mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-07 00:00:17 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
aa0f2c3975 | ||
|
|
4dc8648d81 | ||
|
|
7a703e1176 | ||
|
|
02df1a2924 | ||
|
|
efac7efc97 | ||
|
|
d99b4a80c8 | ||
|
|
2a76744d30 | ||
|
|
843b2d1969 | ||
|
|
b275c96889 | ||
|
|
033b1ce449 | ||
|
|
99a43283be | ||
|
|
55bfd6394b | ||
|
|
14bfe6016e | ||
|
|
0d4ad70875 | ||
|
|
a8bf3859ea | ||
|
|
aa7dcffcea | ||
|
|
be1a5cea8e | ||
|
|
402ea84aa0 | ||
|
|
19a7b06b91 | ||
|
|
96bd643e5b | ||
|
|
6b9293979c | ||
|
|
7d5cee4384 | ||
|
|
c0bab70161 | ||
|
|
32f5a28433 | ||
|
|
ce30179ed1 | ||
|
|
a515b707f3 | ||
|
|
9ab0fa01bd | ||
|
|
c3bffa2929 | ||
|
|
951fee361f | ||
|
|
8c4860b9a7 | ||
|
|
ee221e6a8c | ||
|
|
f5625093bb | ||
|
|
abfd974d70 | ||
|
|
97966930e9 | ||
|
|
a52a2e3ae4 | ||
|
|
c49b30f105 | ||
|
|
b0b4ad2083 | ||
|
|
ee4bee4fef | ||
|
|
c3a0f5a2f6 | ||
|
|
0413a6bf68 | ||
|
|
7bd036d1ae | ||
|
|
112c49a348 | ||
|
|
80878ae611 | ||
|
|
85a69be5b6 | ||
|
|
8dbfd1635a | ||
|
|
8b5ca303e3 | ||
|
|
f90d2aeb6d | ||
|
|
cd4b520f72 | ||
|
|
20d5a26a72 | ||
|
|
9346116485 | ||
|
|
bb8336fcad | ||
|
|
ee96d60983 | ||
|
|
6052b335dc | ||
|
|
ab0a6bbe9f | ||
|
|
9dd6d8ed94 | ||
|
|
3b5d0e3e27 | ||
|
|
37e13cf073 | ||
|
|
9b1b9c26cc | ||
|
|
f7f3024b92 | ||
|
|
11ec71a4ce | ||
|
|
665491adf8 | ||
|
|
55391ccbc0 | ||
|
|
7790d7a0b7 | ||
|
|
f3d8c2cbac | ||
|
|
1226069b4c | ||
|
|
80687c8d2d | ||
|
|
920fe60492 | ||
|
|
8c6ce2cb3b | ||
|
|
61f30d004c | ||
|
|
95919a1ddf | ||
|
|
35ec54f920 | ||
|
|
32e8a56093 | ||
|
|
136f1d0a0b | ||
|
|
0c042d1e85 | ||
|
|
ad13442be4 | ||
|
|
d6b9252760 | ||
|
|
22222ebaf5 | ||
|
|
734258e23b | ||
|
|
74916b3757 | ||
|
|
c5359264a3 | ||
|
|
9d0ff7929e | ||
|
|
d3eed27d17 | ||
|
|
bc1669b163 | ||
|
|
83e417b2c6 | ||
|
|
cb00d9171f | ||
|
|
cf77f2ae5d | ||
|
|
273d086a7b | ||
|
|
94d9cf54bc | ||
|
|
3089e0e6de | ||
|
|
3c088fb414 | ||
|
|
676c9a9be6 | ||
|
|
314fea36b4 | ||
|
|
3b7d30d26a | ||
|
|
a8d32b9a2f | ||
|
|
24cb02f4ff | ||
|
|
725d0e187a | ||
|
|
d8603cb9bd | ||
|
|
04c2cb5feb | ||
|
|
32f2decb24 | ||
|
|
6954ebe3a0 | ||
|
|
2e40da3d6b | ||
|
|
559772bb03 | ||
|
|
2bbcf72e27 | ||
|
|
aa3a92aa60 | ||
|
|
5497240a25 | ||
|
|
79c609d0f5 | ||
|
|
b33788e765 | ||
|
|
8340012466 | ||
|
|
0adcc779cf | ||
|
|
9d0718fbc4 | ||
|
|
53567a6526 | ||
|
|
90fb5f038b | ||
|
|
063b1eb936 | ||
|
|
9f243c8f7b | ||
|
|
2dc600f283 | ||
|
|
a01402d502 | ||
|
|
c90036aeea | ||
|
|
dfb751eea0 | ||
|
|
6e0f5eccb3 | ||
|
|
579fb42458 | ||
|
|
f4b487352c | ||
|
|
4448f84f29 | ||
|
|
9e1e602e09 | ||
|
|
ca70e387ec | ||
|
|
9a483107e3 | ||
|
|
3bac767866 | ||
|
|
7b4e48480b | ||
|
|
101bba4808 | ||
|
|
a31c3c1c15 | ||
|
|
1a18e392f8 | ||
|
|
4c7595c68a | ||
|
|
1a467f0ebd | ||
|
|
06e7360f4c | ||
|
|
ec3b72e17e | ||
|
|
259e1b75a4 | ||
|
|
f9642cba7a | ||
|
|
c01c415030 | ||
|
|
50e56358c3 | ||
|
|
465dbc260f | ||
|
|
9f291f3adb | ||
|
|
f2acc3da4c | ||
|
|
f8f165d96d | ||
|
|
baef95992c | ||
|
|
97e18c8469 | ||
|
|
6e04f7368b | ||
|
|
55a835ebb8 | ||
|
|
85776c2537 | ||
|
|
e3ec25d9db | ||
|
|
ebfcc1e835 | ||
|
|
769b2c2a46 | ||
|
|
3b2100307e | ||
|
|
28cc179214 | ||
|
|
3c0f243a2d | ||
|
|
bbf1563f80 | ||
|
|
ed6b1011f9 | ||
|
|
e3e7f0279c | ||
|
|
a10f984b1c | ||
|
|
663f3d8b5a | ||
|
|
b1f7be2f6c | ||
|
|
b83cbcb33c | ||
|
|
ac1a096bae | ||
|
|
048c8ded88 | ||
|
|
948938bf4b | ||
|
|
bb064c7334 | ||
|
|
2d3d49b900 | ||
|
|
ea7096ed5b | ||
|
|
7a0f6c0a80 | ||
|
|
e4ee35a925 | ||
|
|
d3ab9bdef6 | ||
|
|
926eefc86c | ||
|
|
3eb7a5b998 | ||
|
|
58614ff131 | ||
|
|
bf3a09e5e3 | ||
|
|
83c536c47f | ||
|
|
7b39e57e72 | ||
|
|
5bedf32666 | ||
|
|
1eb7be9870 | ||
|
|
93e4288c57 | ||
|
|
9ca4868833 | ||
|
|
c5f8ea58e9 | ||
|
|
5bee17bee1 | ||
|
|
efe7c54374 | ||
|
|
9e1840e974 | ||
|
|
1f40590f9a | ||
|
|
64a3bc235d | ||
|
|
7d9af246ea | ||
|
|
6d3471bcaa | ||
|
|
a8714dbd49 | ||
|
|
512312fa06 | ||
|
|
010028e381 | ||
|
|
f27f1871e4 | ||
|
|
3a7aa83ab1 | ||
|
|
bcc136c3b9 | ||
|
|
3da31830d1 | ||
|
|
d19b57a52e | ||
|
|
ef6d640a8c | ||
|
|
2cae2f2462 | ||
|
|
1fde5d7fca | ||
|
|
10de2f83ac | ||
|
|
9d86e11a47 | ||
|
|
4d503d3155 | ||
|
|
7e663b91df | ||
|
|
47242dc190 | ||
|
|
e13c8e3295 | ||
|
|
3c3ba62c10 | ||
|
|
34fe56dfb2 | ||
|
|
ecf8cde5e0 | ||
|
|
55284aad7e | ||
|
|
3afc35f7b4 | ||
|
|
1058428a51 | ||
|
|
a2fc51fc7b | ||
|
|
1848629ba5 | ||
|
|
3399577330 | ||
|
|
18bfc8afd0 | ||
|
|
76b023ed3e | ||
|
|
b91b0e9d65 | ||
|
|
74489a4177 | ||
|
|
55d1d6bcd4 | ||
|
|
cd249e2c3a | ||
|
|
61cd835754 | ||
|
|
bd24364c1b | ||
|
|
ab516d7b79 | ||
|
|
d25ed4b0bf | ||
|
|
c521d2b48d | ||
|
|
9ed8165405 | ||
|
|
5099b2b5dc | ||
|
|
d372552593 | ||
|
|
729e32ccc2 | ||
|
|
170204d6f1 | ||
|
|
f7bfecd3f1 | ||
|
|
5f0427c253 | ||
|
|
472860a840 | ||
|
|
eddb7d12cc | ||
|
|
789a9f19c0 | ||
|
|
f70aafb211 | ||
|
|
7519af2819 |
No files matched your search
+1
-1
@@ -7,7 +7,7 @@ AlignConsecutiveAssignments: None
|
||||
AlignConsecutiveBitFields: Consecutive
|
||||
AlignConsecutiveDeclarations: None
|
||||
AlignConsecutiveMacros: None
|
||||
AlignEscapedNewlines: DontAlign
|
||||
AlignEscapedNewlines: Left
|
||||
AlignOperands: Align
|
||||
AlignTrailingComments: true
|
||||
AllowAllParametersOfDeclarationOnNextLine: false
|
||||
|
||||
@@ -8,7 +8,5 @@ FEXCore/Source/Common/SoftFloat-3e/*
|
||||
Source/Common/cpp-optparse/*
|
||||
|
||||
# Files with human-indented tables for readability - don't mess with these
|
||||
FEXCore/Source/Interface/Core/X86Tables/X87Tables.cpp
|
||||
FEXCore/Source/Interface/Core/X86Tables/XOPTables.cpp
|
||||
FEXCore/Source/Interface/Core/X86Tables/*
|
||||
|
||||
@@ -20,6 +20,7 @@ jobs:
|
||||
|
||||
- name: Checkout through merge base
|
||||
uses: rmacklin/fetch-through-merge-base@v0
|
||||
timeout-minutes: 3
|
||||
with:
|
||||
base_ref: ${{ github.event.pull_request.base.ref }}
|
||||
head_ref: ${{ github.event.pull_request.head.sha }}
|
||||
|
||||
+1
-2
@@ -235,8 +235,6 @@ add_definitions(-Wno-trigraphs)
|
||||
add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
|
||||
|
||||
if (BUILD_TESTS)
|
||||
option(CATCH_BUILD_STATIC_LIBRARY "" ON)
|
||||
set(CATCH_BUILD_STATIC_LIBRARY ON)
|
||||
add_subdirectory(External/Catch2/)
|
||||
|
||||
# Pull in catch_discover_tests definition
|
||||
@@ -359,6 +357,7 @@ if (BUILD_TESTS)
|
||||
endif()
|
||||
|
||||
add_subdirectory(FEXHeaderUtils/)
|
||||
add_subdirectory(CodeEmitter/)
|
||||
add_subdirectory(FEXCore/)
|
||||
|
||||
# Binfmt_misc files must be installed prior to Source/ installs
|
||||
|
||||
@@ -0,0 +1,2 @@
|
||||
add_library(CodeEmitter INTERFACE)
|
||||
target_include_directories(CodeEmitter INTERFACE .)
|
||||
File diff suppressed because it is too large.
Load diff
+1302
-1303
File diff suppressed because it is too large.
Load diff
+32
-32
@@ -8,11 +8,11 @@ public:
|
||||
public:
|
||||
// Conditional branch immediate
|
||||
///< Branch conditional
|
||||
void b(FEXCore::ARMEmitter::Condition Cond, uint32_t Imm) {
|
||||
void b(ARMEmitter::Condition Cond, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, Imm);
|
||||
}
|
||||
void b(FEXCore::ARMEmitter::Condition Cond, BackwardLabel const* Label) {
|
||||
void b(ARMEmitter::Condition Cond, BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
@@ -20,13 +20,13 @@ public:
|
||||
}
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void b(FEXCore::ARMEmitter::Condition Cond, LabelType *Label) {
|
||||
void b(ARMEmitter::Condition Cond, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, 0);
|
||||
}
|
||||
|
||||
void b(FEXCore::ARMEmitter::Condition Cond, BiDirectionalLabel *Label) {
|
||||
void b(ARMEmitter::Condition Cond, BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
b(Cond, &Label->Backward);
|
||||
}
|
||||
@@ -36,11 +36,11 @@ public:
|
||||
}
|
||||
|
||||
///< Branch consistent conditional
|
||||
void bc(FEXCore::ARMEmitter::Condition Cond, uint32_t Imm) {
|
||||
void bc(ARMEmitter::Condition Cond, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, Imm);
|
||||
}
|
||||
void bc(FEXCore::ARMEmitter::Condition Cond, BackwardLabel const* Label) {
|
||||
void bc(ARMEmitter::Condition Cond, BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
@@ -49,13 +49,13 @@ public:
|
||||
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void bc(FEXCore::ARMEmitter::Condition Cond, LabelType *Label) {
|
||||
void bc(ARMEmitter::Condition Cond, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, 0);
|
||||
}
|
||||
|
||||
void bc(FEXCore::ARMEmitter::Condition Cond, BiDirectionalLabel *Label) {
|
||||
void bc(ARMEmitter::Condition Cond, BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
bc(Cond, &Label->Backward);
|
||||
}
|
||||
@@ -65,7 +65,7 @@ public:
|
||||
}
|
||||
|
||||
// Unconditional branch register
|
||||
void br(FEXCore::ARMEmitter::Register rn) {
|
||||
void br(ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = 0b1101011 << 25 |
|
||||
0b0'000 << 21 | // opc
|
||||
0b1'1111 << 16 | // op2
|
||||
@@ -74,7 +74,7 @@ public:
|
||||
|
||||
UnconditionalBranch(Op, rn);
|
||||
}
|
||||
void blr(FEXCore::ARMEmitter::Register rn) {
|
||||
void blr(ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = 0b1101011 << 25 |
|
||||
0b0'001 << 21 | // opc
|
||||
0b1'1111 << 16 | // op2
|
||||
@@ -83,7 +83,7 @@ public:
|
||||
|
||||
UnconditionalBranch(Op, rn);
|
||||
}
|
||||
void ret(FEXCore::ARMEmitter::Register rn = FEXCore::ARMEmitter::Reg::r30) {
|
||||
void ret(ARMEmitter::Register rn = ARMEmitter::Reg::r30) {
|
||||
constexpr uint32_t Op = 0b1101011 << 25 |
|
||||
0b0'010 << 21 | // opc
|
||||
0b1'1111 << 16 | // op2
|
||||
@@ -156,13 +156,13 @@ public:
|
||||
}
|
||||
|
||||
// Compare and branch
|
||||
void cbz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, uint32_t Imm) {
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, Imm);
|
||||
}
|
||||
|
||||
void cbz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, BackwardLabel const* Label) {
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
@@ -173,7 +173,7 @@ public:
|
||||
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void cbz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, LabelType *Label) {
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
@@ -181,7 +181,7 @@ public:
|
||||
CompareAndBranch(Op, s, rt, 0);
|
||||
}
|
||||
|
||||
void cbz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, BiDirectionalLabel *Label) {
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
cbz(s, rt, &Label->Backward);
|
||||
}
|
||||
@@ -190,13 +190,13 @@ public:
|
||||
}
|
||||
}
|
||||
|
||||
void cbnz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, uint32_t Imm) {
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, Imm);
|
||||
}
|
||||
|
||||
void cbnz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, BackwardLabel const* Label) {
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
@@ -207,7 +207,7 @@ public:
|
||||
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void cbnz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, LabelType *Label) {
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::BC });
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
@@ -215,7 +215,7 @@ public:
|
||||
CompareAndBranch(Op, s, rt, 0);
|
||||
}
|
||||
|
||||
void cbnz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, BiDirectionalLabel *Label) {
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
cbnz(s, rt, &Label->Backward);
|
||||
}
|
||||
@@ -225,12 +225,12 @@ public:
|
||||
}
|
||||
|
||||
// Test and branch immediate
|
||||
void tbz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm);
|
||||
}
|
||||
void tbz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, BackwardLabel const* Label) {
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
@@ -241,7 +241,7 @@ public:
|
||||
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void tbz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, LabelType *Label) {
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::TEST_BRANCH });
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
@@ -249,7 +249,7 @@ public:
|
||||
TestAndBranch(Op, rt, Bit, 0);
|
||||
}
|
||||
|
||||
void tbz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel *Label) {
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
tbz(rt, Bit, &Label->Backward);
|
||||
}
|
||||
@@ -258,12 +258,12 @@ public:
|
||||
}
|
||||
}
|
||||
|
||||
void tbnz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm);
|
||||
}
|
||||
void tbnz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, BackwardLabel const* Label) {
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, BackwardLabel const* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
@@ -274,14 +274,14 @@ public:
|
||||
|
||||
template<typename LabelType>
|
||||
requires (std::is_same_v<LabelType, ForwardLabel> || std::is_same_v<LabelType, SingleUseForwardLabel>)
|
||||
void tbnz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, LabelType *Label) {
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, LabelType *Label) {
|
||||
AddLocationToLabel(Label, SingleUseForwardLabel{ .Location = GetCursorAddress<uint8_t*>(), .Type = SingleUseForwardLabel::InstType::TEST_BRANCH });
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, 0);
|
||||
}
|
||||
|
||||
void tbnz(FEXCore::ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel *Label) {
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel *Label) {
|
||||
if (Label->Backward.Location) {
|
||||
tbnz(rt, Bit, &Label->Backward);
|
||||
}
|
||||
@@ -292,7 +292,7 @@ public:
|
||||
|
||||
private:
|
||||
// Conditional branch immediate
|
||||
void Branch_Conditional(uint32_t Op, uint32_t Op1, uint32_t Op0, FEXCore::ARMEmitter::Condition Cond, uint32_t Imm) {
|
||||
void Branch_Conditional(uint32_t Op, uint32_t Op1, uint32_t Op0, ARMEmitter::Condition Cond, uint32_t Imm) {
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= Op1 << 24;
|
||||
@@ -304,7 +304,7 @@ private:
|
||||
}
|
||||
|
||||
// Unconditional branch register
|
||||
void UnconditionalBranch(uint32_t Op, FEXCore::ARMEmitter::Register rn) {
|
||||
void UnconditionalBranch(uint32_t Op, ARMEmitter::Register rn) {
|
||||
uint32_t Instr = Op;
|
||||
Instr |= Encode_rn(rn);
|
||||
dc32(Instr);
|
||||
@@ -318,8 +318,8 @@ private:
|
||||
}
|
||||
|
||||
// Compare and branch
|
||||
void CompareAndBranch(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rt, uint32_t Imm) {
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
void CompareAndBranch(uint32_t Op, ARMEmitter::Size s, ARMEmitter::Register rt, uint32_t Imm) {
|
||||
const uint32_t SF = s == ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = Op;
|
||||
|
||||
@@ -330,7 +330,7 @@ private:
|
||||
}
|
||||
|
||||
// Test and branch - immediate
|
||||
void TestAndBranch(uint32_t Op, FEXCore::ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
|
||||
void TestAndBranch(uint32_t Op, ARMEmitter::Register rt, uint32_t Bit, uint32_t Imm) {
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= (Bit >> 5) << 31;
|
||||
+2
-2
@@ -4,7 +4,7 @@
|
||||
#include <cstdint>
|
||||
#include <cstring>
|
||||
|
||||
namespace FEXCore::ARMEmitter {
|
||||
namespace ARMEmitter {
|
||||
class Buffer {
|
||||
public:
|
||||
Buffer() {
|
||||
@@ -103,4 +103,4 @@ protected:
|
||||
uint8_t* CurrentOffset;
|
||||
uint64_t Size;
|
||||
};
|
||||
} // namespace FEXCore::ARMEmitter
|
||||
} // namespace ARMEmitter
|
||||
+14
-15
@@ -1,9 +1,6 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Buffer.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
@@ -11,8 +8,8 @@
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <FEXHeaderUtils/BitUtils.h>
|
||||
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
#include <CodeEmitter/Buffer.h>
|
||||
#include <CodeEmitter/Registers.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
@@ -56,7 +53,7 @@
|
||||
* it easier to select the correct load-store instruction. Mostly because these are a nightmare selecting
|
||||
* the right instruction.
|
||||
*/
|
||||
namespace FEXCore::ARMEmitter {
|
||||
namespace ARMEmitter {
|
||||
/*
|
||||
* This `Size` enum is used for most ALU operations.
|
||||
* These follow the AArch64 encoding style in most cases.
|
||||
@@ -611,7 +608,7 @@ constexpr bool AreVectorsSequential(T first, const Args&... args) {
|
||||
// Choices:
|
||||
// - Size of ops passed as an argument rather than template to let the compiler use csel instead of branching.
|
||||
// - Registers are unsized so they can be passed in a GPR and not need conversion operations
|
||||
class Emitter : public FEXCore::ARMEmitter::Buffer {
|
||||
class Emitter : public ARMEmitter::Buffer {
|
||||
public:
|
||||
Emitter() = default;
|
||||
|
||||
@@ -759,15 +756,17 @@ public:
|
||||
Bind<false>(&Label->Forward);
|
||||
}
|
||||
|
||||
#include <CodeEmitter/VixlUtils.inl>
|
||||
|
||||
public:
|
||||
// TODO: Implement SME when it matters.
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/ALUOps.inl"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/BranchOps.inl"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/LoadstoreOps.inl"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/SystemOps.inl"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/ScalarOps.inl"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/ASIMDOps.inl"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/SVEOps.inl"
|
||||
#include <CodeEmitter/ALUOps.inl>
|
||||
#include <CodeEmitter/BranchOps.inl>
|
||||
#include <CodeEmitter/LoadstoreOps.inl>
|
||||
#include <CodeEmitter/SystemOps.inl>
|
||||
#include <CodeEmitter/ScalarOps.inl>
|
||||
#include <CodeEmitter/ASIMDOps.inl>
|
||||
#include <CodeEmitter/SVEOps.inl>
|
||||
|
||||
private:
|
||||
template<typename T>
|
||||
@@ -829,4 +828,4 @@ private:
|
||||
return FEXCore::ToUnderlying(Reg);
|
||||
}
|
||||
};
|
||||
} // namespace FEXCore::ARMEmitter
|
||||
} // namespace ARMEmitter
|
||||
+477
-477
File diff suppressed because it is too large.
Load diff
+2
-2
@@ -6,7 +6,7 @@
|
||||
#include <compare>
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::ARMEmitter {
|
||||
namespace ARMEmitter {
|
||||
class WRegister;
|
||||
class XRegister;
|
||||
|
||||
@@ -1024,4 +1024,4 @@ enum class OpType : uint32_t {
|
||||
Destructive = 0,
|
||||
Constructive,
|
||||
};
|
||||
} // namespace FEXCore::ARMEmitter
|
||||
} // namespace ARMEmitter
|
||||
+14
-25
@@ -1506,13 +1506,14 @@ public:
|
||||
}
|
||||
|
||||
// SVE broadcast floating-point immediate (unpredicated)
|
||||
void fdup(FEXCore::ARMEmitter::SubRegSize size, FEXCore::ARMEmitter::ZRegister zd, float Value) {
|
||||
LOGMAN_THROW_AA_FMT(size == FEXCore::ARMEmitter::SubRegSize::i16Bit ||
|
||||
size == FEXCore::ARMEmitter::SubRegSize::i32Bit ||
|
||||
size == FEXCore::ARMEmitter::SubRegSize::i64Bit, "Unsupported fmov size");
|
||||
void fdup(ARMEmitter::SubRegSize size, ARMEmitter::ZRegister zd, float Value) {
|
||||
LOGMAN_THROW_AA_FMT(size == ARMEmitter::SubRegSize::i16Bit ||
|
||||
size == ARMEmitter::SubRegSize::i32Bit ||
|
||||
size == ARMEmitter::SubRegSize::i64Bit, "Unsupported fmov size");
|
||||
uint32_t Imm{};
|
||||
if (size == SubRegSize::i16Bit) {
|
||||
Imm = FP16ToImm8(vixl::Float16(Value));
|
||||
LOGMAN_MSG_A_FMT("Unsupported");
|
||||
FEX_UNREACHABLE;
|
||||
} else if (size == SubRegSize::i32Bit) {
|
||||
Imm = FP32ToImm8(Value);
|
||||
} else if (size == SubRegSize::i64Bit) {
|
||||
@@ -1521,7 +1522,7 @@ public:
|
||||
|
||||
SVEBroadcastFloatImmUnpredicated(0b00, 0, Imm, size, zd);
|
||||
}
|
||||
void fmov(FEXCore::ARMEmitter::SubRegSize size, FEXCore::ARMEmitter::ZRegister zd, float Value) {
|
||||
void fmov(ARMEmitter::SubRegSize size, ARMEmitter::ZRegister zd, float Value) {
|
||||
fdup(size, zd, Value);
|
||||
}
|
||||
|
||||
@@ -3513,7 +3514,8 @@ private:
|
||||
size == SubRegSize::i64Bit, "Unsupported fcpy/fmov size");
|
||||
uint32_t imm{};
|
||||
if (size == SubRegSize::i16Bit) {
|
||||
imm = FP16ToImm8(vixl::Float16(value));
|
||||
LOGMAN_MSG_A_FMT("Unsupported");
|
||||
FEX_UNREACHABLE;
|
||||
} else if (size == SubRegSize::i32Bit) {
|
||||
imm = FP32ToImm8(value);
|
||||
} else if (size == SubRegSize::i64Bit) {
|
||||
@@ -3717,7 +3719,7 @@ private:
|
||||
|
||||
// SVE bitwise logical operations (predicated)
|
||||
void SVEBitwiseLogicalPredicated(uint32_t opc, SubRegSize size, PRegister pg, ZRegister zdn, ZRegister zm, ZRegister zd) {
|
||||
LOGMAN_THROW_AA_FMT(size != FEXCore::ARMEmitter::SubRegSize::i128Bit, "Can't use 128-bit size");
|
||||
LOGMAN_THROW_AA_FMT(size != ARMEmitter::SubRegSize::i128Bit, "Can't use 128-bit size");
|
||||
LOGMAN_THROW_A_FMT(zd == zdn, "zd needs to equal zdn");
|
||||
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
|
||||
|
||||
@@ -4743,7 +4745,7 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
void SVEPermuteVector(uint32_t op0, FEXCore::ARMEmitter::ZRegister zd, FEXCore::ARMEmitter::ZRegister zm, uint32_t Imm) {
|
||||
void SVEPermuteVector(uint32_t op0, ARMEmitter::ZRegister zd, ARMEmitter::ZRegister zm, uint32_t Imm) {
|
||||
constexpr uint32_t Op = 0b0000'0101'0010'0000'000 << 13;
|
||||
uint32_t Instr = Op;
|
||||
|
||||
@@ -5228,15 +5230,14 @@ private:
|
||||
|
||||
// Alias that returns the equivalently sized unsigned type for a floating-point type T.
|
||||
template <typename T>
|
||||
requires(std::is_same_v<T, float> || std::is_same_v<T, double> || std::is_same_v<T, vixl::Float16>)
|
||||
using FloatToEquivalentUInt = std::conditional_t<std::is_same_v<T, vixl::Float16>, uint16_t,
|
||||
std::conditional_t<std::is_same_v<T, float>, uint32_t, uint64_t>>;
|
||||
requires(std::is_same_v<T, float> || std::is_same_v<T, double>)
|
||||
using FloatToEquivalentUInt = std::conditional_t<std::is_same_v<T, float>, uint32_t, uint64_t>;
|
||||
|
||||
// Determines if a floating-point value is capable of being converted
|
||||
// into an 8-bit immediate. See pseudocode definition of VFPExpandImm
|
||||
// in ARM A-profile reference manual for a general overview of how this was derived.
|
||||
template <typename T>
|
||||
requires(std::is_same_v<T, float> || std::is_same_v<T, double> || std::is_same_v<T, vixl::Float16>)
|
||||
requires(std::is_same_v<T, float> || std::is_same_v<T, double>)
|
||||
[[nodiscard, maybe_unused]] static bool IsValidFPValueForImm8(T value) {
|
||||
const uint64_t bits = FEXCore::BitCast<FloatToEquivalentUInt<T>>(value);
|
||||
const uint64_t datasize_idx = FEXCore::ilog2(sizeof(T)) - 1;
|
||||
@@ -5277,18 +5278,6 @@ private:
|
||||
return true;
|
||||
}
|
||||
|
||||
static uint32_t FP16ToImm8(vixl::Float16 value) {
|
||||
LOGMAN_THROW_A_FMT(IsValidFPValueForImm8(value),
|
||||
"Value cannot be encoded into an 8-bit immediate");
|
||||
|
||||
const uint32_t bits = vixl::Float16ToRawbits(value);
|
||||
const uint32_t sign = (bits & 0x8000) >> 8;
|
||||
const uint32_t expb2 = (bits & 0x2000) >> 7;
|
||||
const uint32_t b5_to_0 = (bits >> 6) & 0x3F;
|
||||
|
||||
return sign | expb2 | b5_to_0;
|
||||
}
|
||||
|
||||
static uint32_t FP32ToImm8(float value) {
|
||||
LOGMAN_THROW_A_FMT(IsValidFPValueForImm8(value),
|
||||
"Value ({}) cannot be encoded into an 8-bit immediate", value);
|
||||
+9
-9
@@ -33,7 +33,7 @@ public:
|
||||
ASIMDScalarCopy(Op, 1, imm5, 0b0000, rd, rn);
|
||||
}
|
||||
|
||||
void mov(FEXCore::ARMEmitter::ScalarRegSize size, FEXCore::ARMEmitter::VRegister rd, FEXCore::ARMEmitter::VRegister rn, uint32_t Index) {
|
||||
void mov(ARMEmitter::ScalarRegSize size, ARMEmitter::VRegister rd, ARMEmitter::VRegister rn, uint32_t Index) {
|
||||
dup(size, rd, rn, Index);
|
||||
}
|
||||
|
||||
@@ -1052,21 +1052,21 @@ public:
|
||||
}
|
||||
|
||||
// Floating-point immediate
|
||||
void fmov(FEXCore::ARMEmitter::ScalarRegSize size, FEXCore::ARMEmitter::VRegister rd, float Value) {
|
||||
void fmov(ARMEmitter::ScalarRegSize size, ARMEmitter::VRegister rd, float Value) {
|
||||
uint32_t M = 0;
|
||||
uint32_t S = 0;
|
||||
uint32_t ptype;
|
||||
uint32_t imm8;
|
||||
uint32_t imm5 = 0b0'0000;
|
||||
if (size == FEXCore::ARMEmitter::ScalarRegSize::i16Bit) {
|
||||
ptype = 0b11;
|
||||
imm8 = FP16ToImm8(vixl::Float16(Value));
|
||||
if (size == ARMEmitter::ScalarRegSize::i16Bit) {
|
||||
LOGMAN_MSG_A_FMT("Unsupported");
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
else if (size == FEXCore::ARMEmitter::ScalarRegSize::i32Bit) {
|
||||
else if (size == ARMEmitter::ScalarRegSize::i32Bit) {
|
||||
ptype = 0b00;
|
||||
imm8 = FP32ToImm8(Value);
|
||||
}
|
||||
else if (size == FEXCore::ARMEmitter::ScalarRegSize::i64Bit) {
|
||||
else if (size == ARMEmitter::ScalarRegSize::i64Bit) {
|
||||
ptype = 0b01;
|
||||
imm8 = FP64ToImm8(Value);
|
||||
}
|
||||
@@ -1077,7 +1077,7 @@ public:
|
||||
FloatScalarImmediate(M, S, ptype, imm8, imm5, rd);
|
||||
}
|
||||
|
||||
void FloatScalarImmediate(uint32_t M, uint32_t S, uint32_t ptype, uint32_t imm8, uint32_t imm5, FEXCore::ARMEmitter::VRegister rd) {
|
||||
void FloatScalarImmediate(uint32_t M, uint32_t S, uint32_t ptype, uint32_t imm8, uint32_t imm5, ARMEmitter::VRegister rd) {
|
||||
constexpr uint32_t Op = 0b0001'1110'0010'0000'0001'00 << 10;
|
||||
uint32_t Instr = Op;
|
||||
|
||||
@@ -1286,7 +1286,7 @@ public:
|
||||
|
||||
private:
|
||||
// Advanced SIMD scalar copy
|
||||
void ASIMDScalarCopy(uint32_t Op, uint32_t Q, uint32_t imm5, uint32_t imm4, FEXCore::ARMEmitter::VRegister rd, FEXCore::ARMEmitter::VRegister rn) {
|
||||
void ASIMDScalarCopy(uint32_t Op, uint32_t Q, uint32_t imm5, uint32_t imm4, ARMEmitter::VRegister rd, ARMEmitter::VRegister rn) {
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= Q << 30;
|
||||
+26
-26
@@ -11,7 +11,7 @@ public:
|
||||
// TODO: AT
|
||||
// TODO: CFP
|
||||
// TODO: CPP
|
||||
void dc(FEXCore::ARMEmitter::DataCacheOperation DCOp, FEXCore::ARMEmitter::Register rt) {
|
||||
void dc(ARMEmitter::DataCacheOperation DCOp, ARMEmitter::Register rt) {
|
||||
constexpr uint32_t Op = 0b1101'0101'0000'1000'0111 << 12;
|
||||
SystemInstruction(Op, 0, FEXCore::ToUnderlying(DCOp), rt);
|
||||
}
|
||||
@@ -48,67 +48,67 @@ public:
|
||||
ExceptionGeneration(0b101, 0b000, 0b11, Imm);
|
||||
}
|
||||
// System instructions with register argument
|
||||
void wfet(FEXCore::ARMEmitter::Register rt) {
|
||||
void wfet(ARMEmitter::Register rt) {
|
||||
SystemInstructionWithReg(0b0000, 0b000, rt);
|
||||
}
|
||||
void wfit(FEXCore::ARMEmitter::Register rt) {
|
||||
void wfit(ARMEmitter::Register rt) {
|
||||
SystemInstructionWithReg(0b0000, 0b001, rt);
|
||||
}
|
||||
|
||||
// Hints
|
||||
void nop() {
|
||||
Hint(FEXCore::ARMEmitter::HintRegister::NOP);
|
||||
Hint(ARMEmitter::HintRegister::NOP);
|
||||
}
|
||||
void yield() {
|
||||
Hint(FEXCore::ARMEmitter::HintRegister::YIELD);
|
||||
Hint(ARMEmitter::HintRegister::YIELD);
|
||||
}
|
||||
void wfe() {
|
||||
Hint(FEXCore::ARMEmitter::HintRegister::WFE);
|
||||
Hint(ARMEmitter::HintRegister::WFE);
|
||||
}
|
||||
void wfi() {
|
||||
Hint(FEXCore::ARMEmitter::HintRegister::WFI);
|
||||
Hint(ARMEmitter::HintRegister::WFI);
|
||||
}
|
||||
void sev() {
|
||||
Hint(FEXCore::ARMEmitter::HintRegister::SEV);
|
||||
Hint(ARMEmitter::HintRegister::SEV);
|
||||
}
|
||||
void sevl() {
|
||||
Hint(FEXCore::ARMEmitter::HintRegister::SEVL);
|
||||
Hint(ARMEmitter::HintRegister::SEVL);
|
||||
}
|
||||
void dgh() {
|
||||
Hint(FEXCore::ARMEmitter::HintRegister::DGH);
|
||||
Hint(ARMEmitter::HintRegister::DGH);
|
||||
}
|
||||
void csdb() {
|
||||
Hint(FEXCore::ARMEmitter::HintRegister::CSDB);
|
||||
Hint(ARMEmitter::HintRegister::CSDB);
|
||||
}
|
||||
|
||||
// Barriers
|
||||
void clrex(uint32_t imm = 15) {
|
||||
LOGMAN_THROW_AA_FMT(imm < 16, "Immediate out of range");
|
||||
Barrier(FEXCore::ARMEmitter::BarrierRegister::CLREX, imm);
|
||||
Barrier(ARMEmitter::BarrierRegister::CLREX, imm);
|
||||
}
|
||||
void dsb(FEXCore::ARMEmitter::BarrierScope Scope) {
|
||||
Barrier(FEXCore::ARMEmitter::BarrierRegister::DSB, FEXCore::ToUnderlying(Scope));
|
||||
void dsb(ARMEmitter::BarrierScope Scope) {
|
||||
Barrier(ARMEmitter::BarrierRegister::DSB, FEXCore::ToUnderlying(Scope));
|
||||
}
|
||||
void dmb(FEXCore::ARMEmitter::BarrierScope Scope) {
|
||||
Barrier(FEXCore::ARMEmitter::BarrierRegister::DMB, FEXCore::ToUnderlying(Scope));
|
||||
void dmb(ARMEmitter::BarrierScope Scope) {
|
||||
Barrier(ARMEmitter::BarrierRegister::DMB, FEXCore::ToUnderlying(Scope));
|
||||
}
|
||||
void isb() {
|
||||
Barrier(FEXCore::ARMEmitter::BarrierRegister::ISB, FEXCore::ToUnderlying(FEXCore::ARMEmitter::BarrierScope::SY));
|
||||
Barrier(ARMEmitter::BarrierRegister::ISB, FEXCore::ToUnderlying(ARMEmitter::BarrierScope::SY));
|
||||
}
|
||||
void sb() {
|
||||
Barrier(FEXCore::ARMEmitter::BarrierRegister::SB, 0);
|
||||
Barrier(ARMEmitter::BarrierRegister::SB, 0);
|
||||
}
|
||||
void tcommit() {
|
||||
Barrier(FEXCore::ARMEmitter::BarrierRegister::TCOMMIT, 0);
|
||||
Barrier(ARMEmitter::BarrierRegister::TCOMMIT, 0);
|
||||
}
|
||||
|
||||
// System register move
|
||||
void msr(FEXCore::ARMEmitter::SystemRegister reg, FEXCore::ARMEmitter::Register rt) {
|
||||
void msr(ARMEmitter::SystemRegister reg, ARMEmitter::Register rt) {
|
||||
constexpr uint32_t Op = 0b1101'0101'0001 << 20;
|
||||
SystemRegisterMove(Op, rt, reg);
|
||||
}
|
||||
|
||||
void mrs(FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::SystemRegister reg) {
|
||||
void mrs(ARMEmitter::Register rd, ARMEmitter::SystemRegister reg) {
|
||||
constexpr uint32_t Op = 0b1101'0101'0011 << 20;
|
||||
SystemRegisterMove(Op, rd, reg);
|
||||
}
|
||||
@@ -130,7 +130,7 @@ private:
|
||||
}
|
||||
|
||||
// System instructions with register argument
|
||||
void SystemInstructionWithReg(uint32_t CRm, uint32_t op2, FEXCore::ARMEmitter::Register rt) {
|
||||
void SystemInstructionWithReg(uint32_t CRm, uint32_t op2, ARMEmitter::Register rt) {
|
||||
uint32_t Instr = 0b1101'0101'0000'0011'0001 << 12;
|
||||
|
||||
Instr |= CRm << 8;
|
||||
@@ -140,13 +140,13 @@ private:
|
||||
}
|
||||
|
||||
// Hints
|
||||
void Hint(FEXCore::ARMEmitter::HintRegister Reg) {
|
||||
void Hint(ARMEmitter::HintRegister Reg) {
|
||||
uint32_t Instr = 0b1101'0101'0000'0011'0010'0000'0001'1111U;
|
||||
Instr |= FEXCore::ToUnderlying(Reg);
|
||||
dc32(Instr);
|
||||
}
|
||||
// Barriers
|
||||
void Barrier(FEXCore::ARMEmitter::BarrierRegister Reg, uint32_t CRm) {
|
||||
void Barrier(ARMEmitter::BarrierRegister Reg, uint32_t CRm) {
|
||||
uint32_t Instr = 0b1101'0101'0000'0011'0011'0000'0001'1111U;
|
||||
Instr |= CRm << 8;
|
||||
Instr |= FEXCore::ToUnderlying(Reg);
|
||||
@@ -154,7 +154,7 @@ private:
|
||||
}
|
||||
|
||||
// System Instruction
|
||||
void SystemInstruction(uint32_t Op, uint32_t L, uint32_t SubOp, FEXCore::ARMEmitter::Register rt) {
|
||||
void SystemInstruction(uint32_t Op, uint32_t L, uint32_t SubOp, ARMEmitter::Register rt) {
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= L << 21;
|
||||
@@ -165,7 +165,7 @@ private:
|
||||
}
|
||||
|
||||
// System register move
|
||||
void SystemRegisterMove(uint32_t Op, FEXCore::ARMEmitter::Register rt, FEXCore::ARMEmitter::SystemRegister reg) {
|
||||
void SystemRegisterMove(uint32_t Op, ARMEmitter::Register rt, ARMEmitter::SystemRegister reg) {
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= FEXCore::ToUnderlying(reg);
|
||||
@@ -0,0 +1,311 @@
|
||||
// Collection of utilities from vixl.
|
||||
// Following is the vixl license.
|
||||
// Copyright 2015, VIXL authors
|
||||
// All rights reserved.
|
||||
//
|
||||
// Redistribution and use in source and binary forms, with or without
|
||||
// modification, are permitted provided that the following conditions are met:
|
||||
//
|
||||
// * Redistributions of source code must retain the above copyright notice,
|
||||
// this list of conditions and the following disclaimer.
|
||||
// * Redistributions in binary form must reproduce the above copyright notice,
|
||||
// this list of conditions and the following disclaimer in the documentation
|
||||
// and/or other materials provided with the distribution.
|
||||
// * Neither the name of ARM Limited nor the names of its contributors may be
|
||||
// used to endorse or promote products derived from this software without
|
||||
// specific prior written permission.
|
||||
//
|
||||
// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS CONTRIBUTORS "AS IS" AND
|
||||
// ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
|
||||
// WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
||||
// DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE
|
||||
// FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
// DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
// SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
// CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
// OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
// OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
|
||||
|
||||
// Test if a given value can be encoded in the immediate field of a logical
|
||||
// instruction.
|
||||
// If it can be encoded, the function returns true, and values pointed to by n,
|
||||
// imm_s and imm_r are updated with immediates encoded in the format required
|
||||
// by the corresponding fields in the logical instruction.
|
||||
// If it can not be encoded, the function returns false, and the values pointed
|
||||
// to by n, imm_s and imm_r are undefined.
|
||||
static bool IsImmLogical(uint64_t value,
|
||||
unsigned width,
|
||||
unsigned* n,
|
||||
unsigned* imm_s,
|
||||
unsigned* imm_r) {
|
||||
[[maybe_unused]] constexpr auto kBRegSize = 8;
|
||||
[[maybe_unused]] constexpr auto kHRegSize = 16;
|
||||
[[maybe_unused]] constexpr auto kSRegSize = 32;
|
||||
[[maybe_unused]] constexpr auto kDRegSize = 64;
|
||||
|
||||
constexpr auto kWRegSize = 32;
|
||||
constexpr auto kXRegSize = 64;
|
||||
|
||||
LOGMAN_THROW_A_FMT((width == kBRegSize) || (width == kHRegSize) ||
|
||||
(width == kSRegSize) || (width == kDRegSize), "Unexpected imm size");
|
||||
|
||||
bool negate = false;
|
||||
|
||||
// Logical immediates are encoded using parameters n, imm_s and imm_r using
|
||||
// the following table:
|
||||
//
|
||||
// N imms immr size S R
|
||||
// 1 ssssss rrrrrr 64 UInt(ssssss) UInt(rrrrrr)
|
||||
// 0 0sssss xrrrrr 32 UInt(sssss) UInt(rrrrr)
|
||||
// 0 10ssss xxrrrr 16 UInt(ssss) UInt(rrrr)
|
||||
// 0 110sss xxxrrr 8 UInt(sss) UInt(rrr)
|
||||
// 0 1110ss xxxxrr 4 UInt(ss) UInt(rr)
|
||||
// 0 11110s xxxxxr 2 UInt(s) UInt(r)
|
||||
// (s bits must not be all set)
|
||||
//
|
||||
// A pattern is constructed of size bits, where the least significant S+1 bits
|
||||
// are set. The pattern is rotated right by R, and repeated across a 32 or
|
||||
// 64-bit value, depending on destination register width.
|
||||
//
|
||||
// Put another way: the basic format of a logical immediate is a single
|
||||
// contiguous stretch of 1 bits, repeated across the whole word at intervals
|
||||
// given by a power of 2. To identify them quickly, we first locate the
|
||||
// lowest stretch of 1 bits, then the next 1 bit above that; that combination
|
||||
// is different for every logical immediate, so it gives us all the
|
||||
// information we need to identify the only logical immediate that our input
|
||||
// could be, and then we simply check if that's the value we actually have.
|
||||
//
|
||||
// (The rotation parameter does give the possibility of the stretch of 1 bits
|
||||
// going 'round the end' of the word. To deal with that, we observe that in
|
||||
// any situation where that happens the bitwise NOT of the value is also a
|
||||
// valid logical immediate. So we simply invert the input whenever its low bit
|
||||
// is set, and then we know that the rotated case can't arise.)
|
||||
|
||||
if (value & 1) {
|
||||
// If the low bit is 1, negate the value, and set a flag to remember that we
|
||||
// did (so that we can adjust the return values appropriately).
|
||||
negate = true;
|
||||
value = ~value;
|
||||
}
|
||||
|
||||
if (width <= kWRegSize) {
|
||||
// To handle 8/16/32-bit logical immediates, the very easiest thing is to repeat
|
||||
// the input value to fill a 64-bit word. The correct encoding of that as a
|
||||
// logical immediate will also be the correct encoding of the value.
|
||||
|
||||
// Avoid making the assumption that the most-significant 56/48/32 bits are zero by
|
||||
// shifting the value left and duplicating it.
|
||||
for (unsigned bits = width; bits <= kWRegSize; bits *= 2) {
|
||||
value <<= bits;
|
||||
uint64_t mask = (UINT64_C(1) << bits) - 1;
|
||||
value |= ((value >> bits) & mask);
|
||||
}
|
||||
}
|
||||
|
||||
// The basic analysis idea: imagine our input word looks like this.
|
||||
//
|
||||
// 0011111000111110001111100011111000111110001111100011111000111110
|
||||
// c b a
|
||||
// |<--d-->|
|
||||
//
|
||||
// We find the lowest set bit (as an actual power-of-2 value, not its index)
|
||||
// and call it a. Then we add a to our original number, which wipes out the
|
||||
// bottommost stretch of set bits and replaces it with a 1 carried into the
|
||||
// next zero bit. Then we look for the new lowest set bit, which is in
|
||||
// position b, and subtract it, so now our number is just like the original
|
||||
// but with the lowest stretch of set bits completely gone. Now we find the
|
||||
// lowest set bit again, which is position c in the diagram above. Then we'll
|
||||
// measure the distance d between bit positions a and c (using CLZ), and that
|
||||
// tells us that the only valid logical immediate that could possibly be equal
|
||||
// to this number is the one in which a stretch of bits running from a to just
|
||||
// below b is replicated every d bits.
|
||||
uint64_t a = LowestSetBit(value);
|
||||
uint64_t value_plus_a = value + a;
|
||||
uint64_t b = LowestSetBit(value_plus_a);
|
||||
uint64_t value_plus_a_minus_b = value_plus_a - b;
|
||||
uint64_t c = LowestSetBit(value_plus_a_minus_b);
|
||||
|
||||
int d, clz_a, out_n;
|
||||
uint64_t mask;
|
||||
|
||||
if (c != 0) {
|
||||
// The general case, in which there is more than one stretch of set bits.
|
||||
// Compute the repeat distance d, and set up a bitmask covering the basic
|
||||
// unit of repetition (i.e. a word with the bottom d bits set). Also, in all
|
||||
// of these cases the N bit of the output will be zero.
|
||||
clz_a = CountLeadingZeros(a, kXRegSize);
|
||||
int clz_c = CountLeadingZeros(c, kXRegSize);
|
||||
d = clz_a - clz_c;
|
||||
mask = ((UINT64_C(1) << d) - 1);
|
||||
out_n = 0;
|
||||
} else {
|
||||
// Handle degenerate cases.
|
||||
//
|
||||
// If any of those 'find lowest set bit' operations didn't find a set bit at
|
||||
// all, then the word will have been zero thereafter, so in particular the
|
||||
// last lowest_set_bit operation will have returned zero. So we can test for
|
||||
// all the special case conditions in one go by seeing if c is zero.
|
||||
if (a == 0) {
|
||||
// The input was zero (or all 1 bits, which will come to here too after we
|
||||
// inverted it at the start of the function), for which we just return
|
||||
// false.
|
||||
return false;
|
||||
} else {
|
||||
// Otherwise, if c was zero but a was not, then there's just one stretch
|
||||
// of set bits in our word, meaning that we have the trivial case of
|
||||
// d == 64 and only one 'repetition'. Set up all the same variables as in
|
||||
// the general case above, and set the N bit in the output.
|
||||
clz_a = CountLeadingZeros(a, kXRegSize);
|
||||
d = 64;
|
||||
mask = ~UINT64_C(0);
|
||||
out_n = 1;
|
||||
}
|
||||
}
|
||||
|
||||
// If the repeat period d is not a power of two, it can't be encoded.
|
||||
if (!IsPowerOf2(d)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if (((b - a) & ~mask) != 0) {
|
||||
// If the bit stretch (b - a) does not fit within the mask derived from the
|
||||
// repeat period, then fail.
|
||||
return false;
|
||||
}
|
||||
|
||||
// The only possible option is b - a repeated every d bits. Now we're going to
|
||||
// actually construct the valid logical immediate derived from that
|
||||
// specification, and see if it equals our original input.
|
||||
//
|
||||
// To repeat a value every d bits, we multiply it by a number of the form
|
||||
// (1 + 2^d + 2^(2d) + ...), i.e. 0x0001000100010001 or similar. These can
|
||||
// be derived using a table lookup on CLZ(d).
|
||||
static const uint64_t multipliers[] = {
|
||||
0x0000000000000001UL,
|
||||
0x0000000100000001UL,
|
||||
0x0001000100010001UL,
|
||||
0x0101010101010101UL,
|
||||
0x1111111111111111UL,
|
||||
0x5555555555555555UL,
|
||||
};
|
||||
uint64_t multiplier = multipliers[CountLeadingZeros(d, kXRegSize) - 57];
|
||||
uint64_t candidate = (b - a) * multiplier;
|
||||
|
||||
if (value != candidate) {
|
||||
// The candidate pattern doesn't match our input value, so fail.
|
||||
return false;
|
||||
}
|
||||
|
||||
// We have a match! This is a valid logical immediate, so now we have to
|
||||
// construct the bits and pieces of the instruction encoding that generates
|
||||
// it.
|
||||
|
||||
// Count the set bits in our basic stretch. The special case of clz(0) == -1
|
||||
// makes the answer come out right for stretches that reach the very top of
|
||||
// the word (e.g. numbers like 0xffffc00000000000).
|
||||
int clz_b = (b == 0) ? -1 : CountLeadingZeros(b, kXRegSize);
|
||||
int s = clz_a - clz_b;
|
||||
|
||||
// Decide how many bits to rotate right by, to put the low bit of that basic
|
||||
// stretch in position a.
|
||||
int r;
|
||||
if (negate) {
|
||||
// If we inverted the input right at the start of this function, here's
|
||||
// where we compensate: the number of set bits becomes the number of clear
|
||||
// bits, and the rotation count is based on position b rather than position
|
||||
// a (since b is the location of the 'lowest' 1 bit after inversion).
|
||||
s = d - s;
|
||||
r = (clz_b + 1) & (d - 1);
|
||||
} else {
|
||||
r = (clz_a + 1) & (d - 1);
|
||||
}
|
||||
|
||||
// Now we're done, except for having to encode the S output in such a way that
|
||||
// it gives both the number of set bits and the length of the repeated
|
||||
// segment. The s field is encoded like this:
|
||||
//
|
||||
// imms size S
|
||||
// ssssss 64 UInt(ssssss)
|
||||
// 0sssss 32 UInt(sssss)
|
||||
// 10ssss 16 UInt(ssss)
|
||||
// 110sss 8 UInt(sss)
|
||||
// 1110ss 4 UInt(ss)
|
||||
// 11110s 2 UInt(s)
|
||||
//
|
||||
// So we 'or' (2 * -d) with our computed s to form imms.
|
||||
if ((n != NULL) || (imm_s != NULL) || (imm_r != NULL)) {
|
||||
*n = out_n;
|
||||
*imm_s = ((2 * -d) | (s - 1)) & 0x3f;
|
||||
*imm_r = r;
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
private:
|
||||
|
||||
template <typename V>
|
||||
static inline bool IsPowerOf2(V value) {
|
||||
return (value != 0) && ((value & (value - 1)) == 0);
|
||||
}
|
||||
|
||||
// Some compilers dislike negating unsigned integers,
|
||||
// so we provide an equivalent.
|
||||
template <typename T>
|
||||
static inline T UnsignedNegate(T value) {
|
||||
static_assert(std::is_unsigned<T>::value);
|
||||
return ~value + 1;
|
||||
}
|
||||
|
||||
static inline uint64_t LowestSetBit(uint64_t value) {
|
||||
return value & UnsignedNegate(value);
|
||||
}
|
||||
|
||||
template <typename V>
|
||||
static inline int CountLeadingZeros(V value, int width = (sizeof(V) * 8)) {
|
||||
#if COMPILER_HAS_BUILTIN_CLZ
|
||||
if (width == 32) {
|
||||
return (value == 0) ? 32 : __builtin_clz(static_cast<unsigned>(value));
|
||||
} else if (width == 64) {
|
||||
return (value == 0) ? 64 : __builtin_clzll(value);
|
||||
}
|
||||
#endif
|
||||
return CountLeadingZerosFallBack(value, width);
|
||||
}
|
||||
|
||||
static inline int CountLeadingZerosFallBack(uint64_t value, int width) {
|
||||
LOGMAN_THROW_A_FMT(IsPowerOf2(width) && (width <= 64), "Invalid width");
|
||||
if (value == 0) {
|
||||
return width;
|
||||
}
|
||||
int count = 0;
|
||||
value = value << (64 - width);
|
||||
if ((value & UINT64_C(0xffffffff00000000)) == 0) {
|
||||
count += 32;
|
||||
value = value << 32;
|
||||
}
|
||||
if ((value & UINT64_C(0xffff000000000000)) == 0) {
|
||||
count += 16;
|
||||
value = value << 16;
|
||||
}
|
||||
if ((value & UINT64_C(0xff00000000000000)) == 0) {
|
||||
count += 8;
|
||||
value = value << 8;
|
||||
}
|
||||
if ((value & UINT64_C(0xf000000000000000)) == 0) {
|
||||
count += 4;
|
||||
value = value << 4;
|
||||
}
|
||||
if ((value & UINT64_C(0xc000000000000000)) == 0) {
|
||||
count += 2;
|
||||
value = value << 2;
|
||||
}
|
||||
if ((value & UINT64_C(0x8000000000000000)) == 0) {
|
||||
count += 1;
|
||||
}
|
||||
count += (value == 0);
|
||||
return count;
|
||||
}
|
||||
|
||||
public:
|
||||
+11
-1
@@ -173,6 +173,12 @@ class ClangFormatHelper(FormatHelper):
|
||||
name = "clang-format"
|
||||
friendly_name = "C/C++ code formatter"
|
||||
|
||||
@property
|
||||
def cformat_wrapper_path(self) -> str:
|
||||
relpath = "../../Scripts/clang-format.py"
|
||||
curpath = os.path.dirname(os.path.abspath(__file__))
|
||||
return os.path.abspath(os.path.normpath(os.path.join(curpath, relpath)))
|
||||
|
||||
@property
|
||||
def instructions(self) -> str:
|
||||
return " ".join(self.cf_cmd)
|
||||
@@ -210,7 +216,11 @@ class ClangFormatHelper(FormatHelper):
|
||||
if not cpp_files:
|
||||
return None
|
||||
|
||||
cf_cmd = [self.clang_fmt_path, "--diff"]
|
||||
cf_cmd = [
|
||||
self.clang_fmt_path,
|
||||
f"--binary={self.cformat_wrapper_path}",
|
||||
"--diff",
|
||||
]
|
||||
|
||||
if args.start_rev and args.end_rev:
|
||||
cf_cmd.append(args.start_rev)
|
||||
|
||||
@@ -55,6 +55,7 @@ class OpDefinition:
|
||||
DynamicDispatch: bool
|
||||
JITDispatch: bool
|
||||
JITDispatchOverride: str
|
||||
TiedSource: int
|
||||
Arguments: list
|
||||
EmitValidation: list
|
||||
Desc: list
|
||||
@@ -77,6 +78,7 @@ class OpDefinition:
|
||||
self.DynamicDispatch = False
|
||||
self.JITDispatch = True
|
||||
self.JITDispatchOverride = None
|
||||
self.TiedSource = -1
|
||||
self.Arguments = []
|
||||
self.EmitValidation = []
|
||||
self.Desc = []
|
||||
@@ -248,6 +250,9 @@ def parse_ops(ops):
|
||||
if "JITDispatchOverride" in op_val:
|
||||
OpDef.JITDispatchOverride = op_val["JITDispatchOverride"]
|
||||
|
||||
if "TiedSource" in op_val:
|
||||
OpDef.TiedSource = op_val["TiedSource"]
|
||||
|
||||
# Do some fixups of the data here
|
||||
if len(OpDef.EmitValidation) != 0:
|
||||
for i in range(len(OpDef.EmitValidation)):
|
||||
@@ -372,13 +377,30 @@ def print_ir_sizes():
|
||||
|
||||
output_file.write("[[maybe_unused, nodiscard]] static size_t GetSize(IROps Op) { return IRSizes[Op]; }\n\n")
|
||||
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] std::string_view const& GetName(IROps Op);\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] uint8_t GetArgs(IROps Op);\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] uint8_t GetRAArgs(IROps Op);\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] FEXCore::IR::RegisterClassType GetRegClass(IROps Op);\n\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool HasSideEffects(IROps Op);\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool ImplicitFlagClobber(IROps Op);\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool GetHasDest(IROps Op);\n")
|
||||
output_file.write(
|
||||
'[[nodiscard, gnu::const, gnu::visibility("default")]] std::string_view const& GetName(IROps Op);\n'
|
||||
)
|
||||
output_file.write(
|
||||
'[[nodiscard, gnu::const, gnu::visibility("default")]] uint8_t GetArgs(IROps Op);\n'
|
||||
)
|
||||
output_file.write(
|
||||
'[[nodiscard, gnu::const, gnu::visibility("default")]] uint8_t GetRAArgs(IROps Op);\n'
|
||||
)
|
||||
output_file.write(
|
||||
'[[nodiscard, gnu::const, gnu::visibility("default")]] FEXCore::IR::RegisterClassType GetRegClass(IROps Op);\n\n'
|
||||
)
|
||||
output_file.write(
|
||||
'[[nodiscard, gnu::const, gnu::visibility("default")]] bool HasSideEffects(IROps Op);\n'
|
||||
)
|
||||
output_file.write(
|
||||
'[[nodiscard, gnu::const, gnu::visibility("default")]] bool ImplicitFlagClobber(IROps Op);\n'
|
||||
)
|
||||
output_file.write(
|
||||
'[[nodiscard, gnu::const, gnu::visibility("default")]] bool GetHasDest(IROps Op);\n'
|
||||
)
|
||||
output_file.write(
|
||||
'[[nodiscard, gnu::const, gnu::visibility("default")]] int8_t TiedSource(IROps Op);\n'
|
||||
)
|
||||
|
||||
output_file.write("#undef IROP_SIZES\n")
|
||||
output_file.write("#endif\n\n")
|
||||
@@ -471,15 +493,25 @@ def print_ir_getraargs():
|
||||
def print_ir_hassideeffects():
|
||||
output_file.write("#ifdef IROP_HASSIDEEFFECTS_IMPL\n")
|
||||
|
||||
for array, prop in [("SideEffects", "HasSideEffects"),
|
||||
("ImplicitFlagClobbers", "ImplicitFlagClobber")]:
|
||||
output_file.write(f"constexpr std::array<uint8_t, OP_LAST + 1> {array} = {{\n")
|
||||
for array, prop, T in [
|
||||
("SideEffects", "HasSideEffects", "bool"),
|
||||
("ImplicitFlagClobbers", "ImplicitFlagClobber", "bool"),
|
||||
("TiedSources", "TiedSource", "int8_t"),
|
||||
]:
|
||||
output_file.write(
|
||||
f"constexpr std::array<{'uint8_t' if T == 'bool' else T}, OP_LAST + 1> {array} = {{\n"
|
||||
)
|
||||
for op in IROps:
|
||||
output_file.write("\t{},\n".format(("true" if getattr(op, prop) else "false")))
|
||||
if T == "bool":
|
||||
output_file.write(
|
||||
"\t{},\n".format(("true" if getattr(op, prop) else "false"))
|
||||
)
|
||||
else:
|
||||
output_file.write(f"\t{getattr(op, prop)},\n")
|
||||
|
||||
output_file.write("};\n\n")
|
||||
|
||||
output_file.write(f"bool {prop}(IROps Op) {{\n")
|
||||
output_file.write(f"{T} {prop}(IROps Op) {{\n")
|
||||
output_file.write(f" return {array}[Op];\n")
|
||||
output_file.write("}\n")
|
||||
|
||||
|
||||
@@ -134,22 +134,16 @@ set (SRCS
|
||||
Interface/GDBJIT/GDBJIT.cpp
|
||||
Interface/IR/AOTIR.cpp
|
||||
Interface/IR/IRDumper.cpp
|
||||
Interface/IR/IRParser.cpp
|
||||
Interface/IR/IREmitter.cpp
|
||||
Interface/IR/PassManager.cpp
|
||||
Interface/IR/Passes/ConstProp.cpp
|
||||
Interface/IR/Passes/DeadCodeElimination.cpp
|
||||
Interface/IR/Passes/DeadContextStoreElimination.cpp
|
||||
Interface/IR/Passes/IRCompaction.cpp
|
||||
Interface/IR/Passes/IRDumperPass.cpp
|
||||
Interface/IR/Passes/IRValidation.cpp
|
||||
Interface/IR/Passes/RAValidation.cpp
|
||||
Interface/IR/Passes/LongDivideRemovalPass.cpp
|
||||
Interface/IR/Passes/ValueDominanceValidation.cpp
|
||||
Interface/IR/Passes/RedundantFlagCalculationElimination.cpp
|
||||
Interface/IR/Passes/DeadStoreElimination.cpp
|
||||
Interface/IR/Passes/RegisterAllocationPass.cpp
|
||||
Interface/IR/Passes/InlineCallOptimization.cpp
|
||||
Utils/Telemetry.cpp
|
||||
Utils/Threads.cpp
|
||||
Utils/Profiler.cpp
|
||||
@@ -194,7 +188,7 @@ endif()
|
||||
# Some defines for the softfloat library
|
||||
list(APPEND DEFINES "-DSOFTFLOAT_BUILTIN_CLZ")
|
||||
|
||||
set (LIBS fmt::fmt vixl xxHash::xxhash FEXHeaderUtils)
|
||||
set (LIBS fmt::fmt vixl xxHash::xxhash FEXHeaderUtils CodeEmitter)
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
list (APPEND LIBS dl)
|
||||
|
||||
@@ -20,12 +20,12 @@ struct BitSet final {
|
||||
|
||||
ElementType* Memory;
|
||||
void Allocate(size_t Elements) {
|
||||
size_t AllocateSize = AlignUp(Elements, MinimumSizeBits) / MinimumSize;
|
||||
size_t AllocateSize = ToBytes(Elements);
|
||||
LOGMAN_THROW_AA_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
Memory = static_cast<ElementType*>(FEXCore::Allocator::malloc(AllocateSize));
|
||||
}
|
||||
void Realloc(size_t Elements) {
|
||||
size_t AllocateSize = AlignUp(Elements, MinimumSizeBits) / MinimumSize;
|
||||
size_t AllocateSize = ToBytes(Elements);
|
||||
LOGMAN_THROW_AA_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
Memory = static_cast<ElementType*>(FEXCore::Allocator::realloc(Memory, AllocateSize));
|
||||
}
|
||||
@@ -43,10 +43,13 @@ struct BitSet final {
|
||||
Memory[Element / MinimumSizeBits] &= (1ULL << (Element % MinimumSizeBits));
|
||||
}
|
||||
void MemClear(size_t Elements) {
|
||||
memset(Memory, 0, AlignUp(Elements / MinimumSizeBits, MinimumSizeBits));
|
||||
memset(Memory, 0, ToBytes(Elements));
|
||||
}
|
||||
void MemSet(size_t Elements) {
|
||||
memset(Memory, 0xFF, AlignUp(Elements / MinimumSizeBits, MinimumSizeBits));
|
||||
memset(Memory, 0xFF, ToBytes(Elements));
|
||||
}
|
||||
uint32_t ToBytes(size_t Elements) {
|
||||
return AlignUp(Elements, MinimumSizeBits) / MinimumSize;
|
||||
}
|
||||
|
||||
// This very explicitly doesn't let you take an address
|
||||
|
||||
@@ -401,14 +401,14 @@
|
||||
},
|
||||
"VectorTSOEnabled": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"When TSO emulation is enabled, controls if vector loadstores should also be atomic."
|
||||
]
|
||||
},
|
||||
"MemcpySetTSOEnabled": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"When TSO emulation is enabled, controls if memcpy and memset should also be atomic.",
|
||||
"Only affects REP MOVS and REP STOS instructions"
|
||||
|
||||
@@ -57,6 +57,7 @@ namespace HLE {
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class RegisterAllocationData;
|
||||
struct IRListCopy;
|
||||
class IRListView;
|
||||
namespace Validation {
|
||||
class IRValidation;
|
||||
@@ -205,9 +206,6 @@ public:
|
||||
CoreRunningMode RunningMode {CoreRunningMode::MODE_RUN};
|
||||
uint64_t VirtualMemSize {1ULL << 36};
|
||||
|
||||
// this is for internal use
|
||||
bool ValidateIRarser {false};
|
||||
|
||||
// Used if the JIT needs to have its interrupt fault code emitted.
|
||||
bool NeedsPendingInterruptFaultCheck {false};
|
||||
|
||||
@@ -293,8 +291,7 @@ public:
|
||||
void RemoveCustomIREntrypoint(uintptr_t Entrypoint);
|
||||
|
||||
struct GenerateIRResult {
|
||||
FEXCore::IR::IRListView* IRList;
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
|
||||
fextl::unique_ptr<FEXCore::IR::IRStorageBase> IR;
|
||||
uint64_t TotalInstructions;
|
||||
uint64_t TotalInstructionsLength;
|
||||
uint64_t StartAddr;
|
||||
@@ -305,9 +302,8 @@ public:
|
||||
|
||||
struct CompileCodeResult {
|
||||
void* CompiledCode;
|
||||
FEXCore::IR::IRListView* IRData;
|
||||
fextl::unique_ptr<FEXCore::IR::IRStorageBase> IR;
|
||||
FEXCore::Core::DebugData* DebugData;
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
|
||||
bool GeneratedIR;
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
|
||||
@@ -2,8 +2,6 @@
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "FEXCore/Core/X86Enums.h"
|
||||
#include "FEXCore/Utils/AllocatorHooks.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
@@ -13,6 +11,8 @@
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
#include <FEXHeaderUtils/BitUtils.h>
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
#include <CodeEmitter/Registers.h>
|
||||
|
||||
#include <aarch64/cpu-aarch64.h>
|
||||
#include <aarch64/instructions-aarch64.h>
|
||||
@@ -31,110 +31,100 @@ namespace FEXCore::CPU {
|
||||
namespace x64 {
|
||||
#ifndef _M_ARM_64EC
|
||||
// All but x19 and x29 are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 18> SRA = {
|
||||
FEXCore::ARMEmitter::Reg::r4,
|
||||
FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6,
|
||||
FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8,
|
||||
FEXCore::ARMEmitter::Reg::r9,
|
||||
FEXCore::ARMEmitter::Reg::r10,
|
||||
FEXCore::ARMEmitter::Reg::r11,
|
||||
FEXCore::ARMEmitter::Reg::r12,
|
||||
FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14,
|
||||
FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16,
|
||||
FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r19,
|
||||
FEXCore::ARMEmitter::Reg::r29,
|
||||
constexpr std::array<ARMEmitter::Register, 18> SRA = {
|
||||
ARMEmitter::Reg::r4,
|
||||
ARMEmitter::Reg::r5,
|
||||
ARMEmitter::Reg::r6,
|
||||
ARMEmitter::Reg::r7,
|
||||
ARMEmitter::Reg::r8,
|
||||
ARMEmitter::Reg::r9,
|
||||
ARMEmitter::Reg::r10,
|
||||
ARMEmitter::Reg::r11,
|
||||
ARMEmitter::Reg::r12,
|
||||
ARMEmitter::Reg::r13,
|
||||
ARMEmitter::Reg::r14,
|
||||
ARMEmitter::Reg::r15,
|
||||
ARMEmitter::Reg::r16,
|
||||
ARMEmitter::Reg::r17,
|
||||
ARMEmitter::Reg::r19,
|
||||
ARMEmitter::Reg::r29,
|
||||
// PF/AF must be last.
|
||||
REG_PF,
|
||||
REG_AF,
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 7> RA = {
|
||||
constexpr std::array<ARMEmitter::Register, 8> RA = {
|
||||
// All these callee saved
|
||||
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21, FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25, FEXCore::ARMEmitter::Reg::r30,
|
||||
ARMEmitter::Reg::r20, ARMEmitter::Reg::r21, ARMEmitter::Reg::r22, ARMEmitter::Reg::r23,
|
||||
ARMEmitter::Reg::r24, ARMEmitter::Reg::r25, ARMEmitter::Reg::r30, ARMEmitter::Reg::r18,
|
||||
};
|
||||
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 3> RAPair = {{
|
||||
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
|
||||
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
|
||||
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
|
||||
}};
|
||||
constexpr unsigned RAPairs = 6;
|
||||
|
||||
// All are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 16> SRAFPR = {
|
||||
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17, FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
|
||||
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21, FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25, FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29, FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31};
|
||||
constexpr std::array<ARMEmitter::VRegister, 16> SRAFPR = {
|
||||
ARMEmitter::VReg::v16, ARMEmitter::VReg::v17, ARMEmitter::VReg::v18, ARMEmitter::VReg::v19,
|
||||
ARMEmitter::VReg::v20, ARMEmitter::VReg::v21, ARMEmitter::VReg::v22, ARMEmitter::VReg::v23,
|
||||
ARMEmitter::VReg::v24, ARMEmitter::VReg::v25, ARMEmitter::VReg::v26, ARMEmitter::VReg::v27,
|
||||
ARMEmitter::VReg::v28, ARMEmitter::VReg::v29, ARMEmitter::VReg::v30, ARMEmitter::VReg::v31};
|
||||
|
||||
// v8..v15 = (lower 64bits) Callee saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 14> RAFPR = {
|
||||
constexpr std::array<ARMEmitter::VRegister, 14> RAFPR = {
|
||||
// v0 ~ v1 are used as temps.
|
||||
// FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
|
||||
// ARMEmitter::VReg::v0, ARMEmitter::VReg::v1,
|
||||
|
||||
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3, FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7, FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11, FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
|
||||
ARMEmitter::VReg::v2, ARMEmitter::VReg::v3, ARMEmitter::VReg::v4, ARMEmitter::VReg::v5, ARMEmitter::VReg::v6,
|
||||
ARMEmitter::VReg::v7, ARMEmitter::VReg::v8, ARMEmitter::VReg::v9, ARMEmitter::VReg::v10, ARMEmitter::VReg::v11,
|
||||
ARMEmitter::VReg::v12, ARMEmitter::VReg::v13, ARMEmitter::VReg::v14, ARMEmitter::VReg::v15,
|
||||
};
|
||||
#else
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 18> SRA = {
|
||||
FEXCore::ARMEmitter::Reg::r8,
|
||||
FEXCore::ARMEmitter::Reg::r0,
|
||||
FEXCore::ARMEmitter::Reg::r1,
|
||||
FEXCore::ARMEmitter::Reg::r27,
|
||||
constexpr std::array<ARMEmitter::Register, 18> SRA = {
|
||||
ARMEmitter::Reg::r8,
|
||||
ARMEmitter::Reg::r0,
|
||||
ARMEmitter::Reg::r1,
|
||||
ARMEmitter::Reg::r27,
|
||||
// SP's register location isn't specified by the ARM64EC ABI, we choose to use r23
|
||||
FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r29,
|
||||
FEXCore::ARMEmitter::Reg::r25,
|
||||
FEXCore::ARMEmitter::Reg::r26,
|
||||
FEXCore::ARMEmitter::Reg::r2,
|
||||
FEXCore::ARMEmitter::Reg::r3,
|
||||
FEXCore::ARMEmitter::Reg::r4,
|
||||
FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r19,
|
||||
FEXCore::ARMEmitter::Reg::r20,
|
||||
FEXCore::ARMEmitter::Reg::r21,
|
||||
FEXCore::ARMEmitter::Reg::r22,
|
||||
ARMEmitter::Reg::r23,
|
||||
ARMEmitter::Reg::r29,
|
||||
ARMEmitter::Reg::r25,
|
||||
ARMEmitter::Reg::r26,
|
||||
ARMEmitter::Reg::r2,
|
||||
ARMEmitter::Reg::r3,
|
||||
ARMEmitter::Reg::r4,
|
||||
ARMEmitter::Reg::r5,
|
||||
ARMEmitter::Reg::r19,
|
||||
ARMEmitter::Reg::r20,
|
||||
ARMEmitter::Reg::r21,
|
||||
ARMEmitter::Reg::r22,
|
||||
REG_PF,
|
||||
REG_AF,
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 7> RA = {
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7, FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17, FEXCore::ARMEmitter::Reg::r30,
|
||||
constexpr std::array<ARMEmitter::Register, 7> RA = {
|
||||
ARMEmitter::Reg::r6, ARMEmitter::Reg::r7, ARMEmitter::Reg::r14, ARMEmitter::Reg::r15,
|
||||
ARMEmitter::Reg::r16, ARMEmitter::Reg::r17, ARMEmitter::Reg::r30,
|
||||
};
|
||||
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 3> RAPair = {{
|
||||
{FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7},
|
||||
{FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15},
|
||||
{FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17},
|
||||
}};
|
||||
constexpr unsigned RAPairs = 6;
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 16> SRAFPR = {
|
||||
FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1, FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
|
||||
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5, FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9, FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
|
||||
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13, FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
|
||||
constexpr std::array<ARMEmitter::VRegister, 16> SRAFPR = {
|
||||
ARMEmitter::VReg::v0, ARMEmitter::VReg::v1, ARMEmitter::VReg::v2, ARMEmitter::VReg::v3,
|
||||
ARMEmitter::VReg::v4, ARMEmitter::VReg::v5, ARMEmitter::VReg::v6, ARMEmitter::VReg::v7,
|
||||
ARMEmitter::VReg::v8, ARMEmitter::VReg::v9, ARMEmitter::VReg::v10, ARMEmitter::VReg::v11,
|
||||
ARMEmitter::VReg::v12, ARMEmitter::VReg::v13, ARMEmitter::VReg::v14, ARMEmitter::VReg::v15,
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 14> RAFPR = {
|
||||
FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19, FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21,
|
||||
FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23, FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
|
||||
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27, FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
|
||||
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31};
|
||||
constexpr std::array<ARMEmitter::VRegister, 14> RAFPR = {
|
||||
ARMEmitter::VReg::v18, ARMEmitter::VReg::v19, ARMEmitter::VReg::v20, ARMEmitter::VReg::v21, ARMEmitter::VReg::v22,
|
||||
ARMEmitter::VReg::v23, ARMEmitter::VReg::v24, ARMEmitter::VReg::v25, ARMEmitter::VReg::v26, ARMEmitter::VReg::v27,
|
||||
ARMEmitter::VReg::v28, ARMEmitter::VReg::v29, ARMEmitter::VReg::v30, ARMEmitter::VReg::v31};
|
||||
#endif
|
||||
|
||||
// I wish this could get constexpr generated from SRA's definition but impossible until libstdc++12, libc++15.
|
||||
// SRA GPRs that need to be spilled when calling a function with `preserve_all` ABI.
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 7> PreserveAll_SRA = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5, FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
constexpr std::array<ARMEmitter::Register, 7> PreserveAll_SRA = {
|
||||
ARMEmitter::Reg::r4, ARMEmitter::Reg::r5, ARMEmitter::Reg::r6, ARMEmitter::Reg::r7,
|
||||
ARMEmitter::Reg::r8, ARMEmitter::Reg::r16, ARMEmitter::Reg::r17,
|
||||
};
|
||||
|
||||
constexpr uint32_t PreserveAll_SRAMask = {[]() -> uint32_t {
|
||||
@@ -160,12 +150,12 @@ namespace x64 {
|
||||
}()};
|
||||
|
||||
// Dynamic GPRs
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 1> PreserveAll_Dynamic = {
|
||||
constexpr std::array<ARMEmitter::Register, 1> PreserveAll_Dynamic = {
|
||||
// Only LR needs to get saved.
|
||||
FEXCore::ARMEmitter::Reg::r30};
|
||||
ARMEmitter::Reg::r30};
|
||||
|
||||
// SRA FPRs that need to be spilled when calling a function with `preserve_all` ABI.
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 0> PreserveAll_SRAFPR = {
|
||||
constexpr std::array<ARMEmitter::Register, 0> PreserveAll_SRAFPR = {
|
||||
// None.
|
||||
};
|
||||
|
||||
@@ -179,15 +169,14 @@ namespace x64 {
|
||||
|
||||
// Dynamic FPRs
|
||||
// - v0-v7
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 6> PreserveAll_DynamicFPR = {
|
||||
constexpr std::array<ARMEmitter::VRegister, 6> PreserveAll_DynamicFPR = {
|
||||
// v0 ~ v1 are temps
|
||||
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3, FEXCore::ARMEmitter::VReg::v4,
|
||||
FEXCore::ARMEmitter::VReg::v5, FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
ARMEmitter::VReg::v2, ARMEmitter::VReg::v3, ARMEmitter::VReg::v4, ARMEmitter::VReg::v5, ARMEmitter::VReg::v6, ARMEmitter::VReg::v7,
|
||||
};
|
||||
|
||||
// SRA FPRs that need to be spilled when the host supports SVE-256bit with `preserve_all` ABI.
|
||||
// This is /all/ of the SRA registers
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 16> PreserveAll_SRAFPRSVE = SRAFPR;
|
||||
constexpr std::array<ARMEmitter::VRegister, 16> PreserveAll_SRAFPRSVE = SRAFPR;
|
||||
|
||||
constexpr uint32_t PreserveAll_SRAFPRSVEMask = {[]() -> uint32_t {
|
||||
uint32_t Mask {};
|
||||
@@ -198,89 +187,77 @@ namespace x64 {
|
||||
}()};
|
||||
|
||||
// Dynamic FPRs when the host supports SVE-256bit.
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 14> PreserveAll_DynamicFPRSVE = {
|
||||
constexpr std::array<ARMEmitter::VRegister, 14> PreserveAll_DynamicFPRSVE = {
|
||||
// v0 ~ v1 are used as temps.
|
||||
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3, FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7, FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11, FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
|
||||
ARMEmitter::VReg::v2, ARMEmitter::VReg::v3, ARMEmitter::VReg::v4, ARMEmitter::VReg::v5, ARMEmitter::VReg::v6,
|
||||
ARMEmitter::VReg::v7, ARMEmitter::VReg::v8, ARMEmitter::VReg::v9, ARMEmitter::VReg::v10, ARMEmitter::VReg::v11,
|
||||
ARMEmitter::VReg::v12, ARMEmitter::VReg::v13, ARMEmitter::VReg::v14, ARMEmitter::VReg::v15,
|
||||
};
|
||||
} // namespace x64
|
||||
|
||||
namespace x32 {
|
||||
// All but x19 and x29 are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 10> SRA = {
|
||||
FEXCore::ARMEmitter::Reg::r4,
|
||||
FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6,
|
||||
FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8,
|
||||
FEXCore::ARMEmitter::Reg::r9,
|
||||
FEXCore::ARMEmitter::Reg::r10,
|
||||
FEXCore::ARMEmitter::Reg::r11,
|
||||
constexpr std::array<ARMEmitter::Register, 10> SRA = {
|
||||
ARMEmitter::Reg::r4,
|
||||
ARMEmitter::Reg::r5,
|
||||
ARMEmitter::Reg::r6,
|
||||
ARMEmitter::Reg::r7,
|
||||
ARMEmitter::Reg::r8,
|
||||
ARMEmitter::Reg::r9,
|
||||
ARMEmitter::Reg::r10,
|
||||
ARMEmitter::Reg::r11,
|
||||
// PF/AF must be last.
|
||||
REG_PF,
|
||||
REG_AF,
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 15> RA = {
|
||||
constexpr std::array<ARMEmitter::Register, 15> RA = {
|
||||
// All these callee saved
|
||||
FEXCore::ARMEmitter::Reg::r20,
|
||||
FEXCore::ARMEmitter::Reg::r21,
|
||||
FEXCore::ARMEmitter::Reg::r22,
|
||||
FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r24,
|
||||
FEXCore::ARMEmitter::Reg::r25,
|
||||
ARMEmitter::Reg::r20,
|
||||
ARMEmitter::Reg::r21,
|
||||
ARMEmitter::Reg::r22,
|
||||
ARMEmitter::Reg::r23,
|
||||
ARMEmitter::Reg::r24,
|
||||
ARMEmitter::Reg::r25,
|
||||
|
||||
// Registers only available on 32-bit
|
||||
// All these are caller saved (except for r19).
|
||||
FEXCore::ARMEmitter::Reg::r12,
|
||||
FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14,
|
||||
FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16,
|
||||
FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r29,
|
||||
FEXCore::ARMEmitter::Reg::r30,
|
||||
ARMEmitter::Reg::r12,
|
||||
ARMEmitter::Reg::r13,
|
||||
ARMEmitter::Reg::r14,
|
||||
ARMEmitter::Reg::r15,
|
||||
ARMEmitter::Reg::r16,
|
||||
ARMEmitter::Reg::r17,
|
||||
ARMEmitter::Reg::r29,
|
||||
ARMEmitter::Reg::r30,
|
||||
|
||||
FEXCore::ARMEmitter::Reg::r19,
|
||||
ARMEmitter::Reg::r19,
|
||||
};
|
||||
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 7> RAPair = {{
|
||||
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
|
||||
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
|
||||
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
|
||||
|
||||
{FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13},
|
||||
{FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15},
|
||||
{FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17},
|
||||
{FEXCore::ARMEmitter::Reg::r29, FEXCore::ARMEmitter::Reg::r30},
|
||||
}};
|
||||
constexpr unsigned RAPairs = 12;
|
||||
|
||||
// All are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 8> SRAFPR = {
|
||||
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17, FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
|
||||
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21, FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
|
||||
constexpr std::array<ARMEmitter::VRegister, 8> SRAFPR = {
|
||||
ARMEmitter::VReg::v16, ARMEmitter::VReg::v17, ARMEmitter::VReg::v18, ARMEmitter::VReg::v19,
|
||||
ARMEmitter::VReg::v20, ARMEmitter::VReg::v21, ARMEmitter::VReg::v22, ARMEmitter::VReg::v23,
|
||||
};
|
||||
|
||||
// v8..v15 = (lower 64bits) Callee saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 22> RAFPR = {
|
||||
constexpr std::array<ARMEmitter::VRegister, 22> RAFPR = {
|
||||
// v0 ~ v1 are used as temps.
|
||||
// FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
|
||||
// ARMEmitter::VReg::v0, ARMEmitter::VReg::v1,
|
||||
|
||||
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3, FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7, FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11, FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
|
||||
ARMEmitter::VReg::v2, ARMEmitter::VReg::v3, ARMEmitter::VReg::v4, ARMEmitter::VReg::v5, ARMEmitter::VReg::v6,
|
||||
ARMEmitter::VReg::v7, ARMEmitter::VReg::v8, ARMEmitter::VReg::v9, ARMEmitter::VReg::v10, ARMEmitter::VReg::v11,
|
||||
ARMEmitter::VReg::v12, ARMEmitter::VReg::v13, ARMEmitter::VReg::v14, ARMEmitter::VReg::v15,
|
||||
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25, FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29, FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31};
|
||||
ARMEmitter::VReg::v24, ARMEmitter::VReg::v25, ARMEmitter::VReg::v26, ARMEmitter::VReg::v27, ARMEmitter::VReg::v28,
|
||||
ARMEmitter::VReg::v29, ARMEmitter::VReg::v30, ARMEmitter::VReg::v31};
|
||||
|
||||
// I wish this could get constexpr generated from SRA's definition but impossible until libstdc++12, libc++15.
|
||||
// SRA GPRs that need to be spilled when calling a function with `preserve_all` ABI.
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 5> PreserveAll_SRA = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5, FEXCore::ARMEmitter::Reg::r6,
|
||||
FEXCore::ARMEmitter::Reg::r7, FEXCore::ARMEmitter::Reg::r8,
|
||||
constexpr std::array<ARMEmitter::Register, 5> PreserveAll_SRA = {
|
||||
ARMEmitter::Reg::r4, ARMEmitter::Reg::r5, ARMEmitter::Reg::r6, ARMEmitter::Reg::r7, ARMEmitter::Reg::r8,
|
||||
};
|
||||
|
||||
constexpr uint32_t PreserveAll_SRAMask = {[]() -> uint32_t {
|
||||
@@ -306,11 +283,10 @@ namespace x32 {
|
||||
}()};
|
||||
|
||||
// Dynamic GPRs
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 3> PreserveAll_Dynamic = {
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17, FEXCore::ARMEmitter::Reg::r30};
|
||||
constexpr std::array<ARMEmitter::Register, 3> PreserveAll_Dynamic = {ARMEmitter::Reg::r16, ARMEmitter::Reg::r17, ARMEmitter::Reg::r30};
|
||||
|
||||
// SRA FPRs that need to be spilled when calling a function with `preserve_all` ABI.
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 0> PreserveAll_SRAFPR = {
|
||||
constexpr std::array<ARMEmitter::Register, 0> PreserveAll_SRAFPR = {
|
||||
// None.
|
||||
};
|
||||
|
||||
@@ -324,15 +300,14 @@ namespace x32 {
|
||||
|
||||
// Dynamic FPRs
|
||||
// - v0-v7
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 6> PreserveAll_DynamicFPR = {
|
||||
constexpr std::array<ARMEmitter::VRegister, 6> PreserveAll_DynamicFPR = {
|
||||
// v0 ~ v1 are temps
|
||||
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3, FEXCore::ARMEmitter::VReg::v4,
|
||||
FEXCore::ARMEmitter::VReg::v5, FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
ARMEmitter::VReg::v2, ARMEmitter::VReg::v3, ARMEmitter::VReg::v4, ARMEmitter::VReg::v5, ARMEmitter::VReg::v6, ARMEmitter::VReg::v7,
|
||||
};
|
||||
|
||||
// SRA FPRs that need to be spilled when the host supports SVE-256bit with `preserve_all` ABI.
|
||||
// This is /all/ of the SRA registers
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 8> PreserveAll_SRAFPRSVE = SRAFPR;
|
||||
constexpr std::array<ARMEmitter::VRegister, 8> PreserveAll_SRAFPRSVE = SRAFPR;
|
||||
|
||||
constexpr uint32_t PreserveAll_SRAFPRSVEMask = {[]() -> uint32_t {
|
||||
uint32_t Mask {};
|
||||
@@ -343,15 +318,14 @@ namespace x32 {
|
||||
}()};
|
||||
|
||||
// Dynamic FPRs when the host supports SVE-256bit.
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 22> PreserveAll_DynamicFPRSVE = {
|
||||
constexpr std::array<ARMEmitter::VRegister, 22> PreserveAll_DynamicFPRSVE = {
|
||||
// v0 ~ v1 are used as temps.
|
||||
FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3, FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7, FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11, FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
|
||||
ARMEmitter::VReg::v2, ARMEmitter::VReg::v3, ARMEmitter::VReg::v4, ARMEmitter::VReg::v5, ARMEmitter::VReg::v6,
|
||||
ARMEmitter::VReg::v7, ARMEmitter::VReg::v8, ARMEmitter::VReg::v9, ARMEmitter::VReg::v10, ARMEmitter::VReg::v11,
|
||||
ARMEmitter::VReg::v12, ARMEmitter::VReg::v13, ARMEmitter::VReg::v14, ARMEmitter::VReg::v15,
|
||||
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25, FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29, FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31};
|
||||
ARMEmitter::VReg::v24, ARMEmitter::VReg::v25, ARMEmitter::VReg::v26, ARMEmitter::VReg::v27, ARMEmitter::VReg::v28,
|
||||
ARMEmitter::VReg::v29, ARMEmitter::VReg::v30, ARMEmitter::VReg::v31};
|
||||
} // namespace x32
|
||||
|
||||
// We want vixl to not allocate a default buffer. Jit and dispatcher will manually create one.
|
||||
@@ -385,18 +359,18 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr
|
||||
if (EmitterCTX->Config.Is64BitMode()) {
|
||||
StaticRegisters = x64::SRA;
|
||||
GeneralRegisters = x64::RA;
|
||||
GeneralPairRegisters = x64::RAPair;
|
||||
StaticFPRegisters = x64::SRAFPR;
|
||||
GeneralFPRegisters = x64::RAFPR;
|
||||
PairRegisters = x64::RAPairs;
|
||||
#ifdef _M_ARM_64EC
|
||||
ConfiguredDynamicRegisterBase = std::span(x64::RA.begin(), 7);
|
||||
#endif
|
||||
} else {
|
||||
ConfiguredDynamicRegisterBase = std::span(x32::RA.begin() + 6, 8);
|
||||
PairRegisters = x32::RAPairs;
|
||||
|
||||
StaticRegisters = x32::SRA;
|
||||
GeneralRegisters = x32::RA;
|
||||
GeneralPairRegisters = x32::RAPair;
|
||||
|
||||
StaticFPRegisters = x32::SRAFPR;
|
||||
GeneralFPRegisters = x32::RAFPR;
|
||||
@@ -463,6 +437,17 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
|
||||
}
|
||||
}
|
||||
|
||||
// If we can't handle negatives with the orr, try with movn+movk
|
||||
if (Is64Bit && ((~Constant) >> 32) == 0) {
|
||||
movn(s, Reg, (~Constant) & 0xFFFF);
|
||||
movk(s, Reg, (Constant >> 16) & 0xFFFF, 16);
|
||||
if (NOPPad) {
|
||||
nop();
|
||||
nop();
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
// ADRP+ADD is specifically optimized in hardware
|
||||
// Check if we can use this
|
||||
auto PC = GetCursorAddress<uint64_t>();
|
||||
@@ -589,7 +574,7 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FPRs, uint32_t GPRSpillMask, uint32_t FPRSpillMask) {
|
||||
void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs, uint32_t GPRSpillMask, uint32_t FPRSpillMask) {
|
||||
#ifndef VIXL_SIMULATOR
|
||||
if (EmitterCTX->HostFeatures.SupportsAFP) {
|
||||
// Disable AFP features when spilling registers.
|
||||
@@ -683,7 +668,7 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
|
||||
}
|
||||
|
||||
void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRFillMask) {
|
||||
FEXCore::ARMEmitter::Register TmpReg = FEXCore::ARMEmitter::Reg::r0;
|
||||
ARMEmitter::Register TmpReg = ARMEmitter::Reg::r0;
|
||||
LOGMAN_THROW_A_FMT(GPRFillMask != 0, "Must fill at least 1 GPR for a temp");
|
||||
[[maybe_unused]] bool FoundRegister {};
|
||||
for (auto Reg : StaticRegisters) {
|
||||
@@ -798,7 +783,7 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::PushVectorRegisters(FEXCore::ARMEmitter::Register TmpReg, bool SVERegs, std::span<const FEXCore::ARMEmitter::VRegister> VRegs) {
|
||||
void Arm64Emitter::PushVectorRegisters(ARMEmitter::Register TmpReg, bool SVERegs, std::span<const ARMEmitter::VRegister> VRegs) {
|
||||
if (SVERegs) {
|
||||
size_t i = 0;
|
||||
|
||||
@@ -835,7 +820,7 @@ void Arm64Emitter::PushVectorRegisters(FEXCore::ARMEmitter::Register TmpReg, boo
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::PushGeneralRegisters(FEXCore::ARMEmitter::Register TmpReg, std::span<const FEXCore::ARMEmitter::Register> Regs) {
|
||||
void Arm64Emitter::PushGeneralRegisters(ARMEmitter::Register TmpReg, std::span<const ARMEmitter::Register> Regs) {
|
||||
size_t i = 0;
|
||||
for (; i < (Regs.size() % 2); ++i) {
|
||||
const auto Reg1 = Regs[i];
|
||||
@@ -849,7 +834,7 @@ void Arm64Emitter::PushGeneralRegisters(FEXCore::ARMEmitter::Register TmpReg, st
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::PopVectorRegisters(bool SVERegs, std::span<const FEXCore::ARMEmitter::VRegister> VRegs) {
|
||||
void Arm64Emitter::PopVectorRegisters(bool SVERegs, std::span<const ARMEmitter::VRegister> VRegs) {
|
||||
if (SVERegs) {
|
||||
size_t i = 0;
|
||||
for (; i < (VRegs.size() % 4); i += 2) {
|
||||
@@ -885,7 +870,7 @@ void Arm64Emitter::PopVectorRegisters(bool SVERegs, std::span<const FEXCore::ARM
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::PopGeneralRegisters(std::span<const FEXCore::ARMEmitter::Register> Regs) {
|
||||
void Arm64Emitter::PopGeneralRegisters(std::span<const ARMEmitter::Register> Regs) {
|
||||
size_t i = 0;
|
||||
for (; i < (Regs.size() % 2); ++i) {
|
||||
const auto Reg1 = Regs[i];
|
||||
@@ -898,7 +883,7 @@ void Arm64Emitter::PopGeneralRegisters(std::span<const FEXCore::ARMEmitter::Regi
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg) {
|
||||
void Arm64Emitter::PushDynamicRegsAndLR(ARMEmitter::Register TmpReg) {
|
||||
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
|
||||
const auto GPRSize = (ConfiguredDynamicRegisterBase.size() + 1) * Core::CPUState::GPR_REG_SIZE;
|
||||
const auto FPRRegSize = CanUseSVE ? Core::CPUState::XMM_AVX_REG_SIZE : Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
@@ -937,12 +922,12 @@ void Arm64Emitter::PopDynamicRegsAndLR() {
|
||||
#endif
|
||||
}
|
||||
|
||||
void Arm64Emitter::SpillForPreserveAllABICall(FEXCore::ARMEmitter::Register TmpReg, bool FPRs) {
|
||||
void Arm64Emitter::SpillForPreserveAllABICall(ARMEmitter::Register TmpReg, bool FPRs) {
|
||||
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
|
||||
const auto FPRRegSize = CanUseSVE ? Core::CPUState::XMM_AVX_REG_SIZE : Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
std::span<const FEXCore::ARMEmitter::Register> DynamicGPRs {};
|
||||
std::span<const FEXCore::ARMEmitter::VRegister> DynamicFPRs {};
|
||||
std::span<const ARMEmitter::Register> DynamicGPRs {};
|
||||
std::span<const ARMEmitter::VRegister> DynamicFPRs {};
|
||||
uint32_t PreserveSRAMask {};
|
||||
uint32_t PreserveSRAFPRMask {};
|
||||
if (EmitterCTX->Config.Is64BitMode()) {
|
||||
@@ -989,8 +974,8 @@ void Arm64Emitter::SpillForPreserveAllABICall(FEXCore::ARMEmitter::Register TmpR
|
||||
void Arm64Emitter::FillForPreserveAllABICall(bool FPRs) {
|
||||
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
|
||||
|
||||
std::span<const FEXCore::ARMEmitter::Register> DynamicGPRs {};
|
||||
std::span<const FEXCore::ARMEmitter::VRegister> DynamicFPRs {};
|
||||
std::span<const ARMEmitter::Register> DynamicGPRs {};
|
||||
std::span<const ARMEmitter::VRegister> DynamicFPRs {};
|
||||
uint32_t PreserveSRAMask {};
|
||||
uint32_t PreserveSRAFPRMask {};
|
||||
|
||||
|
||||
@@ -2,9 +2,6 @@
|
||||
#pragma once
|
||||
|
||||
#include "FEXCore/Utils/EnumUtils.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
|
||||
|
||||
#include "Interface/Core/ObjectCache/Relocations.h"
|
||||
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
@@ -22,6 +19,8 @@
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
#include <CodeEmitter/Registers.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstddef>
|
||||
@@ -35,85 +34,74 @@ class ContextImpl;
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
// Contains the address to the currently available CPU state
|
||||
constexpr auto STATE = FEXCore::ARMEmitter::XReg::x28;
|
||||
constexpr auto STATE = ARMEmitter::XReg::x28;
|
||||
|
||||
#ifndef _M_ARM_64EC
|
||||
// GPR temporaries. Only x3 can be used across spill boundaries
|
||||
// so if these ever need to change, be very careful about that.
|
||||
constexpr auto TMP1 = FEXCore::ARMEmitter::XReg::x0;
|
||||
constexpr auto TMP2 = FEXCore::ARMEmitter::XReg::x1;
|
||||
constexpr auto TMP3 = FEXCore::ARMEmitter::XReg::x2;
|
||||
constexpr auto TMP4 = FEXCore::ARMEmitter::XReg::x3;
|
||||
constexpr auto TMP1 = ARMEmitter::XReg::x0;
|
||||
constexpr auto TMP2 = ARMEmitter::XReg::x1;
|
||||
constexpr auto TMP3 = ARMEmitter::XReg::x2;
|
||||
constexpr auto TMP4 = ARMEmitter::XReg::x3;
|
||||
constexpr bool TMP_ABIARGS = true;
|
||||
|
||||
// We pin r26/r27 as PF/AF respectively, this is internal FEX ABI.
|
||||
constexpr auto REG_PF = FEXCore::ARMEmitter::Reg::r26;
|
||||
constexpr auto REG_AF = FEXCore::ARMEmitter::Reg::r27;
|
||||
constexpr auto REG_PF = ARMEmitter::Reg::r26;
|
||||
constexpr auto REG_AF = ARMEmitter::Reg::r27;
|
||||
|
||||
// Vector temporaries
|
||||
constexpr auto VTMP1 = FEXCore::ARMEmitter::VReg::v0;
|
||||
constexpr auto VTMP2 = FEXCore::ARMEmitter::VReg::v1;
|
||||
constexpr auto VTMP1 = ARMEmitter::VReg::v0;
|
||||
constexpr auto VTMP2 = ARMEmitter::VReg::v1;
|
||||
#else
|
||||
constexpr auto TMP1 = FEXCore::ARMEmitter::XReg::x10;
|
||||
constexpr auto TMP2 = FEXCore::ARMEmitter::XReg::x11;
|
||||
constexpr auto TMP3 = FEXCore::ARMEmitter::XReg::x12;
|
||||
constexpr auto TMP4 = FEXCore::ARMEmitter::XReg::x13;
|
||||
constexpr auto TMP1 = ARMEmitter::XReg::x10;
|
||||
constexpr auto TMP2 = ARMEmitter::XReg::x11;
|
||||
constexpr auto TMP3 = ARMEmitter::XReg::x12;
|
||||
constexpr auto TMP4 = ARMEmitter::XReg::x13;
|
||||
constexpr bool TMP_ABIARGS = false;
|
||||
|
||||
// We pin r11/r12 as PF/AF respectively for arm64ec, as r26/r27 are used for SRA.
|
||||
constexpr auto REG_PF = FEXCore::ARMEmitter::Reg::r9;
|
||||
constexpr auto REG_AF = FEXCore::ARMEmitter::Reg::r24;
|
||||
constexpr auto REG_PF = ARMEmitter::Reg::r9;
|
||||
constexpr auto REG_AF = ARMEmitter::Reg::r24;
|
||||
|
||||
// Vector temporaries
|
||||
constexpr auto VTMP1 = FEXCore::ARMEmitter::VReg::v16;
|
||||
constexpr auto VTMP2 = FEXCore::ARMEmitter::VReg::v17;
|
||||
constexpr auto VTMP1 = ARMEmitter::VReg::v16;
|
||||
constexpr auto VTMP2 = ARMEmitter::VReg::v17;
|
||||
|
||||
// Entry/Exit ABI
|
||||
constexpr auto EC_CALL_CHECKER_PC_REG = ARMEmitter::XReg::x9;
|
||||
constexpr auto EC_ENTRY_CPUAREA_REG = ARMEmitter::XReg::x17;
|
||||
#endif
|
||||
|
||||
// Predicate register temporaries (used when AVX support is enabled)
|
||||
// PRED_TMP_16B indicates a predicate register that indicates the first 16 bytes set to 1.
|
||||
// PRED_TMP_32B indicates a predicate register that indicates the first 32 bytes set to 1.
|
||||
constexpr FEXCore::ARMEmitter::PRegister PRED_TMP_16B = FEXCore::ARMEmitter::PReg::p6;
|
||||
constexpr FEXCore::ARMEmitter::PRegister PRED_TMP_32B = FEXCore::ARMEmitter::PReg::p7;
|
||||
constexpr ARMEmitter::PRegister PRED_TMP_16B = ARMEmitter::PReg::p6;
|
||||
constexpr ARMEmitter::PRegister PRED_TMP_32B = ARMEmitter::PReg::p7;
|
||||
|
||||
|
||||
// This class contains common emitter utility functions that can
|
||||
// be used by both Arm64 JIT and ARM64 Dispatcher
|
||||
class Arm64Emitter : public FEXCore::ARMEmitter::Emitter {
|
||||
class Arm64Emitter : public ARMEmitter::Emitter {
|
||||
protected:
|
||||
Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr = nullptr, size_t size = 0);
|
||||
|
||||
FEXCore::Context::ContextImpl* EmitterCTX;
|
||||
vixl::aarch64::CPU CPU;
|
||||
|
||||
std::span<const FEXCore::ARMEmitter::Register> ConfiguredDynamicRegisterBase {};
|
||||
std::span<const FEXCore::ARMEmitter::Register> StaticRegisters {};
|
||||
std::span<const FEXCore::ARMEmitter::Register> GeneralRegisters {};
|
||||
std::span<const std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>> GeneralPairRegisters {};
|
||||
std::span<const FEXCore::ARMEmitter::VRegister> StaticFPRegisters {};
|
||||
std::span<const FEXCore::ARMEmitter::VRegister> GeneralFPRegisters {};
|
||||
std::span<const ARMEmitter::Register> ConfiguredDynamicRegisterBase {};
|
||||
std::span<const ARMEmitter::Register> StaticRegisters {};
|
||||
std::span<const ARMEmitter::Register> GeneralRegisters {};
|
||||
std::span<const ARMEmitter::VRegister> StaticFPRegisters {};
|
||||
std::span<const ARMEmitter::VRegister> GeneralFPRegisters {};
|
||||
uint32_t PairRegisters = 0;
|
||||
|
||||
/**
|
||||
* @name Register Allocation
|
||||
* @{ */
|
||||
constexpr static uint32_t RegisterClasses = 6;
|
||||
|
||||
constexpr static uint64_t GPRBase = (0ULL << 32);
|
||||
constexpr static uint64_t FPRBase = (1ULL << 32);
|
||||
constexpr static uint64_t GPRPairBase = (2ULL << 32);
|
||||
|
||||
/** @} */
|
||||
|
||||
constexpr static uint8_t RA_32 = 0;
|
||||
constexpr static uint8_t RA_64 = 1;
|
||||
constexpr static uint8_t RA_FPR = 2;
|
||||
|
||||
void LoadConstant(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad = false);
|
||||
void LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad = false);
|
||||
|
||||
|
||||
// NOTE: These functions WILL clobber the register TMP4 if AVX support is enabled
|
||||
// and FPRs are being spilled or filled. If only GPRs are spilled/filled, then
|
||||
// TMP4 is left alone.
|
||||
void SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FPRs = true, uint32_t GPRSpillMask = ~0U, uint32_t FPRSpillMask = ~0U);
|
||||
void SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs = true, uint32_t GPRSpillMask = ~0U, uint32_t FPRSpillMask = ~0U);
|
||||
void FillStaticRegs(bool FPRs = true, uint32_t GPRFillMask = ~0U, uint32_t FPRFillMask = ~0U);
|
||||
|
||||
// Register 0-18 + 29 + 30 are caller saved
|
||||
@@ -124,13 +112,13 @@ protected:
|
||||
static constexpr uint32_t CALLER_FPR_MASK = ~0U;
|
||||
|
||||
// Generic push and pop vector registers.
|
||||
void PushVectorRegisters(FEXCore::ARMEmitter::Register TmpReg, bool SVERegs, std::span<const FEXCore::ARMEmitter::VRegister> VRegs);
|
||||
void PushGeneralRegisters(FEXCore::ARMEmitter::Register TmpReg, std::span<const FEXCore::ARMEmitter::Register> Regs);
|
||||
void PushVectorRegisters(ARMEmitter::Register TmpReg, bool SVERegs, std::span<const ARMEmitter::VRegister> VRegs);
|
||||
void PushGeneralRegisters(ARMEmitter::Register TmpReg, std::span<const ARMEmitter::Register> Regs);
|
||||
|
||||
void PopVectorRegisters(bool SVERegs, std::span<const FEXCore::ARMEmitter::VRegister> VRegs);
|
||||
void PopGeneralRegisters(std::span<const FEXCore::ARMEmitter::Register> Regs);
|
||||
void PopVectorRegisters(bool SVERegs, std::span<const ARMEmitter::VRegister> VRegs);
|
||||
void PopGeneralRegisters(std::span<const ARMEmitter::Register> Regs);
|
||||
|
||||
void PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg);
|
||||
void PushDynamicRegsAndLR(ARMEmitter::Register TmpReg);
|
||||
void PopDynamicRegsAndLR();
|
||||
|
||||
void PushCalleeSavedRegisters();
|
||||
@@ -146,10 +134,10 @@ protected:
|
||||
// Callee Saved:
|
||||
// - X9-X15, X19-X31
|
||||
// - Low 128-bits of v8-v31
|
||||
void SpillForPreserveAllABICall(FEXCore::ARMEmitter::Register TmpReg, bool FPRs = true);
|
||||
void SpillForPreserveAllABICall(ARMEmitter::Register TmpReg, bool FPRs = true);
|
||||
void FillForPreserveAllABICall(bool FPRs = true);
|
||||
|
||||
void SpillForABICall(bool SupportsPreserveAllABI, FEXCore::ARMEmitter::Register TmpReg, bool FPRs = true) {
|
||||
void SpillForABICall(bool SupportsPreserveAllABI, ARMEmitter::Register TmpReg, bool FPRs = true) {
|
||||
if (SupportsPreserveAllABI) {
|
||||
SpillForPreserveAllABICall(TmpReg, FPRs);
|
||||
} else {
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -140,7 +140,7 @@ namespace CPU {
|
||||
*/
|
||||
[[nodiscard]]
|
||||
virtual CompiledCode CompileCode(uint64_t Entry, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData,
|
||||
FEXCore::IR::RegisterAllocationData* RAData) = 0;
|
||||
const FEXCore::IR::RegisterAllocationData* RAData) = 0;
|
||||
|
||||
/**
|
||||
* @brief Relocates a block of code from the JIT code object cache
|
||||
|
||||
@@ -69,6 +69,8 @@ namespace ProductNames {
|
||||
|
||||
static const char ARM_Firestorm[] = "Apple Firestorm";
|
||||
static const char ARM_Icestorm[] = "Apple Icestorm";
|
||||
|
||||
static const char ARM_ORYON_1[] = "Oryon-1";
|
||||
#else
|
||||
#endif
|
||||
} // namespace ProductNames
|
||||
@@ -140,8 +142,10 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
// CPU priority order
|
||||
// This is mostly arbitrary but will sort by some sort of CPU priority by performance
|
||||
// Relative list so things they will commonly end up in big.little configurations sort of relate
|
||||
static constexpr std::array<CPUMIDR, 42> CPUMIDRs = {{
|
||||
static constexpr std::array<CPUMIDR, 43> CPUMIDRs = {{
|
||||
// Typically big CPU cores
|
||||
{0x51, 0x001, 1, ProductNames::ARM_ORYON_1}, // Qualcomm Oryon-1
|
||||
|
||||
{0x61, 0x023, 1, ProductNames::ARM_Firestorm}, // Apple M1 Firestorm
|
||||
|
||||
{0x41, 0xd82, 1, ProductNames::ARM_X4}, // X4
|
||||
@@ -355,14 +359,14 @@ void CPUIDEmu::SetupFeatures() {
|
||||
return;
|
||||
}
|
||||
|
||||
#define ENABLE_DISABLE_OPTION(FeatureName, name, enum_name) \
|
||||
do { \
|
||||
const bool Disable##name = (CPUIDFeatures() & FEXCore::Config::CPUID::DISABLE##enum_name) != 0; \
|
||||
const bool Enable##name = (CPUIDFeatures() & FEXCore::Config::CPUID::ENABLE##enum_name) != 0; \
|
||||
#define ENABLE_DISABLE_OPTION(FeatureName, name, enum_name) \
|
||||
do { \
|
||||
const bool Disable##name = (CPUIDFeatures() & FEXCore::Config::CPUID::DISABLE##enum_name) != 0; \
|
||||
const bool Enable##name = (CPUIDFeatures() & FEXCore::Config::CPUID::ENABLE##enum_name) != 0; \
|
||||
LogMan::Throw::AFmt(!(Disable##name && Enable##name), "Disabling and Enabling CPU feature (" #name ") is mutually exclusive"); \
|
||||
const bool AlreadyEnabled = Features.FeatureName; \
|
||||
const bool Result = (AlreadyEnabled | Enable##name) & !Disable##name; \
|
||||
Features.FeatureName = Result; \
|
||||
const bool AlreadyEnabled = Features.FeatureName; \
|
||||
const bool Result = (AlreadyEnabled | Enable##name) & !Disable##name; \
|
||||
Features.FeatureName = Result; \
|
||||
} while (0)
|
||||
|
||||
ENABLE_DISABLE_OPTION(SHA, SHA, SHA);
|
||||
|
||||
@@ -366,7 +366,7 @@ void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread)
|
||||
|
||||
Thread->CTX = this;
|
||||
|
||||
Thread->PassManager->AddDefaultPasses(this, Config.Core == FEXCore::Config::CONFIG_IRJIT);
|
||||
Thread->PassManager->AddDefaultPasses(this);
|
||||
Thread->PassManager->AddDefaultValidationPasses();
|
||||
|
||||
Thread->PassManager->RegisterSyscallHandler(SyscallHandler);
|
||||
@@ -374,7 +374,7 @@ void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread)
|
||||
// Create CPU backend
|
||||
switch (Config.Core) {
|
||||
case FEXCore::Config::CONFIG_IRJIT:
|
||||
Thread->PassManager->InsertRegisterAllocationPass(HostFeatures.SupportsAVX);
|
||||
Thread->PassManager->InsertRegisterAllocationPass();
|
||||
Thread->CPUBackend = FEXCore::CPU::CreateArm64JITCore(this, Thread);
|
||||
break;
|
||||
case FEXCore::Config::CONFIG_CUSTOM: Thread->CPUBackend = CustomCPUFactory(this, Thread); break;
|
||||
@@ -403,8 +403,6 @@ ContextImpl::CreateThread(uint64_t InitialRIP, uint64_t StackPointer, FEXCore::C
|
||||
InitializeCompiler(Thread);
|
||||
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Store(0);
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress =
|
||||
reinterpret_cast<Core::NonAtomicRefCounter<uint64_t>*>(FEXCore::Allocator::VirtualAlloc(4096));
|
||||
|
||||
if (Config.BlockJITNaming() || Config.GlobalJITNaming() || Config.LibraryJITNaming()) {
|
||||
// Allocate a JIT symbol buffer only if enabled.
|
||||
@@ -421,7 +419,8 @@ void ContextImpl::DestroyThread(FEXCore::Core::InternalThreadState* Thread, bool
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCore::Allocator::VirtualFree(reinterpret_cast<void*>(Thread->CurrentFrame->State.DeferredSignalFaultAddress), 4096);
|
||||
FEXCore::Allocator::VirtualProtect(&Thread->InterruptFaultPage, sizeof(Thread->InterruptFaultPage),
|
||||
Allocator::ProtectOptions::Read | Allocator::ProtectOptions::Write);
|
||||
delete Thread;
|
||||
}
|
||||
|
||||
@@ -469,29 +468,39 @@ static void IRDumper(FEXCore::Core::InternalThreadState* Thread, IR::IREmitter*
|
||||
fextl::fmt::print(FD, "IR-ShouldDump-{} 0x{:x}:\n{}\n@@@@@\n", RA ? "post" : "pre", GuestRIP, out.str());
|
||||
};
|
||||
|
||||
static void ValidateIR(ContextImpl* ctx, IR::IREmitter* IREmitter) {
|
||||
// Convert to text, Parse, Convert to text again and make sure the texts match
|
||||
fextl::stringstream out;
|
||||
static auto compaction = IR::CreateIRCompaction(ctx->OpDispatcherAllocator);
|
||||
compaction->Run(IREmitter);
|
||||
auto NewIR = IREmitter->ViewIR();
|
||||
Dump(&out, &NewIR, nullptr);
|
||||
out.seekg(0);
|
||||
FEXCore::Utils::PooledAllocatorMalloc Allocator;
|
||||
auto reparsed = IR::Parse(Allocator, out);
|
||||
if (reparsed == nullptr) {
|
||||
LOGMAN_MSG_A_FMT("Failed to parse IR\n");
|
||||
} else {
|
||||
fextl::stringstream out2;
|
||||
auto NewIR2 = reparsed->ViewIR();
|
||||
Dump(&out2, &NewIR2, nullptr);
|
||||
if (out.str() != out2.str()) {
|
||||
LogMan::Msg::IFmt("one:\n {}", out.str());
|
||||
LogMan::Msg::IFmt("two:\n {}", out2.str());
|
||||
LOGMAN_MSG_A_FMT("Parsed IR doesn't match\n");
|
||||
}
|
||||
// IRStorageBase with fully owned memory
|
||||
struct IRListCopy : public IR::IRStorageBase {
|
||||
std::span<std::byte> IRData;
|
||||
std::span<std::byte> ListData;
|
||||
|
||||
// TODO: Consider defaulting to empty RAData instead?
|
||||
IR::RegisterAllocationData::UniquePtr RADataInternal;
|
||||
|
||||
IRListCopy(const IR::IRListView& view, IR::RegisterAllocationData::UniquePtr RAData)
|
||||
: RADataInternal(std::move(RAData)) {
|
||||
std::byte* Storage = reinterpret_cast<std::byte*>(FEXCore::Allocator::malloc(view.GetDataSize() + view.GetListSize()));
|
||||
|
||||
IRData = {Storage, Storage + view.GetDataSize()};
|
||||
ListData = {Storage + view.GetDataSize(), Storage + view.GetDataSize() + view.GetListSize()};
|
||||
memcpy(IRData.data(), (char*)view.GetData(), IRData.size());
|
||||
memcpy(ListData.data(), (char*)view.GetListData(), ListData.size());
|
||||
}
|
||||
}
|
||||
|
||||
IRListCopy(const IRListCopy& other) = delete;
|
||||
IRListCopy(IRListCopy&& other) = delete;
|
||||
|
||||
~IRListCopy() {
|
||||
FEXCore::Allocator::free(IRData.data());
|
||||
}
|
||||
|
||||
const IR::RegisterAllocationData* RAData() override {
|
||||
return RADataInternal.get();
|
||||
}
|
||||
IR::IRListView GetIRView() override {
|
||||
return IR::IRListView {IRData.data(), ListData.data(), IRData.size(), ListData.size()};
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
ContextImpl::GenerateIRResult
|
||||
ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, bool ExtendedDebugInfo, uint64_t MaxInst) {
|
||||
@@ -615,7 +624,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
if (HadDispatchError && TotalInstructions == 0) {
|
||||
// Couldn't handle any instruction in op dispatcher
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
return {nullptr, nullptr, 0, 0, 0, 0};
|
||||
return {nullptr, 0, 0, 0, 0};
|
||||
}
|
||||
|
||||
if (NeedsBlockEnd) {
|
||||
@@ -643,35 +652,26 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
auto ShouldDump = Thread->OpDispatcher->ShouldDumpIR();
|
||||
// Debug
|
||||
{
|
||||
if (ShouldDump) {
|
||||
IRDumper(Thread, IREmitter, GuestRIP, nullptr);
|
||||
}
|
||||
|
||||
if (static_cast<ContextImpl*>(Thread->CTX)->Config.ValidateIRarser) {
|
||||
ValidateIR(this, IREmitter);
|
||||
}
|
||||
if (ShouldDump) {
|
||||
IRDumper(Thread, IREmitter, GuestRIP, nullptr);
|
||||
}
|
||||
|
||||
// Run the passmanager over the IR from the dispatcher
|
||||
Thread->PassManager->Run(IREmitter);
|
||||
|
||||
// Debug
|
||||
{
|
||||
if (ShouldDump) {
|
||||
IRDumper(Thread, IREmitter, GuestRIP,
|
||||
Thread->PassManager->HasPass("RA") ? Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA")->GetAllocationData() : nullptr);
|
||||
}
|
||||
if (ShouldDump) {
|
||||
IRDumper(Thread, IREmitter, GuestRIP,
|
||||
Thread->PassManager->HasPass("RA") ? Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA")->GetAllocationData() : nullptr);
|
||||
}
|
||||
|
||||
auto RAData = Thread->PassManager->HasPass("RA") ? Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA")->PullAllocationData() : nullptr;
|
||||
auto IRList = IREmitter->CreateIRCopy();
|
||||
auto IRList = fextl::make_unique<IRListCopy>(IREmitter->ViewIR(), std::move(RAData));
|
||||
|
||||
IREmitter->DelayedDisownBuffer();
|
||||
|
||||
return {
|
||||
.IRList = IRList,
|
||||
.RAData = std::move(RAData),
|
||||
.IR = std::move(IRList),
|
||||
.TotalInstructions = TotalInstructions,
|
||||
.TotalInstructionsLength = TotalInstructionsLength,
|
||||
.StartAddr = Thread->FrontendDecoder->DecodedMinAddress,
|
||||
@@ -680,13 +680,6 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
}
|
||||
|
||||
ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst) {
|
||||
FEXCore::IR::IRListView* IRList {};
|
||||
FEXCore::Core::DebugData* DebugData {};
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData {};
|
||||
bool GeneratedIR {};
|
||||
uint64_t StartAddr {};
|
||||
uint64_t Length {};
|
||||
|
||||
// JIT Code object cache lookup
|
||||
if (CodeObjectCacheService) {
|
||||
auto CodeCacheEntry = CodeObjectCacheService->FetchCodeObjectFromCache(GuestRIP);
|
||||
@@ -695,9 +688,8 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
if (CompiledCode) {
|
||||
return {
|
||||
.CompiledCode = CompiledCode,
|
||||
.IRData = nullptr, // No IR data generated
|
||||
.IR = nullptr, // No IR/RA data generated
|
||||
.DebugData = nullptr, // nullptr here ensures that code serialization doesn't occur on from cache read
|
||||
.RAData = nullptr, // No RA data generated
|
||||
.GeneratedIR = false, // nullptr here ensures IR cache mechanisms won't run
|
||||
.StartAddr = 0, // Unused
|
||||
.Length = 0, // Unused
|
||||
@@ -713,49 +705,47 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::IRStorageBase> IR;
|
||||
FEXCore::Core::DebugData* DebugData {};
|
||||
uint64_t StartAddr {};
|
||||
uint64_t Length {};
|
||||
|
||||
// AOT IR bookkeeping and cache
|
||||
{
|
||||
auto [IRCopy, RACopy, DebugDataCopy, _StartAddr, _Length, _GeneratedIR] = IRCaptureCache.PreGenerateIRFetch(Thread, GuestRIP, IRList);
|
||||
if (_GeneratedIR) {
|
||||
auto IRFromAOT = IRCaptureCache.PreGenerateIRFetch(Thread, GuestRIP);
|
||||
if (IRFromAOT) {
|
||||
// Setup pointers to internal structures
|
||||
IRList = IRCopy;
|
||||
RAData = std::move(RACopy);
|
||||
DebugData = DebugDataCopy;
|
||||
StartAddr = _StartAddr;
|
||||
Length = _Length;
|
||||
GeneratedIR = _GeneratedIR;
|
||||
IR = std::move(IRFromAOT->IR);
|
||||
DebugData = IRFromAOT->DebugData;
|
||||
StartAddr = IRFromAOT->StartAddr;
|
||||
Length = IRFromAOT->Length;
|
||||
}
|
||||
}
|
||||
|
||||
if (IRList == nullptr) {
|
||||
if (!IR) {
|
||||
// Generate IR + Meta Info
|
||||
auto [IRCopy, RACopy, TotalInstructions, TotalInstructionsLength, _StartAddr, _Length] =
|
||||
GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst);
|
||||
auto [IRCopy, TotalInstructions, TotalInstructionsLength, _StartAddr, _Length] = GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst);
|
||||
|
||||
// Setup pointers to internal structures
|
||||
IRList = IRCopy;
|
||||
RAData = std::move(RACopy);
|
||||
IR = std::move(IRCopy);
|
||||
DebugData = new FEXCore::Core::DebugData();
|
||||
StartAddr = _StartAddr;
|
||||
Length = _Length;
|
||||
|
||||
// These blocks aren't already in the cache
|
||||
GeneratedIR = true;
|
||||
}
|
||||
|
||||
if (IRList == nullptr) {
|
||||
if (!IR) {
|
||||
return {};
|
||||
}
|
||||
// Attempt to get the CPU backend to compile this code
|
||||
auto IRView = IR->GetIRView();
|
||||
return {
|
||||
// FEX currently throws away the CPUBackend::CompiledCode object other than the entrypoint
|
||||
// In the future with code caching getting wired up, we will pass the rest of the data forward.
|
||||
// TODO: Pass the data forward when code caching is wired up to this.
|
||||
.CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, IRList, DebugData, RAData.get()).BlockEntry,
|
||||
.IRData = IRList,
|
||||
.CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, &IRView, DebugData, IR->RAData()).BlockEntry,
|
||||
.IR = std::move(IR),
|
||||
.DebugData = DebugData,
|
||||
.RAData = std::move(RAData),
|
||||
.GeneratedIR = GeneratedIR,
|
||||
.GeneratedIR = true,
|
||||
.StartAddr = StartAddr,
|
||||
.Length = Length,
|
||||
};
|
||||
@@ -782,21 +772,7 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
void* CodePtr {};
|
||||
FEXCore::IR::IRListView* IRList {};
|
||||
FEXCore::Core::DebugData* DebugData {};
|
||||
|
||||
bool GeneratedIR {};
|
||||
uint64_t StartAddr {}, Length {};
|
||||
|
||||
auto [Code, IR, Data, RAData, Generated, _StartAddr, _Length] = CompileCode(Thread, GuestRIP, MaxInst);
|
||||
CodePtr = Code;
|
||||
IRList = IR;
|
||||
DebugData = Data;
|
||||
GeneratedIR = Generated;
|
||||
StartAddr = _StartAddr;
|
||||
Length = _Length;
|
||||
|
||||
auto [CodePtr, IR, DebugData, GeneratedIR, StartAddr, Length] = CompileCode(Thread, GuestRIP, MaxInst);
|
||||
if (CodePtr == nullptr) {
|
||||
return 0;
|
||||
}
|
||||
@@ -847,7 +823,7 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
// Clear any relocations that might have been generated
|
||||
Thread->CPUBackend->ClearRelocations();
|
||||
|
||||
if (IRCaptureCache.PostCompileCode(Thread, CodePtr, GuestRIP, StartAddr, Length, std::move(RAData), IRList, DebugData, GeneratedIR)) {
|
||||
if (IRCaptureCache.PostCompileCode(Thread, CodePtr, GuestRIP, StartAddr, Length, std::move(IR), DebugData, GeneratedIR)) {
|
||||
// Early exit
|
||||
return (uintptr_t)CodePtr;
|
||||
}
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
@@ -17,6 +16,8 @@
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <condition_variable>
|
||||
#include <csignal>
|
||||
@@ -62,6 +63,11 @@ void Dispatcher::EmitDispatcher() {
|
||||
ARMEmitter::ForwardLabel l_CTX;
|
||||
ARMEmitter::SingleUseForwardLabel l_Sleep;
|
||||
#ifdef _M_ARM_64EC
|
||||
// These structures are not included in the standard Windows headers, define them here
|
||||
static constexpr size_t TEBCPUAreaOffset = 0x1788;
|
||||
static constexpr size_t CPUAreaInSyscallCallbackOffset = 0x1;
|
||||
static constexpr size_t CPUAreaEmulatorStackLimitOffset = 0x8;
|
||||
static constexpr size_t CPUAreaEmulatorDataOffset = 0x30;
|
||||
ARMEmitter::SingleUseForwardLabel ExitEC;
|
||||
#endif
|
||||
ARMEmitter::SingleUseForwardLabel l_CompileBlock;
|
||||
@@ -82,20 +88,45 @@ void Dispatcher::EmitDispatcher() {
|
||||
AbsoluteLoopTopAddressFillSRA = GetCursorAddress<uint64_t>();
|
||||
|
||||
FillStaticRegs();
|
||||
ARMEmitter::BiDirectionalLabel LoopTop {};
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
b(&LoopTop);
|
||||
|
||||
AbsoluteLoopTopAddressEnterECFillSRA = GetCursorAddress<uint64_t>();
|
||||
ldr(STATE, EC_ENTRY_CPUAREA_REG, CPUAreaEmulatorDataOffset);
|
||||
FillStaticRegs();
|
||||
|
||||
// Enter JIT
|
||||
b(&LoopTop);
|
||||
|
||||
AbsoluteLoopTopAddressEnterEC = GetCursorAddress<uint64_t>();
|
||||
// Load ThreadState and write the target PC there
|
||||
ldr(STATE, EC_ENTRY_CPUAREA_REG, CPUAreaEmulatorDataOffset);
|
||||
str(EC_CALL_CHECKER_PC_REG, STATE_PTR(CpuStateFrame, State.rip));
|
||||
|
||||
// Swap stacks to the emulator stack
|
||||
ldr(TMP1, EC_ENTRY_CPUAREA_REG, CPUAreaEmulatorStackLimitOffset);
|
||||
add(ARMEmitter::Size::i64Bit, StaticRegisters[X86State::REG_RSP], ARMEmitter::Reg::rsp, 0);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, TMP1, 0);
|
||||
|
||||
if (EmitterCTX->HostFeatures.SupportsSVE) {
|
||||
ptrue(ARMEmitter::SubRegSize::i8Bit, PRED_TMP_16B, ARMEmitter::PredicatePattern::SVE_VL16);
|
||||
}
|
||||
|
||||
// Enter JIT
|
||||
#endif
|
||||
|
||||
// We want to ensure that we are 16 byte aligned at the top of this loop
|
||||
Align16B();
|
||||
ARMEmitter::BiDirectionalLabel FullLookup {};
|
||||
ARMEmitter::BiDirectionalLabel CallBlock {};
|
||||
ARMEmitter::BackwardLabel LoopTop {};
|
||||
|
||||
Bind(&LoopTop);
|
||||
AbsoluteLoopTopAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
// Load in our RIP
|
||||
// Don't modify TMP3 since it contains our RIP once the block doesn't exist
|
||||
// IMPORTANT: Pointers.Common.ExitFunctionEC callsites/implementations need to be
|
||||
// adjusted accordingly if this changes.
|
||||
auto RipReg = TMP3;
|
||||
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
|
||||
|
||||
@@ -177,7 +208,8 @@ void Dispatcher::EmitDispatcher() {
|
||||
#ifdef _M_ARM_64EC
|
||||
{
|
||||
Bind(&ExitEC);
|
||||
// Target PC is already loaded into TMP3 at the start of the dispatcher
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, StaticRegisters[X86State::REG_RSP], 0);
|
||||
mov(EC_CALL_CHECKER_PC_REG, RipReg);
|
||||
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionEC));
|
||||
br(TMP2);
|
||||
}
|
||||
@@ -204,6 +236,12 @@ void Dispatcher::EmitDispatcher() {
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ldr(ARMEmitter::XReg::x0, ARMEmitter::XReg::x18, TEBCPUAreaOffset);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, 1);
|
||||
strb(ARMEmitter::WReg::w1, ARMEmitter::XReg::x0, CPUAreaInSyscallCallbackOffset);
|
||||
#endif
|
||||
|
||||
mov(ARMEmitter::XReg::x0, STATE);
|
||||
mov(ARMEmitter::XReg::x1, ARMEmitter::XReg::lr);
|
||||
|
||||
@@ -220,13 +258,18 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ldr(TMP2, ARMEmitter::XReg::x18, TEBCPUAreaOffset);
|
||||
strb(ARMEmitter::WReg::zr, TMP2, CPUAreaInSyscallCallbackOffset);
|
||||
#endif
|
||||
|
||||
ldr(TMP2, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, 1);
|
||||
str(TMP2, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
|
||||
// Trigger segfault if any deferred signals are pending
|
||||
ldr(TMP2, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress));
|
||||
str(ARMEmitter::XReg::zr, TMP2, 0);
|
||||
strb(ARMEmitter::XReg::zr, STATE,
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
|
||||
br(TMP1);
|
||||
}
|
||||
@@ -245,6 +288,12 @@ void Dispatcher::EmitDispatcher() {
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ldr(ARMEmitter::XReg::x0, ARMEmitter::XReg::x18, TEBCPUAreaOffset);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, 1);
|
||||
strb(ARMEmitter::WReg::w1, ARMEmitter::XReg::x0, CPUAreaInSyscallCallbackOffset);
|
||||
#endif
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, &l_CTX);
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
// x2 contains guest RIP
|
||||
@@ -259,13 +308,18 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ldr(TMP1, ARMEmitter::XReg::x18, TEBCPUAreaOffset);
|
||||
strb(ARMEmitter::WReg::zr, TMP1, CPUAreaInSyscallCallbackOffset);
|
||||
#endif
|
||||
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 1);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
|
||||
// Trigger segfault if any deferred signals are pending
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress));
|
||||
str(ARMEmitter::XReg::zr, TMP1, 0);
|
||||
strb(ARMEmitter::XReg::zr, STATE,
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
|
||||
b(&LoopTop);
|
||||
}
|
||||
@@ -494,6 +548,8 @@ void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState* Thread)
|
||||
|
||||
Common.DispatcherLoopTop = AbsoluteLoopTopAddress;
|
||||
Common.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
|
||||
Common.DispatcherLoopTopEnterEC = AbsoluteLoopTopAddressEnterEC;
|
||||
Common.DispatcherLoopTopEnterECFillSRA = AbsoluteLoopTopAddressEnterECFillSRA;
|
||||
Common.ExitFunctionLinker = ExitFunctionLinkerAddress;
|
||||
Common.ThreadStopHandlerSpillSRA = ThreadStopHandlerAddressSpillSRA;
|
||||
Common.ThreadPauseHandlerSpillSRA = ThreadPauseHandlerAddressSpillSRA;
|
||||
|
||||
@@ -47,6 +47,8 @@ public:
|
||||
uint64_t ThreadStopHandlerAddressSpillSRA {};
|
||||
uint64_t AbsoluteLoopTopAddress {};
|
||||
uint64_t AbsoluteLoopTopAddressFillSRA {};
|
||||
uint64_t AbsoluteLoopTopAddressEnterEC {};
|
||||
uint64_t AbsoluteLoopTopAddressEnterECFillSRA {};
|
||||
uint64_t ThreadPauseHandlerAddress {};
|
||||
uint64_t ThreadPauseHandlerAddressSpillSRA {};
|
||||
uint64_t ExitFunctionLinkerAddress {};
|
||||
|
||||
@@ -59,19 +59,19 @@ static void OverrideFeatures(HostFeatures* Features) {
|
||||
return;
|
||||
}
|
||||
|
||||
#define ENABLE_DISABLE_OPTION(FeatureName, name, enum_name) \
|
||||
do { \
|
||||
const bool Disable##name = (HostFeatures() & FEXCore::Config::HostFeatures::DISABLE##enum_name) != 0; \
|
||||
const bool Enable##name = (HostFeatures() & FEXCore::Config::HostFeatures::ENABLE##enum_name) != 0; \
|
||||
#define ENABLE_DISABLE_OPTION(FeatureName, name, enum_name) \
|
||||
do { \
|
||||
const bool Disable##name = (HostFeatures() & FEXCore::Config::HostFeatures::DISABLE##enum_name) != 0; \
|
||||
const bool Enable##name = (HostFeatures() & FEXCore::Config::HostFeatures::ENABLE##enum_name) != 0; \
|
||||
LogMan::Throw::AFmt(!(Disable##name && Enable##name), "Disabling and Enabling CPU feature (" #name ") is mutually exclusive"); \
|
||||
const bool AlreadyEnabled = Features->FeatureName; \
|
||||
const bool Result = (AlreadyEnabled | Enable##name) & !Disable##name; \
|
||||
Features->FeatureName = Result; \
|
||||
const bool AlreadyEnabled = Features->FeatureName; \
|
||||
const bool Result = (AlreadyEnabled | Enable##name) & !Disable##name; \
|
||||
Features->FeatureName = Result; \
|
||||
} while (0)
|
||||
|
||||
#define GET_SINGLE_OPTION(name, enum_name) \
|
||||
#define GET_SINGLE_OPTION(name, enum_name) \
|
||||
const bool Disable##name = (HostFeatures() & FEXCore::Config::HostFeatures::DISABLE##enum_name) != 0; \
|
||||
const bool Enable##name = (HostFeatures() & FEXCore::Config::HostFeatures::ENABLE##enum_name) != 0; \
|
||||
const bool Enable##name = (HostFeatures() & FEXCore::Config::HostFeatures::ENABLE##enum_name) != 0; \
|
||||
LogMan::Throw::AFmt(!(Disable##name && Enable##name), "Disabling and Enabling CPU feature (" #name ") is mutually exclusive");
|
||||
|
||||
ENABLE_DISABLE_OPTION(SupportsAVX, AVX, AVX);
|
||||
@@ -155,11 +155,6 @@ HostFeatures::HostFeatures() {
|
||||
SupportsBMI2 = true;
|
||||
SupportsCLWB = true;
|
||||
|
||||
// TODO: AFP is disabled until the scalar usage in the codebase can be audited to be working as expected.
|
||||
SupportsAFP = false;
|
||||
// RPRES has a dependency on AFP. Disable it until AFP is enabled.
|
||||
SupportsRPRES = false;
|
||||
|
||||
if (!SupportsAtomics) {
|
||||
WARN_ONCE_FMT("Host CPU doesn't support atomics. Expect bad performance");
|
||||
}
|
||||
|
||||
@@ -185,22 +185,22 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
break;
|
||||
}
|
||||
|
||||
#define COMMON_UNARY_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
#define COMMON_UNARY_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = {FABI_F80_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, SupportsPreserveAllABI}; \
|
||||
return true; \
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_BINARY_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
#define COMMON_BINARY_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = {FABI_F80_I16_F80_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, SupportsPreserveAllABI}; \
|
||||
return true; \
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
#define COMMON_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F64##OP>::handle, Core::OPINDEX_F64##OP); \
|
||||
return true; \
|
||||
return true; \
|
||||
}
|
||||
|
||||
// Unary
|
||||
|
||||
@@ -50,22 +50,22 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
|
||||
|
||||
// Bits are arranged as:
|
||||
// Bit #: 3 2 1 0
|
||||
// [OF | CF | SF | ZF]
|
||||
// [SF | ZF | CF | OF]
|
||||
uint32_t flags = 0;
|
||||
flags |= (valid_rhs < upper_limit) ? 0b01 : 0b00;
|
||||
flags |= (valid_lhs < upper_limit) ? 0b10 : 0b00;
|
||||
flags |= (valid_rhs < upper_limit) ? 0b0100 : 0b0000;
|
||||
flags |= (valid_lhs < upper_limit) ? 0b1000 : 0b0000;
|
||||
|
||||
const uint32_t result = HandlePolarity(aggregation, control, upper_limit, valid_rhs);
|
||||
if (result != 0) {
|
||||
flags |= 0b0100;
|
||||
flags |= 0b0010;
|
||||
}
|
||||
if ((result & 1) != 0) {
|
||||
flags |= 0b1000;
|
||||
flags |= 0b0001;
|
||||
}
|
||||
|
||||
// We tack the flags on top of the result to avoid needing to handle
|
||||
// multiple return values in the JITs.
|
||||
return result | (flags << 16);
|
||||
// We track the flags in the usual NZCV bit position so we can msr them
|
||||
// later. Avoids handling flags natively in JIT.
|
||||
return result | (flags << 28);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t GetExplicitLength(uint64_t reg, uint16_t control) {
|
||||
|
||||
@@ -7,8 +7,6 @@ $end_info$
|
||||
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
@@ -18,6 +16,31 @@ namespace FEXCore::CPU {
|
||||
#define GRS(Node) (IROp->Size <= 4 ? GetReg<RA_32>(Node) : GetReg<RA_64>(Node))
|
||||
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
|
||||
#define DEF_BINOP_WITH_CONSTANT(FEXOp, VarOp, ConstOp) \
|
||||
DEF_OP(FEXOp) { \
|
||||
auto Op = IROp->C<IR::IROp_##FEXOp>(); \
|
||||
\
|
||||
uint64_t Const; \
|
||||
if (IsInlineConstant(Op->Src2, &Const)) { \
|
||||
ConstOp(ConvertSize(IROp), GetReg(Node), GetReg(Op->Src1.ID()), Const); \
|
||||
} else { \
|
||||
VarOp(ConvertSize(IROp), GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2.ID())); \
|
||||
} \
|
||||
}
|
||||
|
||||
DEF_BINOP_WITH_CONSTANT(Add, add, add)
|
||||
DEF_BINOP_WITH_CONSTANT(Sub, sub, sub)
|
||||
DEF_BINOP_WITH_CONSTANT(AddWithFlags, adds, adds)
|
||||
DEF_BINOP_WITH_CONSTANT(SubWithFlags, subs, subs)
|
||||
DEF_BINOP_WITH_CONSTANT(Or, orr, orr)
|
||||
DEF_BINOP_WITH_CONSTANT(And, and_, and_)
|
||||
DEF_BINOP_WITH_CONSTANT(Andn, bic, bic)
|
||||
DEF_BINOP_WITH_CONSTANT(Xor, eor, eor)
|
||||
DEF_BINOP_WITH_CONSTANT(Lshl, lslv, lsl)
|
||||
DEF_BINOP_WITH_CONSTANT(Lshr, lsrv, lsr)
|
||||
DEF_BINOP_WITH_CONSTANT(Ror, rorv, ror)
|
||||
|
||||
DEF_OP(TruncElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_TruncElementPair>();
|
||||
|
||||
@@ -69,133 +92,71 @@ DEF_OP(CycleCounter) {
|
||||
#endif
|
||||
}
|
||||
|
||||
DEF_OP(Add) {
|
||||
auto Op = IROp->C<IR::IROp_Add>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
add(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), Const);
|
||||
} else {
|
||||
add(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AddWithFlags) {
|
||||
auto Op = IROp->C<IR::IROp_AddWithFlags>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
adds(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), Const);
|
||||
} else {
|
||||
adds(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AddShift) {
|
||||
auto Op = IROp->C<IR::IROp_AddShift>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
add(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
|
||||
add(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
|
||||
}
|
||||
|
||||
DEF_OP(AddNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_AddNZCV>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
LOGMAN_THROW_AA_FMT(OpSize >= 4, "Constant not allowed here");
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size >= 4, "Constant not allowed here");
|
||||
cmn(EmitSize, Src1, Const);
|
||||
} else {
|
||||
unsigned Shift = OpSize < 4 ? (32 - (8 * OpSize)) : 0;
|
||||
} else if (IROp->Size < 4) {
|
||||
unsigned Shift = 32 - (8 * IROp->Size);
|
||||
|
||||
if (OpSize < 4) {
|
||||
lsl(ARMEmitter::Size::i32Bit, TMP1, Src1, Shift);
|
||||
cmn(EmitSize, TMP1, GetReg(Op->Src2.ID()), ARMEmitter::ShiftType::LSL, Shift);
|
||||
} else {
|
||||
cmn(EmitSize, Src1, GetReg(Op->Src2.ID()));
|
||||
}
|
||||
lsl(ARMEmitter::Size::i32Bit, TMP1, Src1, Shift);
|
||||
cmn(EmitSize, TMP1, GetReg(Op->Src2.ID()), ARMEmitter::ShiftType::LSL, Shift);
|
||||
} else {
|
||||
cmn(EmitSize, Src1, GetReg(Op->Src2.ID()));
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AdcNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_AdcNZCV>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
adcs(EmitSize, ARMEmitter::Reg::zr, GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
adcs(ConvertSize48(IROp), ARMEmitter::Reg::zr, GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(AdcWithFlags) {
|
||||
auto Op = IROp->C<IR::IROp_AdcWithFlags>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
adcs(EmitSize, GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2.ID()));
|
||||
adcs(ConvertSize48(IROp), GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(Adc) {
|
||||
auto Op = IROp->C<IR::IROp_Adc>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
adc(EmitSize, GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2.ID()));
|
||||
adc(ConvertSize48(IROp), GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(SbbWithFlags) {
|
||||
auto Op = IROp->C<IR::IROp_SbbWithFlags>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
sbcs(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
sbcs(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(SbbNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_SbbNZCV>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
sbcs(EmitSize, ARMEmitter::Reg::zr, GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
sbcs(ConvertSize48(IROp), ARMEmitter::Reg::zr, GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(Sbb) {
|
||||
auto Op = IROp->C<IR::IROp_Sbb>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
sbc(EmitSize, GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2.ID()));
|
||||
sbc(ConvertSize48(IROp), GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(TestNZ) {
|
||||
auto Op = IROp->C<IR::IROp_TestNZ>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
uint64_t Const;
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
@@ -203,7 +164,7 @@ DEF_OP(TestNZ) {
|
||||
// Shift the sign bit into place, clearing out the garbage in upper bits.
|
||||
// Adding zero does an effective test, setting NZ according to the result and
|
||||
// zeroing CV.
|
||||
if (OpSize < 4) {
|
||||
if (IROp->Size < 4) {
|
||||
// Cheaper to and+cmn than to lsl+lsl+tst, so do the and ourselves if
|
||||
// needed.
|
||||
if (Op->Src1 != Op->Src2) {
|
||||
@@ -217,7 +178,7 @@ DEF_OP(TestNZ) {
|
||||
Src1 = TMP1;
|
||||
}
|
||||
|
||||
unsigned Shift = 32 - (OpSize * 8);
|
||||
unsigned Shift = 32 - (IROp->Size * 8);
|
||||
cmn(EmitSize, ARMEmitter::Reg::zr, Src1, ARMEmitter::ShiftType::LSL, Shift);
|
||||
} else {
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
@@ -229,51 +190,16 @@ DEF_OP(TestNZ) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Sub) {
|
||||
auto Op = IROp->C<IR::IROp_Sub>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
sub(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), Const);
|
||||
} else {
|
||||
sub(EmitSize, GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(SubShift) {
|
||||
auto Op = IROp->C<IR::IROp_SubShift>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
sub(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
|
||||
}
|
||||
|
||||
DEF_OP(SubWithFlags) {
|
||||
auto Op = IROp->C<IR::IROp_SubWithFlags>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
subs(EmitSize, GetReg(Node), GetZeroableReg(Op->Src1), Const);
|
||||
} else {
|
||||
subs(EmitSize, GetReg(Node), GetZeroableReg(Op->Src1), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
sub(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
|
||||
}
|
||||
|
||||
DEF_OP(SubNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_SubNZCV>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
@@ -300,9 +226,7 @@ DEF_OP(SubNZCV) {
|
||||
|
||||
DEF_OP(CmpPairZ) {
|
||||
auto Op = IROp->C<IR::IROp_CmpPairZ>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
// Save NZCV
|
||||
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
|
||||
@@ -354,100 +278,54 @@ DEF_OP(AXFlag) {
|
||||
axflag();
|
||||
}
|
||||
|
||||
ARMEmitter::Condition MapSelectCC(IR::CondClassType Cond) {
|
||||
switch (Cond.Val) {
|
||||
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
|
||||
case FEXCore::IR::COND_NEQ: return ARMEmitter::Condition::CC_NE;
|
||||
case FEXCore::IR::COND_SGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_SLT: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_SGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_SLE: return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_UGE: return ARMEmitter::Condition::CC_CS;
|
||||
case FEXCore::IR::COND_ULT: return ARMEmitter::Condition::CC_CC;
|
||||
case FEXCore::IR::COND_UGT: return ARMEmitter::Condition::CC_HI;
|
||||
case FEXCore::IR::COND_ULE: return ARMEmitter::Condition::CC_LS;
|
||||
case FEXCore::IR::COND_FLU: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_FGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_FLEU: return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_FGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_FU: return ARMEmitter::Condition::CC_VS;
|
||||
case FEXCore::IR::COND_FNU: return ARMEmitter::Condition::CC_VC;
|
||||
case FEXCore::IR::COND_VS:
|
||||
case FEXCore::IR::COND_VC:
|
||||
case FEXCore::IR::COND_MI: return ARMEmitter::Condition::CC_MI;
|
||||
case FEXCore::IR::COND_PL: return ARMEmitter::Condition::CC_PL;
|
||||
default: LOGMAN_MSG_A_FMT("Unsupported compare type"); return ARMEmitter::Condition::CC_NV;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(CondAddNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_CondAddNZCV>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
ARMEmitter::StatusFlags Flags = (ARMEmitter::StatusFlags)Op->FalseNZCV;
|
||||
uint64_t Const = 0;
|
||||
auto Src1 = GetZeroableReg(Op->Src1);
|
||||
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
ccmn(EmitSize, Src1, Const, Flags, MapSelectCC(Op->Cond));
|
||||
ccmn(ConvertSize48(IROp), Src1, Const, Flags, MapCC(Op->Cond));
|
||||
} else {
|
||||
ccmn(EmitSize, Src1, GetReg(Op->Src2.ID()), Flags, MapSelectCC(Op->Cond));
|
||||
ccmn(ConvertSize48(IROp), Src1, GetReg(Op->Src2.ID()), Flags, MapCC(Op->Cond));
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(CondSubNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_CondSubNZCV>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == IR::i32Bit || OpSize == IR::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == IR::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
ARMEmitter::StatusFlags Flags = (ARMEmitter::StatusFlags)Op->FalseNZCV;
|
||||
uint64_t Const = 0;
|
||||
auto Src1 = GetZeroableReg(Op->Src1);
|
||||
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
ccmp(EmitSize, Src1, Const, Flags, MapSelectCC(Op->Cond));
|
||||
ccmp(ConvertSize48(IROp), Src1, Const, Flags, MapCC(Op->Cond));
|
||||
} else {
|
||||
ccmp(EmitSize, Src1, GetReg(Op->Src2.ID()), Flags, MapSelectCC(Op->Cond));
|
||||
ccmp(ConvertSize48(IROp), Src1, GetReg(Op->Src2.ID()), Flags, MapCC(Op->Cond));
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Neg) {
|
||||
auto Op = IROp->C<IR::IROp_Neg>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
if (Op->Cond == FEXCore::IR::COND_AL) {
|
||||
neg(EmitSize, GetReg(Node), GetReg(Op->Src.ID()));
|
||||
neg(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src.ID()));
|
||||
} else {
|
||||
cneg(EmitSize, GetReg(Node), GetReg(Op->Src.ID()), MapSelectCC(Op->Cond));
|
||||
cneg(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src.ID()), MapCC(Op->Cond));
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Mul) {
|
||||
auto Op = IROp->C<IR::IROp_Mul>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
mul(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
mul(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(UMul) {
|
||||
auto Op = IROp->C<IR::IROp_UMul>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
mul(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
mul(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(UMull) {
|
||||
@@ -466,13 +344,12 @@ DEF_OP(Div) {
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
auto Src2 = GetReg(Op->Src2.ID());
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
if (OpSize == 1) {
|
||||
sxtb(EmitSize, TMP1, Src1);
|
||||
sxtb(EmitSize, TMP2, Src2);
|
||||
@@ -496,13 +373,12 @@ DEF_OP(UDiv) {
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
auto Src2 = GetReg(Op->Src2.ID());
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
if (OpSize == 1) {
|
||||
uxtb(EmitSize, TMP1, Src1);
|
||||
uxtb(EmitSize, TMP2, Src2);
|
||||
@@ -525,13 +401,12 @@ DEF_OP(Rem) {
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
auto Src2 = GetReg(Op->Src2.ID());
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
if (OpSize == 1) {
|
||||
sxtb(EmitSize, TMP1, Src1);
|
||||
sxtb(EmitSize, TMP2, Src2);
|
||||
@@ -555,12 +430,12 @@ DEF_OP(URem) {
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
auto Src1 = GetReg(Op->Src1.ID());
|
||||
auto Src2 = GetReg(Op->Src2.ID());
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
if (OpSize == 1) {
|
||||
uxtb(EmitSize, TMP1, Src1);
|
||||
uxtb(EmitSize, TMP2, Src2);
|
||||
@@ -619,90 +494,49 @@ DEF_OP(UMulH) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Or) {
|
||||
auto Op = IROp->C<IR::IROp_Or>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
orr(EmitSize, Dst, Src1, Const);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
orr(EmitSize, Dst, Src1, Src2);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Orlshl) {
|
||||
auto Op = IROp->C<IR::IROp_Orlshl>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
orr(EmitSize, Dst, Src1, Const << Op->BitShift);
|
||||
orr(ConvertSize(IROp), Dst, Src1, Const << Op->BitShift);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
orr(EmitSize, Dst, Src1, Src2, ARMEmitter::ShiftType::LSL, Op->BitShift);
|
||||
orr(ConvertSize(IROp), Dst, Src1, Src2, ARMEmitter::ShiftType::LSL, Op->BitShift);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Orlshr) {
|
||||
auto Op = IROp->C<IR::IROp_Orlshr>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
orr(EmitSize, Dst, Src1, Const >> Op->BitShift);
|
||||
orr(ConvertSize(IROp), Dst, Src1, Const >> Op->BitShift);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
orr(EmitSize, Dst, Src1, Src2, ARMEmitter::ShiftType::LSR, Op->BitShift);
|
||||
orr(ConvertSize(IROp), Dst, Src1, Src2, ARMEmitter::ShiftType::LSR, Op->BitShift);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Ornror) {
|
||||
auto Op = IROp->C<IR::IROp_Ornror>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
orn(EmitSize, Dst, Src1, Src2, ARMEmitter::ShiftType::ROR, Op->BitShift);
|
||||
}
|
||||
|
||||
DEF_OP(And) {
|
||||
auto Op = IROp->C<IR::IROp_And>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
and_(EmitSize, Dst, Src1, Const);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
and_(EmitSize, Dst, Src1, Src2);
|
||||
}
|
||||
orn(ConvertSize(IROp), Dst, Src1, Src2, ARMEmitter::ShiftType::ROR, Op->BitShift);
|
||||
}
|
||||
|
||||
DEF_OP(AndWithFlags) {
|
||||
auto Op = IROp->C<IR::IROp_AndWithFlags>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
uint64_t Const;
|
||||
const auto Dst = GetReg(Node);
|
||||
@@ -734,99 +568,22 @@ DEF_OP(AndWithFlags) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Andn) {
|
||||
auto Op = IROp->C<IR::IROp_Andn>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
bic(EmitSize, Dst, Src1, Const);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
bic(EmitSize, Dst, Src1, Src2);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Xor) {
|
||||
auto Op = IROp->C<IR::IROp_Xor>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
eor(EmitSize, Dst, Src1, Const);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
eor(EmitSize, Dst, Src1, Src2);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(XorShift) {
|
||||
auto Op = IROp->C<IR::IROp_XorShift>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
eor(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
|
||||
eor(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
|
||||
}
|
||||
|
||||
DEF_OP(XornShift) {
|
||||
auto Op = IROp->C<IR::IROp_XornShift>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
eon(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
|
||||
}
|
||||
|
||||
DEF_OP(Lshl) {
|
||||
auto Op = IROp->C<IR::IROp_Lshl>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
lsl(EmitSize, Dst, Src1, Const);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
lslv(EmitSize, Dst, Src1, Src2);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Lshr) {
|
||||
auto Op = IROp->C<IR::IROp_Lshr>();
|
||||
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
lsr(EmitSize, Dst, Src1, Const);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
lsrv(EmitSize, Dst, Src1, Src2);
|
||||
}
|
||||
eon(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
|
||||
}
|
||||
|
||||
DEF_OP(Ashr) {
|
||||
auto Op = IROp->C<IR::IROp_Ashr>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
@@ -932,45 +689,18 @@ DEF_OP(ShiftFlags) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Ror) {
|
||||
auto Op = IROp->C<IR::IROp_Ror>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
ror(EmitSize, Dst, Src1, Const);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
rorv(EmitSize, Dst, Src1, Src2);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Extr) {
|
||||
auto Op = IROp->C<IR::IROp_Extr>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Upper = GetReg(Op->Upper.ID());
|
||||
const auto Lower = GetReg(Op->Lower.ID());
|
||||
|
||||
extr(EmitSize, Dst, Upper, Lower, Op->LSB);
|
||||
extr(ConvertSize48(IROp), Dst, Upper, Lower, Op->LSB);
|
||||
}
|
||||
|
||||
DEF_OP(PDep) {
|
||||
auto Op = IROp->C<IR::IROp_PExt>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSize = ConvertSize48(IROp);
|
||||
|
||||
const auto Dest = GetReg(Node);
|
||||
|
||||
@@ -1033,9 +763,7 @@ DEF_OP(PExt) {
|
||||
auto Op = IROp->C<IR::IROp_PExt>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto OpSizeBitsM1 = (OpSize * 8) - 1;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSize = ConvertSize48(IROp);
|
||||
|
||||
const auto Input = GetReg(Op->Input.ID());
|
||||
const auto Mask = GetReg(Op->Mask.ID());
|
||||
@@ -1351,15 +1079,11 @@ DEF_OP(LURem) {
|
||||
|
||||
DEF_OP(Not) {
|
||||
auto Op = IROp->C<IR::IROp_Not>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
mvn(EmitSize, Dst, Src);
|
||||
mvn(ConvertSize48(IROp), Dst, Src);
|
||||
}
|
||||
|
||||
DEF_OP(Popcount) {
|
||||
@@ -1373,23 +1097,23 @@ DEF_OP(Popcount) {
|
||||
case 0x1:
|
||||
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), Src);
|
||||
// only use lowest byte
|
||||
cnt(FEXCore::ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
|
||||
cnt(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
|
||||
break;
|
||||
case 0x2:
|
||||
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), Src);
|
||||
cnt(FEXCore::ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
|
||||
cnt(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
|
||||
// only count two lowest bytes
|
||||
addp(FEXCore::ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D(), VTMP1.D());
|
||||
addp(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D(), VTMP1.D());
|
||||
break;
|
||||
case 0x4:
|
||||
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), Src);
|
||||
cnt(FEXCore::ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
|
||||
cnt(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
|
||||
// fmov has zero extended, unused bytes are zero
|
||||
addv(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
|
||||
break;
|
||||
case 0x8:
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), Src);
|
||||
cnt(FEXCore::ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
|
||||
cnt(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
|
||||
// fmov has zero extended, unused bytes are zero
|
||||
addv(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
|
||||
break;
|
||||
@@ -1401,15 +1125,13 @@ DEF_OP(Popcount) {
|
||||
|
||||
DEF_OP(FindLSB) {
|
||||
auto Op = IROp->C<IR::IROp_FindLSB>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
if (OpSize != 8) {
|
||||
ubfx(EmitSize, TMP1, Src, 0, OpSize * 8);
|
||||
if (IROp->Size != 8) {
|
||||
ubfx(EmitSize, TMP1, Src, 0, IROp->Size * 8);
|
||||
cmp(EmitSize, TMP1, 0);
|
||||
rbit(EmitSize, TMP1, TMP1);
|
||||
} else {
|
||||
@@ -1426,7 +1148,7 @@ DEF_OP(FindMSB) {
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 2 || OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
@@ -1449,7 +1171,7 @@ DEF_OP(FindTrailingZeroes) {
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 2 || OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
@@ -1473,7 +1195,7 @@ DEF_OP(CountLeadingZeroes) {
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 2 || OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
@@ -1494,7 +1216,7 @@ DEF_OP(Rev) {
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 2 || OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
@@ -1507,9 +1229,7 @@ DEF_OP(Rev) {
|
||||
|
||||
DEF_OP(Bfi) {
|
||||
auto Op = IROp->C<IR::IROp_Bfi>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto SrcDst = GetReg(Op->Dest.ID());
|
||||
@@ -1528,19 +1248,17 @@ DEF_OP(Bfi) {
|
||||
mov(EmitSize, TMP1, SrcDst);
|
||||
bfi(EmitSize, TMP1, Src, Op->lsb, Op->Width);
|
||||
|
||||
if (OpSize >= 4) {
|
||||
if (IROp->Size >= 4) {
|
||||
mov(EmitSize, Dst, TMP1.R());
|
||||
} else {
|
||||
ubfx(EmitSize, Dst, TMP1, 0, OpSize * 8);
|
||||
ubfx(EmitSize, Dst, TMP1, 0, IROp->Size * 8);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Bfxil) {
|
||||
auto Op = IROp->C<IR::IROp_Bfxil>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto SrcDst = GetReg(Op->Dest.ID());
|
||||
@@ -1566,8 +1284,7 @@ DEF_OP(Bfe) {
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size <= 8, "OpSize is too large for BFE: {}", IROp->Size);
|
||||
LOGMAN_THROW_AA_FMT(Op->Width != 0, "Invalid BFE width of 0");
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
@@ -1575,7 +1292,7 @@ DEF_OP(Bfe) {
|
||||
if (Op->lsb == 0 && Op->Width == 32) {
|
||||
mov(ARMEmitter::Size::i32Bit, Dst, Src);
|
||||
} else if (Op->lsb == 0 && Op->Width == 64) {
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 8, "Must be 64-bit wide register");
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size == 8, "Must be 64-bit wide register");
|
||||
mov(ARMEmitter::Size::i64Bit, Dst, Src);
|
||||
} else {
|
||||
ubfx(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
@@ -1584,23 +1301,20 @@ DEF_OP(Bfe) {
|
||||
|
||||
DEF_OP(Sbfe) {
|
||||
auto Op = IROp->C<IR::IROp_Sbfe>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
sbfx(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
sbfx(ConvertSize(IROp), Dst, Src, Op->lsb, Op->Width);
|
||||
}
|
||||
|
||||
DEF_OP(Select) {
|
||||
auto Op = IROp->C<IR::IROp_Select>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto CompareEmitSize = Op->CompareSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
uint64_t Const;
|
||||
auto cc = MapSelectCC(Op->Cond);
|
||||
auto cc = MapCC(Op->Cond);
|
||||
|
||||
if (IsGPR(Op->Cmp1.ID())) {
|
||||
const auto Src1 = GetReg(Op->Cmp1.ID());
|
||||
@@ -1649,16 +1363,15 @@ DEF_OP(Select) {
|
||||
|
||||
DEF_OP(NZCVSelect) {
|
||||
auto Op = IROp->C<IR::IROp_NZCVSelect>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
|
||||
auto cc = MapSelectCC(Op->Cond);
|
||||
auto cc = MapCC(Op->Cond);
|
||||
|
||||
uint64_t const_true, const_false;
|
||||
bool is_const_true = IsInlineConstant(Op->TrueVal, &const_true);
|
||||
bool is_const_false = IsInlineConstant(Op->FalseVal, &const_false);
|
||||
|
||||
uint64_t all_ones = OpSize == 8 ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
uint64_t all_ones = IROp->Size == 8 ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
|
||||
ARMEmitter::Register Dst = GetReg(Node);
|
||||
|
||||
@@ -1740,12 +1453,11 @@ DEF_OP(Float_ToGPR_ZS) {
|
||||
|
||||
ARMEmitter::Register Dst = GetReg(Node);
|
||||
ARMEmitter::VRegister Src = GetVReg(Op->Scalar.ID());
|
||||
const auto DestSize = IROp->Size == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
if (Op->SrcElementSize == 8) {
|
||||
fcvtzs(DestSize, Dst, Src.D());
|
||||
fcvtzs(ConvertSize(IROp), Dst, Src.D());
|
||||
} else {
|
||||
fcvtzs(DestSize, Dst, Src.S());
|
||||
fcvtzs(ConvertSize(IROp), Dst, Src.S());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1754,14 +1466,13 @@ DEF_OP(Float_ToGPR_S) {
|
||||
|
||||
ARMEmitter::Register Dst = GetReg(Node);
|
||||
ARMEmitter::VRegister Src = GetVReg(Op->Scalar.ID());
|
||||
const auto DestSize = IROp->Size == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
if (Op->SrcElementSize == 8) {
|
||||
frinti(VTMP1.D(), Src.D());
|
||||
fcvtzs(DestSize, Dst, VTMP1.D());
|
||||
fcvtzs(ConvertSize(IROp), Dst, VTMP1.D());
|
||||
} else {
|
||||
frinti(VTMP1.S(), Src.S());
|
||||
fcvtzs(DestSize, Dst, VTMP1.S());
|
||||
fcvtzs(ConvertSize(IROp), Dst, VTMP1.S());
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -6,7 +6,6 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
|
||||
@@ -69,8 +68,8 @@ DEF_OP(CASPair) {
|
||||
|
||||
DEF_OP(CAS) {
|
||||
auto Op = IROp->C<IR::IROp_CAS>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
// DataSrc = *Src1
|
||||
// if (DataSrc == Src3) { *Src1 == Src2; } Src2 = DataSrc
|
||||
// This will write to memory! Careful!
|
||||
@@ -79,13 +78,6 @@ DEF_OP(CAS) {
|
||||
auto Desired = GetReg(Op->Desired.ID());
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mov(EmitSize, TMP2, Expected);
|
||||
casal(SubEmitSize, TMP2, Desired, MemSrc);
|
||||
@@ -96,9 +88,9 @@ DEF_OP(CAS) {
|
||||
ARMEmitter::SingleUseForwardLabel LoopExpected;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
if (OpSize == 1) {
|
||||
if (IROp->Size == 1) {
|
||||
cmp(EmitSize, TMP2, Expected, ARMEmitter::ExtendedType::UXTB, 0);
|
||||
} else if (OpSize == 2) {
|
||||
} else if (IROp->Size == 2) {
|
||||
cmp(EmitSize, TMP2, Expected, ARMEmitter::ExtendedType::UXTH, 0);
|
||||
} else {
|
||||
cmp(EmitSize, TMP2, Expected);
|
||||
@@ -120,19 +112,12 @@ DEF_OP(CAS) {
|
||||
|
||||
DEF_OP(AtomicAdd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicAdd>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
staddl(SubEmitSize, Src, MemSrc);
|
||||
} else {
|
||||
@@ -147,19 +132,12 @@ DEF_OP(AtomicAdd) {
|
||||
|
||||
DEF_OP(AtomicSub) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicSub>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
neg(EmitSize, TMP2, Src);
|
||||
staddl(SubEmitSize, TMP2, MemSrc);
|
||||
@@ -175,19 +153,12 @@ DEF_OP(AtomicSub) {
|
||||
|
||||
DEF_OP(AtomicAnd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicAnd>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mvn(EmitSize, TMP2, Src);
|
||||
stclrl(SubEmitSize, TMP2, MemSrc);
|
||||
@@ -203,19 +174,12 @@ DEF_OP(AtomicAnd) {
|
||||
|
||||
DEF_OP(AtomicCLR) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicCLR>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
stclrl(SubEmitSize, Src, MemSrc);
|
||||
} else {
|
||||
@@ -230,19 +194,12 @@ DEF_OP(AtomicCLR) {
|
||||
|
||||
DEF_OP(AtomicOr) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicOr>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
stsetl(SubEmitSize, Src, MemSrc);
|
||||
} else {
|
||||
@@ -257,19 +214,12 @@ DEF_OP(AtomicOr) {
|
||||
|
||||
DEF_OP(AtomicXor) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicXor>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
steorl(SubEmitSize, Src, MemSrc);
|
||||
} else {
|
||||
@@ -284,18 +234,11 @@ DEF_OP(AtomicXor) {
|
||||
|
||||
DEF_OP(AtomicNeg) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicNeg>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
@@ -312,7 +255,7 @@ DEF_OP(AtomicSwap) {
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
@@ -333,19 +276,12 @@ DEF_OP(AtomicSwap) {
|
||||
|
||||
DEF_OP(AtomicFetchAdd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchAdd>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
ldaddal(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
} else {
|
||||
@@ -361,19 +297,12 @@ DEF_OP(AtomicFetchAdd) {
|
||||
|
||||
DEF_OP(AtomicFetchSub) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchSub>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
neg(EmitSize, TMP2, Src);
|
||||
ldaddal(SubEmitSize, TMP2, GetReg(Node), MemSrc);
|
||||
@@ -390,19 +319,12 @@ DEF_OP(AtomicFetchSub) {
|
||||
|
||||
DEF_OP(AtomicFetchAnd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchAnd>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mvn(EmitSize, TMP2, Src);
|
||||
ldclral(SubEmitSize, TMP2, GetReg(Node), MemSrc);
|
||||
@@ -419,19 +341,12 @@ DEF_OP(AtomicFetchAnd) {
|
||||
|
||||
DEF_OP(AtomicFetchCLR) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchCLR>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
ldclral(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
} else {
|
||||
@@ -447,19 +362,12 @@ DEF_OP(AtomicFetchCLR) {
|
||||
|
||||
DEF_OP(AtomicFetchOr) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchOr>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
ldsetal(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
} else {
|
||||
@@ -475,19 +383,12 @@ DEF_OP(AtomicFetchOr) {
|
||||
|
||||
DEF_OP(AtomicFetchXor) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchXor>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
ldeoral(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
} else {
|
||||
@@ -503,18 +404,11 @@ DEF_OP(AtomicFetchXor) {
|
||||
|
||||
DEF_OP(AtomicFetchNeg) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchNeg>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
|
||||
@@ -7,7 +7,6 @@ $end_info$
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
@@ -55,7 +54,8 @@ DEF_OP(ExitFunction) {
|
||||
if (IsInlineConstant(Op->NewRIP, &NewRIP) || IsInlineEntrypointOffset(Op->NewRIP, &NewRIP)) {
|
||||
#ifdef _M_ARM_64EC
|
||||
if (RtlIsEcCode(NewRIP)) {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP3, NewRIP);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, StaticRegisters[X86State::REG_RSP], 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, EC_CALL_CHECKER_PC_REG, NewRIP);
|
||||
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionEC));
|
||||
br(TMP2);
|
||||
} else {
|
||||
@@ -101,39 +101,13 @@ DEF_OP(Jump) {
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(Target).first->second;
|
||||
}
|
||||
|
||||
static ARMEmitter::Condition MapBranchCC(IR::CondClassType Cond) {
|
||||
switch (Cond.Val) {
|
||||
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
|
||||
case FEXCore::IR::COND_NEQ: return ARMEmitter::Condition::CC_NE;
|
||||
case FEXCore::IR::COND_SGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_SLT: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_SGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_SLE: return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_UGE: return ARMEmitter::Condition::CC_CS;
|
||||
case FEXCore::IR::COND_ULT: return ARMEmitter::Condition::CC_CC;
|
||||
case FEXCore::IR::COND_UGT: return ARMEmitter::Condition::CC_HI;
|
||||
case FEXCore::IR::COND_ULE: return ARMEmitter::Condition::CC_LS;
|
||||
case FEXCore::IR::COND_FLU: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_FGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_FLEU: return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_FGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_FU: return ARMEmitter::Condition::CC_VS;
|
||||
case FEXCore::IR::COND_FNU: return ARMEmitter::Condition::CC_VC;
|
||||
case FEXCore::IR::COND_VS:
|
||||
case FEXCore::IR::COND_VC:
|
||||
case FEXCore::IR::COND_MI: return ARMEmitter::Condition::CC_MI;
|
||||
case FEXCore::IR::COND_PL: return ARMEmitter::Condition::CC_PL;
|
||||
default: LOGMAN_MSG_A_FMT("Unsupported compare type"); return ARMEmitter::Condition::CC_NV;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(CondJump) {
|
||||
auto Op = IROp->C<IR::IROp_CondJump>();
|
||||
|
||||
auto TrueTargetLabel = &JumpTargets.try_emplace(Op->TrueBlock.ID()).first->second;
|
||||
|
||||
if (Op->FromNZCV) {
|
||||
b(MapBranchCC(Op->Cond), TrueTargetLabel);
|
||||
b(MapCC(Op->Cond), TrueTargetLabel);
|
||||
} else {
|
||||
[[maybe_unused]] uint64_t Const;
|
||||
[[maybe_unused]] const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
|
||||
|
||||
@@ -5,7 +5,6 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
@@ -18,12 +17,7 @@ DEF_OP(VInsGPR) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2 || ElementSize == 1, "Unexpected {} size", __func__);
|
||||
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp);
|
||||
const auto ElementsPer128Bit = 16 / ElementSize;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -66,7 +60,7 @@ DEF_OP(VInsGPR) {
|
||||
// Inserts the GPR value into the given V register.
|
||||
// Also automatically adjusts the index in the case of using the
|
||||
// moved upper lane.
|
||||
const auto Insert = [&](const FEXCore::ARMEmitter::VRegister& reg, int index) {
|
||||
const auto Insert = [&](const ARMEmitter::VRegister& reg, int index) {
|
||||
if (InUpperLane) {
|
||||
index -= ElementsPer128Bit;
|
||||
}
|
||||
@@ -117,16 +111,7 @@ DEF_OP(VDupFromGPR) {
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2 || ElementSize == 1, "Unexpected {} element size: {}",
|
||||
__func__, ElementSize);
|
||||
|
||||
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
dup(SubEmitSize, Dst.Z(), Src);
|
||||
@@ -216,14 +201,9 @@ DEF_OP(Vector_SToF) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2, "Unexpected {} size", __func__);
|
||||
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ARMEmitter::SubRegSize::i16Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
@@ -253,14 +233,9 @@ DEF_OP(Vector_FToZS) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2, "Unexpected {} size", __func__);
|
||||
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ARMEmitter::SubRegSize::i16Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
@@ -289,14 +264,8 @@ DEF_OP(Vector_FToS) {
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2, "Unexpected {} size", __func__);
|
||||
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ARMEmitter::SubRegSize::i16Bit;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -323,15 +292,10 @@ DEF_OP(Vector_FToF) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2, "Unexpected {} size", __func__);
|
||||
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ARMEmitter::SubRegSize::i16Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
|
||||
@@ -353,23 +317,23 @@ DEF_OP(Vector_FToF) {
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0402: { // Float <- Half
|
||||
zip1(FEXCore::ARMEmitter::SubRegSize::i16Bit, Dst.Z(), Vector.Z(), Vector.Z());
|
||||
fcvtlt(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Mask, Dst.Z());
|
||||
zip1(ARMEmitter::SubRegSize::i16Bit, Dst.Z(), Vector.Z(), Vector.Z());
|
||||
fcvtlt(ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Mask, Dst.Z());
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- Float
|
||||
zip1(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Vector.Z(), Vector.Z());
|
||||
fcvtlt(FEXCore::ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Dst.Z());
|
||||
zip1(ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Vector.Z(), Vector.Z());
|
||||
fcvtlt(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Dst.Z());
|
||||
break;
|
||||
}
|
||||
case 0x0204: { // Half <- Float
|
||||
fcvtnt(FEXCore::ARMEmitter::SubRegSize::i16Bit, Dst.Z(), Mask, Vector.Z());
|
||||
uzp2(FEXCore::ARMEmitter::SubRegSize::i16Bit, Dst.Z(), Dst.Z(), Dst.Z());
|
||||
fcvtnt(ARMEmitter::SubRegSize::i16Bit, Dst.Z(), Mask, Vector.Z());
|
||||
uzp2(ARMEmitter::SubRegSize::i16Bit, Dst.Z(), Dst.Z(), Dst.Z());
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvtnt(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Mask, Vector.Z());
|
||||
uzp2(FEXCore::ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Dst.Z(), Dst.Z());
|
||||
fcvtnt(ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Mask, Vector.Z());
|
||||
uzp2(ARMEmitter::SubRegSize::i32Bit, Dst.Z(), Dst.Z(), Dst.Z());
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv); break;
|
||||
@@ -396,13 +360,8 @@ DEF_OP(Vector_FToI) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 8 || ElementSize == 4 || ElementSize == 2, "Unexpected {} size", __func__);
|
||||
|
||||
const auto SubEmitSize = ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ARMEmitter::SubRegSize::i16Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -425,15 +384,15 @@ DEF_OP(Vector_FToI) {
|
||||
// frinti having AdvSIMD, AdvSIMD scalar, and an SVE version),
|
||||
// we can't just use a lambda without some seriously ugly casting.
|
||||
// This is fairly self-contained otherwise.
|
||||
#define ROUNDING_FN(name) \
|
||||
if (ElementSize == 2) { \
|
||||
name(Dst.H(), Vector.H()); \
|
||||
#define ROUNDING_FN(name) \
|
||||
if (ElementSize == 2) { \
|
||||
name(Dst.H(), Vector.H()); \
|
||||
} else if (ElementSize == 4) { \
|
||||
name(Dst.S(), Vector.S()); \
|
||||
name(Dst.S(), Vector.S()); \
|
||||
} else if (ElementSize == 8) { \
|
||||
name(Dst.D(), Vector.D()); \
|
||||
} else { \
|
||||
FEX_UNREACHABLE; \
|
||||
name(Dst.D(), Vector.D()); \
|
||||
} else { \
|
||||
FEX_UNREACHABLE; \
|
||||
}
|
||||
|
||||
switch (Op->Round) {
|
||||
|
||||
@@ -5,9 +5,7 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
|
||||
@@ -13,7 +13,6 @@ $end_info$
|
||||
|
||||
#include "FEXCore/Utils/Telemetry.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
@@ -469,13 +468,13 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
static void DirectBlockDelinker(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
|
||||
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
|
||||
uintptr_t branch = (uintptr_t)(Record)-8;
|
||||
FEXCore::ARMEmitter::Emitter emit((uint8_t*)(branch), 8);
|
||||
FEXCore::ARMEmitter::SingleUseForwardLabel l_BranchHost;
|
||||
ARMEmitter::Emitter emit((uint8_t*)(branch), 8);
|
||||
ARMEmitter::SingleUseForwardLabel l_BranchHost;
|
||||
emit.ldr(TMP1, &l_BranchHost);
|
||||
emit.blr(TMP1);
|
||||
emit.Bind(&l_BranchHost);
|
||||
emit.dc64(LinkerAddress);
|
||||
FEXCore::ARMEmitter::Emitter::ClearICache((void*)branch, 8);
|
||||
ARMEmitter::Emitter::ClearICache((void*)branch, 8);
|
||||
}
|
||||
|
||||
static void IndirectBlockDelinker(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
|
||||
@@ -500,9 +499,9 @@ static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame* Fram
|
||||
if (vixl::IsInt26(offset)) {
|
||||
// optimal case - can branch directly
|
||||
// patch the code
|
||||
FEXCore::ARMEmitter::Emitter emit((uint8_t*)(branch), 4);
|
||||
ARMEmitter::Emitter emit((uint8_t*)(branch), 4);
|
||||
emit.b(offset);
|
||||
FEXCore::ARMEmitter::Emitter::ClearICache((void*)branch, 4);
|
||||
ARMEmitter::Emitter::ClearICache((void*)branch, 4);
|
||||
|
||||
// Add de-linking handler
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, Record, DirectBlockDelinker);
|
||||
@@ -530,19 +529,11 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
|
||||
|
||||
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
|
||||
RAPass->AllocateRegisterSet(RegisterClasses);
|
||||
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRClass, GeneralRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRFixedClass, StaticRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRClass, GeneralFPRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRFixedClass, StaticFPRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRPairClass, GeneralPairRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::ComplexClass, 1);
|
||||
|
||||
for (uint32_t i = 0; i < GeneralPairRegisters.size(); ++i) {
|
||||
RAPass->AddRegisterConflict(FEXCore::IR::GPRClass, i * 2, FEXCore::IR::GPRPairClass, i);
|
||||
RAPass->AddRegisterConflict(FEXCore::IR::GPRClass, i * 2 + 1, FEXCore::IR::GPRPairClass, i);
|
||||
}
|
||||
RAPass->PairRegs = PairRegisters;
|
||||
|
||||
{
|
||||
// Set up pointers that the JIT needs to load
|
||||
@@ -669,7 +660,7 @@ bool Arm64JITCore::IsGPRPair(IR::NodeID Node) const {
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData,
|
||||
FEXCore::IR::RegisterAllocationData* RAData) {
|
||||
const FEXCore::IR::RegisterAllocationData* RAData) {
|
||||
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
|
||||
|
||||
JumpTargets.clear();
|
||||
|
||||
@@ -8,7 +8,6 @@ $end_info$
|
||||
#pragma once
|
||||
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
@@ -24,6 +23,8 @@ $end_info$
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <utility>
|
||||
@@ -46,7 +47,7 @@ public:
|
||||
|
||||
[[nodiscard]]
|
||||
CPUBackend::CompiledCode CompileCode(uint64_t Entry, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData,
|
||||
FEXCore::IR::RegisterAllocationData* RAData) override;
|
||||
const FEXCore::IR::RegisterAllocationData* RAData) override;
|
||||
|
||||
[[nodiscard]]
|
||||
void* MapRegion(void* HostPtr, uint64_t, uint64_t) override {
|
||||
@@ -83,7 +84,7 @@ private:
|
||||
fextl::map<IR::NodeID, ARMEmitter::BiDirectionalLabel> JumpTargets;
|
||||
|
||||
[[nodiscard]]
|
||||
FEXCore::ARMEmitter::Register GetReg(IR::NodeID Node) const {
|
||||
ARMEmitter::Register GetReg(IR::NodeID Node) const {
|
||||
const auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
@@ -98,7 +99,7 @@ private:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
FEXCore::ARMEmitter::VRegister GetVReg(IR::NodeID Node) const {
|
||||
ARMEmitter::VRegister GetVReg(IR::NodeID Node) const {
|
||||
const auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
@@ -113,12 +114,12 @@ private:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register> GetRegPair(IR::NodeID Node) const {
|
||||
std::pair<ARMEmitter::Register, ARMEmitter::Register> GetRegPair(IR::NodeID Node) const {
|
||||
const auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRPairClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
return GeneralPairRegisters[Reg.Reg];
|
||||
return std::make_pair(GeneralRegisters[Reg.Reg], GeneralRegisters[Reg.Reg + 1]);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
@@ -134,7 +135,7 @@ private:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
FEXCore::ARMEmitter::Register GetZeroableReg(IR::OrderedNodeWrapper Src) const {
|
||||
ARMEmitter::Register GetZeroableReg(IR::OrderedNodeWrapper Src) const {
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Src, &Const)) {
|
||||
LOGMAN_THROW_AA_FMT(Const == 0, "Only valid constant");
|
||||
@@ -154,6 +155,99 @@ private:
|
||||
ARMEmitter::ShiftType::ROR;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Size ConvertSize(const IR::IROp_Header* Op) {
|
||||
return Op->Size == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Size ConvertSize48(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_AA_FMT(Op->Size == 4 || Op->Size == 8, "Invalid size");
|
||||
return ConvertSize(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize16(uint8_t ElementSize) {
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 1 || ElementSize == 2 || ElementSize == 4 || ElementSize == 8 || ElementSize == 16, "Invalid size");
|
||||
return ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ARMEmitter::SubRegSize::i128Bit;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize16(const IR::IROp_Header* Op) {
|
||||
return ConvertSubRegSize16(Op->ElementSize);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize8(uint8_t ElementSize) {
|
||||
LOGMAN_THROW_AA_FMT(ElementSize != 16, "Invalid size");
|
||||
return ConvertSubRegSize16(ElementSize);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize8(const IR::IROp_Header* Op) {
|
||||
return ConvertSubRegSize8(Op->ElementSize);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize4(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_AA_FMT(Op->ElementSize != 8, "Invalid size");
|
||||
return ConvertSubRegSize8(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize248(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_AA_FMT(Op->ElementSize != 1, "Invalid size");
|
||||
return ConvertSubRegSize8(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair16(const IR::IROp_Header* Op) {
|
||||
return ARMEmitter::ToVectorSizePair(ConvertSubRegSize16(Op));
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair8(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_AA_FMT(Op->ElementSize != 16, "Invalid size");
|
||||
return ConvertSubRegSizePair16(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair248(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_AA_FMT(Op->ElementSize != 1, "Invalid size");
|
||||
return ConvertSubRegSizePair8(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Condition MapCC(IR::CondClassType Cond) {
|
||||
switch (Cond.Val) {
|
||||
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
|
||||
case FEXCore::IR::COND_NEQ: return ARMEmitter::Condition::CC_NE;
|
||||
case FEXCore::IR::COND_SGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_SLT: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_SGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_SLE: return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_UGE: return ARMEmitter::Condition::CC_CS;
|
||||
case FEXCore::IR::COND_ULT: return ARMEmitter::Condition::CC_CC;
|
||||
case FEXCore::IR::COND_UGT: return ARMEmitter::Condition::CC_HI;
|
||||
case FEXCore::IR::COND_ULE: return ARMEmitter::Condition::CC_LS;
|
||||
case FEXCore::IR::COND_FLU: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_FGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_FLEU: return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_FGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_FU: return ARMEmitter::Condition::CC_VS;
|
||||
case FEXCore::IR::COND_FNU: return ARMEmitter::Condition::CC_VC;
|
||||
case FEXCore::IR::COND_VS:
|
||||
case FEXCore::IR::COND_VC:
|
||||
case FEXCore::IR::COND_MI: return ARMEmitter::Condition::CC_MI;
|
||||
case FEXCore::IR::COND_PL: return ARMEmitter::Condition::CC_PL;
|
||||
default: LOGMAN_MSG_A_FMT("Unsupported compare type"); return ARMEmitter::Condition::CC_NV;
|
||||
}
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsFPR(IR::NodeID Node) const;
|
||||
[[nodiscard]]
|
||||
@@ -162,8 +256,8 @@ private:
|
||||
bool IsGPRPair(IR::NodeID Node) const;
|
||||
|
||||
[[nodiscard]]
|
||||
FEXCore::ARMEmitter::ExtendedMemOperand GenerateMemOperand(
|
||||
uint8_t AccessSize, FEXCore::ARMEmitter::Register Base, IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale);
|
||||
ARMEmitter::ExtendedMemOperand GenerateMemOperand(uint8_t AccessSize, ARMEmitter::Register Base, IR::OrderedNodeWrapper Offset,
|
||||
IR::MemOffsetType OffsetType, uint8_t OffsetScale);
|
||||
|
||||
// NOTE: Will use TMP1 as a way to encode immediates that happen to fall outside
|
||||
// the limits of the scalar plus immediate variant of SVE load/stores.
|
||||
@@ -171,8 +265,8 @@ private:
|
||||
// TMP1 is safe to use again once this memory operand is used with its
|
||||
// equivalent loads or stores that this was called for.
|
||||
[[nodiscard]]
|
||||
FEXCore::ARMEmitter::SVEMemOperand GenerateSVEMemOperand(uint8_t AccessSize, FEXCore::ARMEmitter::Register Base,
|
||||
IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale);
|
||||
ARMEmitter::SVEMemOperand GenerateSVEMemOperand(uint8_t AccessSize, ARMEmitter::Register Base, IR::OrderedNodeWrapper Offset,
|
||||
IR::MemOffsetType OffsetType, uint8_t OffsetScale);
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsInlineConstant(const IR::OrderedNodeWrapper& Node, uint64_t* Value = nullptr) const;
|
||||
@@ -187,7 +281,7 @@ private:
|
||||
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
|
||||
void EmitDetectionString();
|
||||
IR::RegisterAllocationPass* RAPass;
|
||||
IR::RegisterAllocationData* RAData;
|
||||
const IR::RegisterAllocationData* RAData;
|
||||
FEXCore::Core::DebugData* DebugData;
|
||||
|
||||
void ResetStack();
|
||||
|
||||
@@ -7,8 +7,6 @@ $end_info$
|
||||
|
||||
#include "FEXCore/Core/X86Enums.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
@@ -86,170 +84,33 @@ DEF_OP(LoadRegister) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
const auto regId = Op->Offset == offsetof(Core::CpuStateFrame, State.pf_raw) ?
|
||||
(StaticRegisters.size() - 2) :
|
||||
Op->Offset == offsetof(Core::CpuStateFrame, State.af_raw) ?
|
||||
(StaticRegisters.size() - 1) :
|
||||
(Op->Offset - offsetof(Core::CpuStateFrame, State.gregs[0])) / Core::CPUState::GPR_REG_SIZE;
|
||||
unsigned Reg = Op->Reg == Core::CPUState::PF_AS_GREG ? (StaticRegisters.size() - 2) :
|
||||
Op->Reg == Core::CPUState::AF_AS_GREG ? (StaticRegisters.size() - 1) :
|
||||
Op->Reg;
|
||||
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
LOGMAN_THROW_A_FMT(Reg < StaticRegisters.size(), "out of range reg");
|
||||
const auto reg = StaticRegisters[Reg];
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < StaticRegisters.size(), "out of range regId");
|
||||
|
||||
const auto reg = StaticRegisters[regId];
|
||||
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
if (GetReg(Node).Idx() != reg.Idx()) {
|
||||
if (GetReg(Node).Idx() != reg.Idx()) {
|
||||
if (OpSize == 4) {
|
||||
mov(GetReg(Node).W(), reg.W());
|
||||
}
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
if (GetReg(Node).Idx() != reg.Idx()) {
|
||||
} else {
|
||||
mov(GetReg(Node).X(), reg.X());
|
||||
}
|
||||
break;
|
||||
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadRegister GPR size: {}", OpSize); break;
|
||||
}
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
const auto regSize = HostSupportsSVE256 ? Core::CPUState::XMM_AVX_REG_SIZE : Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
LOGMAN_THROW_A_FMT(Op->Reg < StaticFPRegisters.size(), "out of range reg");
|
||||
LOGMAN_THROW_A_FMT(OpSize == regSize, "expected sized");
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
|
||||
const auto guest = StaticFPRegisters[regId];
|
||||
const auto guest = StaticFPRegisters[Op->Reg];
|
||||
const auto host = GetVReg(Node);
|
||||
|
||||
if (HostSupportsSVE256) {
|
||||
const auto regOffs = Op->Offset & 31;
|
||||
|
||||
ARMEmitter::SingleUseForwardLabel DataLocation;
|
||||
const auto LoadPredicate = [this, &DataLocation] {
|
||||
const auto Predicate = ARMEmitter::PReg::p0;
|
||||
adr(TMP1, &DataLocation);
|
||||
ldr(Predicate, TMP1);
|
||||
return Predicate.Merging();
|
||||
};
|
||||
|
||||
const auto EmitData = [this, &DataLocation](uint32_t Value) {
|
||||
ARMEmitter::SingleUseForwardLabel PastConstant;
|
||||
b(&PastConstant);
|
||||
Bind(&DataLocation);
|
||||
dc32(Value);
|
||||
Bind(&PastConstant);
|
||||
};
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
|
||||
dup(ARMEmitter::ScalarRegSize::i8Bit, host, guest, 0);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
|
||||
fmov(host.H(), guest.H());
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 3) == 0, "unexpected regOffs: {}", regOffs);
|
||||
if (regOffs == 0) {
|
||||
if (host.Idx() != guest.Idx()) {
|
||||
fmov(host.S(), guest.S());
|
||||
}
|
||||
} else {
|
||||
const auto Predicate = LoadPredicate();
|
||||
|
||||
dup(FEXCore::ARMEmitter::SubRegSize::i32Bit, VTMP1.Z(), host.Z(), 0);
|
||||
mov(FEXCore::ARMEmitter::SubRegSize::i32Bit, guest.Z(), Predicate, VTMP1.Z());
|
||||
|
||||
EmitData(1U << regOffs);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case 8: {
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 7) == 0, "unexpected regOffs: {}", regOffs);
|
||||
if (regOffs == 0) {
|
||||
if (host.Idx() != guest.Idx()) {
|
||||
dup(ARMEmitter::ScalarRegSize::i64Bit, host, guest, 0);
|
||||
}
|
||||
} else {
|
||||
const auto Predicate = LoadPredicate();
|
||||
|
||||
dup(FEXCore::ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), host.Z(), 0);
|
||||
mov(FEXCore::ARMEmitter::SubRegSize::i64Bit, guest.Z(), Predicate, VTMP1.Z());
|
||||
|
||||
EmitData(1U << regOffs);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case 16: {
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
|
||||
if (host.Idx() != guest.Idx()) {
|
||||
mov(host.Q(), guest.Q());
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case 32: {
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
|
||||
if (host.Idx() != guest.Idx()) {
|
||||
mov(ARMEmitter::SubRegSize::i64Bit, host.Z(), PRED_TMP_32B.Merging(), guest.Z());
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadRegister FPR size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
const auto regOffs = Op->Offset & 15;
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
|
||||
dup(ARMEmitter::ScalarRegSize::i8Bit, host, guest, 0);
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
|
||||
fmov(host.H(), guest.H());
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 3) == 0, "unexpected regOffs: {}", regOffs);
|
||||
if (regOffs == 0) {
|
||||
if (host.Idx() != guest.Idx()) {
|
||||
fmov(host.S(), guest.S());
|
||||
}
|
||||
} else {
|
||||
ins(ARMEmitter::SubRegSize::i32Bit, host, 0, guest, regOffs / 4);
|
||||
}
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 7) == 0, "unexpected regOffs: {}", regOffs);
|
||||
if (regOffs == 0) {
|
||||
if (host.Idx() != guest.Idx()) {
|
||||
dup(ARMEmitter::ScalarRegSize::i64Bit, host, guest, 0);
|
||||
}
|
||||
} else {
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, host, 0, guest, regOffs / 8);
|
||||
}
|
||||
break;
|
||||
|
||||
case 16:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
|
||||
if (host.Idx() != guest.Idx()) {
|
||||
mov(host.Q(), guest.Q());
|
||||
}
|
||||
break;
|
||||
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadRegister FPR size: {}", OpSize); break;
|
||||
if (host.Idx() != guest.Idx()) {
|
||||
if (HostSupportsSVE256) {
|
||||
mov(ARMEmitter::SubRegSize::i64Bit, host.Z(), PRED_TMP_32B.Merging(), guest.Z());
|
||||
} else {
|
||||
mov(host.Q(), guest.Q());
|
||||
}
|
||||
}
|
||||
} else {
|
||||
@@ -261,171 +122,33 @@ DEF_OP(StoreRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
unsigned Reg = Op->Reg == Core::CPUState::PF_AS_GREG ? (StaticRegisters.size() - 2) :
|
||||
Op->Reg == Core::CPUState::AF_AS_GREG ? (StaticRegisters.size() - 1) :
|
||||
Op->Reg;
|
||||
|
||||
const auto regId = Op->Offset == offsetof(Core::CpuStateFrame, State.pf_raw) ?
|
||||
(StaticRegisters.size() - 2) :
|
||||
Op->Offset == offsetof(Core::CpuStateFrame, State.af_raw) ?
|
||||
(StaticRegisters.size() - 1) :
|
||||
(Op->Offset - offsetof(Core::CpuStateFrame, State.gregs[0])) / Core::CPUState::GPR_REG_SIZE;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < StaticRegisters.size(), "out of range regId");
|
||||
|
||||
const auto reg = StaticRegisters[regId];
|
||||
LOGMAN_THROW_A_FMT(Reg < StaticRegisters.size(), "out of range reg");
|
||||
const auto reg = StaticRegisters[Reg];
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
if (Src.Idx() != reg.Idx()) {
|
||||
mov(ARMEmitter::Size::i32Bit, reg, Src);
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
if (Src.Idx() != reg.Idx()) {
|
||||
mov(ARMEmitter::Size::i64Bit, reg, Src);
|
||||
}
|
||||
break;
|
||||
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreRegister GPR size: {}", OpSize); break;
|
||||
if (Src.Idx() != reg.Idx()) {
|
||||
// Always use 64-bit, it's faster. Upper bits ignored for 32-bit mode.
|
||||
mov(ARMEmitter::Size::i64Bit, reg, Src);
|
||||
}
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
const auto regSize = HostSupportsSVE256 ? Core::CPUState::XMM_AVX_REG_SIZE : Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
LOGMAN_THROW_A_FMT(Op->Reg < StaticFPRegisters.size(), "reg out of range");
|
||||
LOGMAN_THROW_A_FMT(OpSize == regSize, "expected sized");
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "regId out of range");
|
||||
|
||||
const auto guest = StaticFPRegisters[regId];
|
||||
const auto guest = StaticFPRegisters[Op->Reg];
|
||||
const auto host = GetVReg(Op->Value.ID());
|
||||
|
||||
if (HostSupportsSVE256) {
|
||||
// 256-bit capable hardware allows us to expand the allowed
|
||||
// offsets used, however we cannot use Adv. SIMD's INS instruction
|
||||
// at all, since it will zero out the upper lanes of the 256-bit SVE
|
||||
// vectors, so we'll need to set up a proper predicate for performing
|
||||
// the insert.
|
||||
|
||||
const auto regOffs = Op->Offset & 31;
|
||||
|
||||
// Compartmentalized setting up of the predicate for the cases that need it.
|
||||
ARMEmitter::SingleUseForwardLabel DataLocation;
|
||||
const auto LoadPredicate = [this, &DataLocation] {
|
||||
const auto Predicate = ARMEmitter::PReg::p0;
|
||||
adr(TMP1, &DataLocation);
|
||||
ldr(Predicate, TMP1);
|
||||
return Predicate.Merging();
|
||||
};
|
||||
|
||||
// Emits the predicate data and provides the necessary jump to go around the
|
||||
// emitted data instead of trying to execute it. Place at end of necessary code.
|
||||
// It's helpful to treat LoadPredicate and EmitData as a prologue and epilogue
|
||||
// respectfully.
|
||||
const auto EmitData = [this, &DataLocation](uint32_t Data) {
|
||||
ARMEmitter::SingleUseForwardLabel PastConstant;
|
||||
b(&PastConstant);
|
||||
Bind(&DataLocation);
|
||||
dc32(Data);
|
||||
Bind(&PastConstant);
|
||||
};
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
LOGMAN_THROW_AA_FMT(regOffs <= 31, "unexpected reg index: {}", regOffs);
|
||||
|
||||
const auto Predicate = LoadPredicate();
|
||||
dup(ARMEmitter::SubRegSize::i8Bit, VTMP1.Z(), host.Z(), 0);
|
||||
mov(ARMEmitter::SubRegSize::i8Bit, guest.Z(), Predicate, VTMP1.Z());
|
||||
|
||||
EmitData(1U << regOffs);
|
||||
break;
|
||||
}
|
||||
|
||||
case 2: {
|
||||
LOGMAN_THROW_AA_FMT((regOffs / 2) <= 15, "unexpected reg index: {}", regOffs / 2);
|
||||
|
||||
const auto Predicate = LoadPredicate();
|
||||
dup(ARMEmitter::SubRegSize::i16Bit, VTMP1.Z(), host.Z(), 0);
|
||||
mov(ARMEmitter::SubRegSize::i16Bit, guest.Z(), Predicate, VTMP1.Z());
|
||||
|
||||
EmitData(1U << regOffs);
|
||||
break;
|
||||
}
|
||||
|
||||
case 4: {
|
||||
LOGMAN_THROW_AA_FMT((regOffs / 4) <= 7, "unexpected reg index: {}", regOffs / 4);
|
||||
|
||||
const auto Predicate = LoadPredicate();
|
||||
|
||||
dup(ARMEmitter::SubRegSize::i32Bit, VTMP1.Z(), host.Z(), 0);
|
||||
mov(ARMEmitter::SubRegSize::i32Bit, guest.Z(), Predicate, VTMP1.Z());
|
||||
|
||||
EmitData(1U << regOffs);
|
||||
break;
|
||||
}
|
||||
|
||||
case 8: {
|
||||
LOGMAN_THROW_AA_FMT((regOffs / 8) <= 3, "unexpected reg index: {}", regOffs / 8);
|
||||
|
||||
const auto Predicate = LoadPredicate();
|
||||
|
||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), host.Z(), 0);
|
||||
mov(ARMEmitter::SubRegSize::i64Bit, guest.Z(), Predicate, VTMP1.Z());
|
||||
|
||||
EmitData(1U << regOffs);
|
||||
break;
|
||||
}
|
||||
|
||||
case 16: {
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
|
||||
if (guest.Idx() != host.Idx()) {
|
||||
mov(guest.Q(), host.Q());
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case 32: {
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
|
||||
if (guest.Idx() != host.Idx()) {
|
||||
mov(ARMEmitter::SubRegSize::i64Bit, guest.Z(), PRED_TMP_32B.Merging(), host.Z());
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreRegister FPR size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
const auto regOffs = Op->Offset & 15;
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: ins(ARMEmitter::SubRegSize::i8Bit, guest, regOffs, host, 0); break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 1) == 0, "unexpected regOffs: {}", regOffs);
|
||||
ins(ARMEmitter::SubRegSize::i16Bit, guest, regOffs / 2, host, 0);
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 3) == 0, "unexpected regOffs: {}", regOffs);
|
||||
// XXX: This had a bug with insert of size 16bit
|
||||
ins(ARMEmitter::SubRegSize::i32Bit, guest, regOffs / 4, host, 0);
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 7) == 0, "unexpected regOffs: {}", regOffs);
|
||||
// XXX: This had a bug with insert of size 16bit
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, guest, regOffs / 8, host, 0);
|
||||
break;
|
||||
|
||||
case 16:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
|
||||
if (guest.Idx() != host.Idx()) {
|
||||
mov(guest.Q(), host.Q());
|
||||
}
|
||||
break;
|
||||
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreRegister FPR size: {}", OpSize); break;
|
||||
if (guest.Idx() != host.Idx()) {
|
||||
if (HostSupportsSVE256) {
|
||||
mov(ARMEmitter::SubRegSize::i64Bit, guest.Z(), PRED_TMP_32B.Merging(), host.Z());
|
||||
} else {
|
||||
mov(guest.Q(), host.Q());
|
||||
}
|
||||
}
|
||||
} else {
|
||||
@@ -445,7 +168,7 @@ DEF_OP(LoadContextIndexed) {
|
||||
case 2:
|
||||
case 4:
|
||||
case 8: {
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, Index, FEXCore::ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Op->Stride));
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, Index, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Op->Stride));
|
||||
const auto Dst = GetReg(Node);
|
||||
switch (OpSize) {
|
||||
case 1: ldrb(Dst, TMP1, Op->BaseOffset); break;
|
||||
@@ -467,7 +190,7 @@ DEF_OP(LoadContextIndexed) {
|
||||
case 8:
|
||||
case 16:
|
||||
case 32: {
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, Index, FEXCore::ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Op->Stride));
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, Index, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Op->Stride));
|
||||
const auto Dst = GetVReg(Node);
|
||||
|
||||
switch (OpSize) {
|
||||
@@ -510,7 +233,7 @@ DEF_OP(StoreContextIndexed) {
|
||||
case 2:
|
||||
case 4:
|
||||
case 8: {
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, Index, FEXCore::ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Op->Stride));
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, Index, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Op->Stride));
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: strb(Value, TMP1, Op->BaseOffset); break;
|
||||
@@ -534,7 +257,7 @@ DEF_OP(StoreContextIndexed) {
|
||||
case 8:
|
||||
case 16:
|
||||
case 32: {
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, Index, FEXCore::ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Op->Stride));
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, Index, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Op->Stride));
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: strb(Value, TMP1, Op->BaseOffset); break;
|
||||
@@ -827,8 +550,8 @@ DEF_OP(StoreFlag) {
|
||||
strb(GetReg(Op->Value.ID()), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
}
|
||||
|
||||
FEXCore::ARMEmitter::ExtendedMemOperand Arm64JITCore::GenerateMemOperand(
|
||||
uint8_t AccessSize, FEXCore::ARMEmitter::Register Base, IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale) {
|
||||
ARMEmitter::ExtendedMemOperand Arm64JITCore::GenerateMemOperand(
|
||||
uint8_t AccessSize, ARMEmitter::Register Base, IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale) {
|
||||
if (Offset.IsInvalid()) {
|
||||
return ARMEmitter::ExtendedMemOperand(Base.X(), ARMEmitter::IndexType::OFFSET, 0);
|
||||
} else {
|
||||
@@ -855,17 +578,16 @@ FEXCore::ARMEmitter::ExtendedMemOperand Arm64JITCore::GenerateMemOperand(
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
FEXCore::ARMEmitter::SVEMemOperand Arm64JITCore::GenerateSVEMemOperand(uint8_t AccessSize, FEXCore::ARMEmitter::Register Base,
|
||||
IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType,
|
||||
[[maybe_unused]] uint8_t OffsetScale) {
|
||||
ARMEmitter::SVEMemOperand Arm64JITCore::GenerateSVEMemOperand(uint8_t AccessSize, ARMEmitter::Register Base, IR::OrderedNodeWrapper Offset,
|
||||
IR::MemOffsetType OffsetType, [[maybe_unused]] uint8_t OffsetScale) {
|
||||
if (Offset.IsInvalid()) {
|
||||
return FEXCore::ARMEmitter::SVEMemOperand(Base.X(), 0);
|
||||
return ARMEmitter::SVEMemOperand(Base.X(), 0);
|
||||
}
|
||||
|
||||
uint64_t Const {};
|
||||
if (IsInlineConstant(Offset, &Const)) {
|
||||
if (Const == 0) {
|
||||
return FEXCore::ARMEmitter::SVEMemOperand(Base.X(), 0);
|
||||
return ARMEmitter::SVEMemOperand(Base.X(), 0);
|
||||
}
|
||||
|
||||
const auto SignedConst = static_cast<int64_t>(Const);
|
||||
@@ -888,13 +610,13 @@ FEXCore::ARMEmitter::SVEMemOperand Arm64JITCore::GenerateSVEMemOperand(uint8_t A
|
||||
// then we can encode it as an immediate offset.
|
||||
//
|
||||
if (IsCleanlyDivisible && Index >= -8 && Index <= 7) {
|
||||
return FEXCore::ARMEmitter::SVEMemOperand(Base.X(), static_cast<uint64_t>(Index));
|
||||
return ARMEmitter::SVEMemOperand(Base.X(), static_cast<uint64_t>(Index));
|
||||
}
|
||||
|
||||
// If we can't do that for whatever reason, then unfortunately, we need
|
||||
// to move it over to a temporary to use as an offset.
|
||||
mov(TMP1, Const);
|
||||
return FEXCore::ARMEmitter::SVEMemOperand(Base.X(), TMP1);
|
||||
return ARMEmitter::SVEMemOperand(Base.X(), TMP1);
|
||||
}
|
||||
|
||||
// Otherwise handle it like normal.
|
||||
@@ -904,7 +626,7 @@ FEXCore::ARMEmitter::SVEMemOperand Arm64JITCore::GenerateSVEMemOperand(uint8_t A
|
||||
LOGMAN_THROW_A_FMT(OffsetType.Val == IR::MEM_OFFSET_SXTX.Val, "Currently only the default offset type (SXTX) is supported.");
|
||||
|
||||
const auto RegOffset = GetReg(Offset.ID());
|
||||
return FEXCore::ARMEmitter::SVEMemOperand(Base.X(), RegOffset.X());
|
||||
return ARMEmitter::SVEMemOperand(Base.X(), RegOffset.X());
|
||||
}
|
||||
|
||||
DEF_OP(LoadMem) {
|
||||
@@ -949,11 +671,17 @@ DEF_OP(LoadMemTSO) {
|
||||
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid() || CTX->HostFeatures.SupportsTSOImm9, "unexpected offset");
|
||||
LOGMAN_THROW_A_FMT(Op->OffsetScale == 1, "unexpected offset scale");
|
||||
LOGMAN_THROW_A_FMT(Op->OffsetType == IR::MEM_OFFSET_SXTX, "unexpected offset type");
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
(void)IsInlineConstant(Op->Offset, &Offset);
|
||||
LOGMAN_THROW_A_FMT(IsInlineConstant(Op->Offset, &Offset), "expected immediate");
|
||||
}
|
||||
|
||||
if (OpSize == 1) {
|
||||
@@ -1018,7 +746,7 @@ DEF_OP(LoadMemTSO) {
|
||||
}
|
||||
if (VectorTSOEnabled()) {
|
||||
// Half-barrier.
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISHLD);
|
||||
dmb(ARMEmitter::BarrierScope::ISHLD);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1030,7 +758,7 @@ DEF_OP(VLoadVectorMasked) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
const auto CMPPredicate = ARMEmitter::PReg::p0;
|
||||
const auto GoverningPredicate = Is256Bit ? PRED_TMP_32B : PRED_TMP_16B;
|
||||
@@ -1040,17 +768,10 @@ DEF_OP(VLoadVectorMasked) {
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto MemSrc = GenerateSVEMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 1 || ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize = ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
// Check if the sign bit is set for the given element size.
|
||||
cmplt(SubRegSize, CMPPredicate, GoverningPredicate.Zeroing(), MaskReg.Z(), 0);
|
||||
|
||||
switch (ElementSize) {
|
||||
switch (IROp->ElementSize) {
|
||||
case 1: {
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), CMPPredicate.Zeroing(), MemSrc);
|
||||
break;
|
||||
@@ -1067,7 +788,7 @@ DEF_OP(VLoadVectorMasked) {
|
||||
ld1d(Dst.Z(), CMPPredicate.Zeroing(), MemSrc);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled VLoadVectorMasked size: {}", ElementSize); break;
|
||||
default: break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1078,7 +799,7 @@ DEF_OP(VStoreVectorMasked) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
const auto CMPPredicate = ARMEmitter::PReg::p0;
|
||||
const auto GoverningPredicate = Is256Bit ? PRED_TMP_32B : PRED_TMP_16B;
|
||||
@@ -1088,17 +809,10 @@ DEF_OP(VStoreVectorMasked) {
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto MemDst = GenerateSVEMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 1 || ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize = ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
// Check if the sign bit is set for the given element size.
|
||||
cmplt(SubRegSize, CMPPredicate, GoverningPredicate.Zeroing(), MaskReg.Z(), 0);
|
||||
|
||||
switch (ElementSize) {
|
||||
switch (IROp->ElementSize) {
|
||||
case 1: {
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(RegData.Z(), CMPPredicate.Zeroing(), MemDst);
|
||||
break;
|
||||
@@ -1115,7 +829,7 @@ DEF_OP(VStoreVectorMasked) {
|
||||
st1d(RegData.Z(), CMPPredicate.Zeroing(), MemDst);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled VStoreVectorMasked size: {}", ElementSize); break;
|
||||
default: break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1170,7 +884,7 @@ DEF_OP(VStoreVectorElement) {
|
||||
|
||||
// Emit a half-barrier if TSO is enabled.
|
||||
if (CTX->IsAtomicTSOEnabled() && VectorTSOEnabled()) {
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
}
|
||||
|
||||
if (Is256Bit) {
|
||||
@@ -1377,11 +1091,17 @@ DEF_OP(StoreMemTSO) {
|
||||
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid() || CTX->HostFeatures.SupportsTSOImm9, "unexpected offset");
|
||||
LOGMAN_THROW_A_FMT(Op->OffsetScale == 1, "unexpected offset scale");
|
||||
LOGMAN_THROW_A_FMT(Op->OffsetType == IR::MEM_OFFSET_SXTX, "unexpected offset type");
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
(void)IsInlineConstant(Op->Offset, &Offset);
|
||||
LOGMAN_THROW_A_FMT(IsInlineConstant(Op->Offset, &Offset), "expected immediate");
|
||||
}
|
||||
|
||||
if (OpSize == 1) {
|
||||
@@ -1416,7 +1136,7 @@ DEF_OP(StoreMemTSO) {
|
||||
} else {
|
||||
if (VectorTSOEnabled()) {
|
||||
// Half-Barrier.
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
}
|
||||
const auto Src = GetVReg(Op->Value.ID());
|
||||
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
@@ -1455,7 +1175,7 @@ DEF_OP(MemSet) {
|
||||
|
||||
uint64_t DirectionConstant;
|
||||
bool DirectionIsInline = IsInlineConstant(Op->Direction, &DirectionConstant);
|
||||
FEXCore::ARMEmitter::Register DirectionReg = ARMEmitter::Reg::r0;
|
||||
ARMEmitter::Register DirectionReg = ARMEmitter::Reg::r0;
|
||||
if (!DirectionIsInline) {
|
||||
DirectionReg = GetReg(Op->Direction.ID());
|
||||
}
|
||||
@@ -1638,13 +1358,13 @@ DEF_OP(MemCpy) {
|
||||
|
||||
const bool IsAtomic = Op->IsAtomic && MemcpySetTSOEnabled();
|
||||
const int32_t Size = Op->Size;
|
||||
const auto MemRegDest = GetReg(Op->AddrDest.ID());
|
||||
const auto MemRegSrc = GetReg(Op->AddrSrc.ID());
|
||||
const auto MemRegDest = GetReg(Op->Dest.ID());
|
||||
const auto MemRegSrc = GetReg(Op->Src.ID());
|
||||
|
||||
const auto Length = GetReg(Op->Length.ID());
|
||||
uint64_t DirectionConstant;
|
||||
bool DirectionIsInline = IsInlineConstant(Op->Direction, &DirectionConstant);
|
||||
FEXCore::ARMEmitter::Register DirectionReg = ARMEmitter::Reg::r0;
|
||||
ARMEmitter::Register DirectionReg = ARMEmitter::Reg::r0;
|
||||
if (!DirectionIsInline) {
|
||||
DirectionReg = GetReg(Op->Direction.ID());
|
||||
}
|
||||
@@ -1663,19 +1383,8 @@ DEF_OP(MemCpy) {
|
||||
ARMEmitter::SingleUseForwardLabel Done {};
|
||||
|
||||
mov(TMP1, Length.X());
|
||||
if (Op->PrefixDest.IsInvalid()) {
|
||||
mov(TMP2, MemRegDest.X());
|
||||
} else {
|
||||
const auto Prefix = GetReg(Op->PrefixDest.ID());
|
||||
add(TMP2, Prefix.X(), MemRegDest.X());
|
||||
}
|
||||
|
||||
if (Op->PrefixSrc.IsInvalid()) {
|
||||
mov(TMP3, MemRegSrc.X());
|
||||
} else {
|
||||
const auto Prefix = GetReg(Op->PrefixSrc.ID());
|
||||
add(TMP3, Prefix.X(), MemRegSrc.X());
|
||||
}
|
||||
mov(TMP2, MemRegDest.X());
|
||||
mov(TMP3, MemRegSrc.X());
|
||||
|
||||
// TMP1 = Length
|
||||
// TMP2 = Dest
|
||||
@@ -1987,9 +1696,9 @@ DEF_OP(ParanoidLoadMemTSO) {
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 1, TMP2);
|
||||
break;
|
||||
case 32:
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), PRED_TMP_32B.Zeroing(), MemReg);
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
@@ -2063,9 +1772,9 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
break;
|
||||
}
|
||||
case 32: {
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(Src.Z(), PRED_TMP_32B, MemReg, 0);
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize); break;
|
||||
@@ -2093,7 +1802,7 @@ DEF_OP(CacheLineClear) {
|
||||
|
||||
if (Op->Serialize) {
|
||||
// If requested, serialized all of the data cache operations.
|
||||
dsb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
dsb(ARMEmitter::BarrierScope::ISH);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -10,7 +10,6 @@ $end_info$
|
||||
#endif
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
|
||||
@@ -28,9 +27,9 @@ DEF_OP(GuestOpcode) {
|
||||
DEF_OP(Fence) {
|
||||
auto Op = IROp->C<IR::IROp_Fence>();
|
||||
switch (Op->Fence) {
|
||||
case IR::Fence_Load.Val: dmb(FEXCore::ARMEmitter::BarrierScope::LD); break;
|
||||
case IR::Fence_LoadStore.Val: dmb(FEXCore::ARMEmitter::BarrierScope::SY); break;
|
||||
case IR::Fence_Store.Val: dmb(FEXCore::ARMEmitter::BarrierScope::ST); break;
|
||||
case IR::Fence_Load.Val: dmb(ARMEmitter::BarrierScope::LD); break;
|
||||
case IR::Fence_LoadStore.Val: dmb(ARMEmitter::BarrierScope::SY); break;
|
||||
case IR::Fence_Store.Val: dmb(ARMEmitter::BarrierScope::ST); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Fence: {}", Op->Fence); break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -11,12 +11,14 @@ namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
DEF_OP(ExtractElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_ExtractElementPair>();
|
||||
LOGMAN_THROW_AA_FMT(Op->Header.Size == 4 || Op->Header.Size == 8, "Invalid size");
|
||||
const auto EmitSize = Op->Header.Size == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Src = GetRegPair(Op->Pair.ID());
|
||||
const std::array<ARMEmitter::Register, 2> Regs = {Src.first, Src.second};
|
||||
mov(EmitSize, GetReg(Node), Regs[Op->Element]);
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Pair = GetRegPair(Op->Pair.ID());
|
||||
const auto Src = Op->Element == 0 ? Pair.first : Pair.second;
|
||||
|
||||
if (Dst != Src) {
|
||||
mov(ConvertSize48(IROp), Dst, Src);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(CreateElementPair) {
|
||||
@@ -42,5 +44,25 @@ DEF_OP(CreateElementPair) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Copy) {
|
||||
auto Op = IROp->C<IR::IROp_Copy>();
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), GetReg(Op->Source.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(Swap1) {
|
||||
auto Op = IROp->C<IR::IROp_Swap1>();
|
||||
auto A = GetReg(Op->A.ID()), B = GetReg(Op->B.ID());
|
||||
LOGMAN_THROW_AA_FMT(B == GetReg(Node), "Invariant");
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, TMP1, A);
|
||||
mov(ARMEmitter::Size::i64Bit, A, B);
|
||||
mov(ARMEmitter::Size::i64Bit, B, TMP1);
|
||||
}
|
||||
|
||||
DEF_OP(Swap2) {
|
||||
// Implemented above
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
} // namespace FEXCore::CPU
|
||||
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
@@ -70,6 +70,17 @@ struct LoadSourceOptions {
|
||||
bool AllowUpperGarbage = false;
|
||||
};
|
||||
|
||||
struct AddressMode {
|
||||
Ref Base {nullptr};
|
||||
Ref Index {nullptr};
|
||||
MemOffsetType IndexType = MEM_OFFSET_SXTX;
|
||||
uint8_t IndexScale = 1;
|
||||
int64_t Offset = 0;
|
||||
|
||||
// Size in bytes for the address calculation. 8 for an arm64 hardware mode.
|
||||
uint8_t AddrSize;
|
||||
};
|
||||
|
||||
class OpDispatchBuilder final : public IREmitter {
|
||||
friend class FEXCore::IR::Pass;
|
||||
friend class FEXCore::IR::PassManager;
|
||||
@@ -95,7 +106,7 @@ public:
|
||||
TYPE_RDRAND,
|
||||
};
|
||||
|
||||
OrderedNode* GetNewJumpBlock(uint64_t RIP) {
|
||||
Ref GetNewJumpBlock(uint64_t RIP) {
|
||||
auto it = JumpTargets.find(RIP);
|
||||
LOGMAN_THROW_A_FMT(it != JumpTargets.end(), "Couldn't find block generated for 0x{:x}", RIP);
|
||||
return it->second.BlockEntry;
|
||||
@@ -135,20 +146,20 @@ public:
|
||||
CalculateDeferredFlags();
|
||||
return _Jump();
|
||||
}
|
||||
IRPair<IROp_Jump> Jump(OrderedNode* _TargetBlock) {
|
||||
IRPair<IROp_Jump> Jump(Ref _TargetBlock) {
|
||||
CalculateDeferredFlags();
|
||||
return _Jump(_TargetBlock);
|
||||
}
|
||||
IRPair<IROp_CondJump> CondJump(OrderedNode* _Cmp1, OrderedNode* _Cmp2, OrderedNode* _TrueBlock, OrderedNode* _FalseBlock,
|
||||
CondClassType _Cond = {COND_NEQ}, uint8_t _CompareSize = 0) {
|
||||
IRPair<IROp_CondJump>
|
||||
CondJump(Ref _Cmp1, Ref _Cmp2, Ref _TrueBlock, Ref _FalseBlock, CondClassType _Cond = {COND_NEQ}, uint8_t _CompareSize = 0) {
|
||||
CalculateDeferredFlags();
|
||||
return _CondJump(_Cmp1, _Cmp2, _TrueBlock, _FalseBlock, _Cond, _CompareSize);
|
||||
}
|
||||
IRPair<IROp_CondJump> CondJump(OrderedNode* ssa0, CondClassType cond = {COND_NEQ}) {
|
||||
IRPair<IROp_CondJump> CondJump(Ref ssa0, CondClassType cond = {COND_NEQ}) {
|
||||
CalculateDeferredFlags();
|
||||
return _CondJump(ssa0, cond);
|
||||
}
|
||||
IRPair<IROp_CondJump> CondJump(OrderedNode* ssa0, OrderedNode* ssa1, OrderedNode* ssa2, CondClassType cond = {COND_NEQ}) {
|
||||
IRPair<IROp_CondJump> CondJump(Ref ssa0, Ref ssa1, Ref ssa2, CondClassType cond = {COND_NEQ}) {
|
||||
CalculateDeferredFlags();
|
||||
return _CondJump(ssa0, ssa1, ssa2, cond);
|
||||
}
|
||||
@@ -355,9 +366,8 @@ public:
|
||||
void SHLDImmediateOp(OpcodeArgs);
|
||||
void SHRDOp(OpcodeArgs);
|
||||
void SHRDImmediateOp(OpcodeArgs);
|
||||
template<bool IsImmediate, bool Is1Bit>
|
||||
void ASHROp(OpcodeArgs);
|
||||
template<bool SHR1Bit>
|
||||
void ASHRImmediateOp(OpcodeArgs);
|
||||
template<bool Left, bool IsImmediate, bool Is1Bit>
|
||||
void RotateOp(OpcodeArgs);
|
||||
void RCROp1Bit(OpcodeArgs);
|
||||
@@ -384,8 +394,8 @@ public:
|
||||
void POPFOp(OpcodeArgs);
|
||||
|
||||
struct CycleCounterPair {
|
||||
OrderedNode* CounterLow;
|
||||
OrderedNode* CounterHigh;
|
||||
Ref CounterLow;
|
||||
Ref CounterHigh;
|
||||
};
|
||||
CycleCounterPair CycleCounter();
|
||||
void RDTSCOp(OpcodeArgs);
|
||||
@@ -717,9 +727,9 @@ public:
|
||||
void VZEROOp(OpcodeArgs);
|
||||
|
||||
// X87 Ops
|
||||
OrderedNode* ReconstructFSW();
|
||||
Ref ReconstructFSW();
|
||||
// Returns new x87 stack top from FSW.
|
||||
OrderedNode* ReconstructX87StateFromFSW(OrderedNode* FSW);
|
||||
Ref ReconstructX87StateFromFSW(Ref FSW);
|
||||
template<size_t width>
|
||||
void FLD(OpcodeArgs);
|
||||
template<NamedVectorConstant constant>
|
||||
@@ -843,7 +853,7 @@ public:
|
||||
void FXSaveOp(OpcodeArgs);
|
||||
void FXRStoreOp(OpcodeArgs);
|
||||
|
||||
OrderedNode* XSaveBase(X86Tables::DecodedOp Op);
|
||||
Ref XSaveBase(X86Tables::DecodedOp Op);
|
||||
void XSaveOp(OpcodeArgs);
|
||||
|
||||
void PAlignrOp(OpcodeArgs);
|
||||
@@ -909,7 +919,7 @@ public:
|
||||
|
||||
void PSADBW(OpcodeArgs);
|
||||
|
||||
OrderedNode* BitwiseAtLeastTwo(OrderedNode* A, OrderedNode* B, OrderedNode* C);
|
||||
Ref BitwiseAtLeastTwo(Ref A, Ref B, Ref C);
|
||||
|
||||
void SHA1NEXTEOp(OpcodeArgs);
|
||||
void SHA1MSG1Op(OpcodeArgs);
|
||||
@@ -948,12 +958,14 @@ public:
|
||||
|
||||
void CRC32(OpcodeArgs);
|
||||
|
||||
void BreakOp(OpcodeArgs, FEXCore::IR::BreakDefinition BreakDefinition);
|
||||
void UnimplementedOp(OpcodeArgs);
|
||||
void PermissionRestrictedOp(OpcodeArgs);
|
||||
|
||||
void InvalidOp(OpcodeArgs);
|
||||
|
||||
void SetPackedRFLAG(bool Lower8, OrderedNode* Src);
|
||||
OrderedNode* GetPackedRFLAG(uint32_t FlagsMask = ~0U);
|
||||
void SetPackedRFLAG(bool Lower8, Ref Src);
|
||||
Ref GetPackedRFLAG(uint32_t FlagsMask = ~0U);
|
||||
|
||||
void SetMultiblock(bool _Multiblock) {
|
||||
Multiblock = _Multiblock;
|
||||
@@ -1002,7 +1014,7 @@ protected:
|
||||
|
||||
private:
|
||||
struct JumpTargetInfo {
|
||||
OrderedNode* BlockEntry;
|
||||
Ref BlockEntry;
|
||||
bool HaveEmitted;
|
||||
};
|
||||
|
||||
@@ -1016,10 +1028,10 @@ private:
|
||||
}
|
||||
|
||||
static bool IsNZCV(unsigned BitOffset) {
|
||||
return ContainsNZCV(1U << BitOffset);
|
||||
return BitOffset < 32 && ContainsNZCV(1U << BitOffset);
|
||||
}
|
||||
|
||||
OrderedNode* CachedNZCV {};
|
||||
Ref CachedNZCV {};
|
||||
bool NZCVDirty {};
|
||||
uint32_t PossiblySetNZCVBits {};
|
||||
|
||||
@@ -1034,7 +1046,7 @@ private:
|
||||
|
||||
// Opcode helpers for generalizing behavior across VEX and non-VEX variants.
|
||||
|
||||
OrderedNode* ADDSUBPOpImpl(OpcodeArgs, size_t ElementSize, OrderedNode* Src1, OrderedNode* Src2);
|
||||
Ref ADDSUBPOpImpl(OpcodeArgs, size_t ElementSize, Ref Src1, Ref Src2);
|
||||
|
||||
void AVXVectorALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void AVXVectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
@@ -1044,73 +1056,71 @@ private:
|
||||
|
||||
void AVXVariableShiftImpl(OpcodeArgs, IROps IROp);
|
||||
|
||||
OrderedNode* AESKeyGenAssistImpl(OpcodeArgs);
|
||||
Ref AESKeyGenAssistImpl(OpcodeArgs);
|
||||
|
||||
OrderedNode*
|
||||
CVTGPR_To_FPRImpl(OpcodeArgs, size_t DstElementSize, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op);
|
||||
Ref CVTGPR_To_FPRImpl(OpcodeArgs, size_t DstElementSize, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op);
|
||||
|
||||
OrderedNode* DPPOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm, size_t ElementSize);
|
||||
Ref DPPOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm, size_t ElementSize);
|
||||
|
||||
OrderedNode* VDPPSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm);
|
||||
Ref VDPPSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2, const X86Tables::DecodedOperand& Imm);
|
||||
|
||||
OrderedNode* ExtendVectorElementsImpl(OpcodeArgs, size_t ElementSize, size_t DstElementSize, bool Signed);
|
||||
Ref ExtendVectorElementsImpl(OpcodeArgs, size_t ElementSize, size_t DstElementSize, bool Signed);
|
||||
|
||||
OrderedNode* HSUBPOpImpl(OpcodeArgs, size_t ElementSize, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op);
|
||||
Ref HSUBPOpImpl(OpcodeArgs, size_t ElementSize, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op);
|
||||
|
||||
OrderedNode* InsertPSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm);
|
||||
Ref InsertPSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm);
|
||||
|
||||
OrderedNode* MPSADBWOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op,
|
||||
const X86Tables::DecodedOperand& ImmOp);
|
||||
Ref MPSADBWOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op,
|
||||
const X86Tables::DecodedOperand& ImmOp);
|
||||
|
||||
OrderedNode* PACKSSOpImpl(OpcodeArgs, size_t ElementSize, OrderedNode* Src1, OrderedNode* Src2);
|
||||
Ref PACKSSOpImpl(OpcodeArgs, size_t ElementSize, Ref Src1, Ref Src2);
|
||||
|
||||
OrderedNode* PACKUSOpImpl(OpcodeArgs, size_t ElementSize, OrderedNode* Src1, OrderedNode* Src2);
|
||||
Ref PACKUSOpImpl(OpcodeArgs, size_t ElementSize, Ref Src1, Ref Src2);
|
||||
|
||||
OrderedNode* PALIGNROpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm, bool IsAVX);
|
||||
Ref PALIGNROpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm, bool IsAVX);
|
||||
|
||||
void PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask);
|
||||
|
||||
OrderedNode* PHADDSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2);
|
||||
Ref PHADDSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2);
|
||||
|
||||
OrderedNode* PHMINPOSUWOpImpl(OpcodeArgs);
|
||||
Ref PHMINPOSUWOpImpl(OpcodeArgs);
|
||||
|
||||
OrderedNode* PHSUBOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2, size_t ElementSize);
|
||||
Ref PHSUBOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2, size_t ElementSize);
|
||||
|
||||
OrderedNode* PHSUBSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op);
|
||||
Ref PHSUBSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op);
|
||||
|
||||
OrderedNode* PINSROpImpl(OpcodeArgs, size_t ElementSize, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op,
|
||||
const X86Tables::DecodedOperand& Imm);
|
||||
Ref PINSROpImpl(OpcodeArgs, size_t ElementSize, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op,
|
||||
const X86Tables::DecodedOperand& Imm);
|
||||
|
||||
OrderedNode* PMADDWDOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2);
|
||||
Ref PMADDWDOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2);
|
||||
|
||||
OrderedNode* PMADDUBSWOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op);
|
||||
Ref PMADDUBSWOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op);
|
||||
|
||||
OrderedNode* PMULHRSWOpImpl(OpcodeArgs, OrderedNode* Src1, OrderedNode* Src2);
|
||||
Ref PMULHRSWOpImpl(OpcodeArgs, Ref Src1, Ref Src2);
|
||||
|
||||
OrderedNode* PMULHWOpImpl(OpcodeArgs, bool Signed, OrderedNode* Src1, OrderedNode* Src2);
|
||||
Ref PMULHWOpImpl(OpcodeArgs, bool Signed, Ref Src1, Ref Src2);
|
||||
|
||||
OrderedNode* PMULLOpImpl(OpcodeArgs, size_t ElementSize, bool Signed, OrderedNode* Src1, OrderedNode* Src2);
|
||||
Ref PMULLOpImpl(OpcodeArgs, size_t ElementSize, bool Signed, Ref Src1, Ref Src2);
|
||||
|
||||
OrderedNode* PSADBWOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op);
|
||||
Ref PSADBWOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op);
|
||||
|
||||
OrderedNode* PSHUFBOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2);
|
||||
Ref PSHUFBOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2);
|
||||
|
||||
OrderedNode* PSIGNImpl(OpcodeArgs, size_t ElementSize, OrderedNode* Src1, OrderedNode* Src2);
|
||||
Ref PSIGNImpl(OpcodeArgs, size_t ElementSize, Ref Src1, Ref Src2);
|
||||
|
||||
OrderedNode* PSLLIImpl(OpcodeArgs, size_t ElementSize, OrderedNode* Src, uint64_t Shift);
|
||||
Ref PSLLIImpl(OpcodeArgs, size_t ElementSize, Ref Src, uint64_t Shift);
|
||||
|
||||
OrderedNode* PSLLImpl(OpcodeArgs, size_t ElementSize, OrderedNode* Src, OrderedNode* ShiftVec);
|
||||
Ref PSLLImpl(OpcodeArgs, size_t ElementSize, Ref Src, Ref ShiftVec);
|
||||
|
||||
OrderedNode* PSRAOpImpl(OpcodeArgs, size_t ElementSize, OrderedNode* Src, OrderedNode* ShiftVec);
|
||||
Ref PSRAOpImpl(OpcodeArgs, size_t ElementSize, Ref Src, Ref ShiftVec);
|
||||
|
||||
OrderedNode* PSRLDOpImpl(OpcodeArgs, size_t ElementSize, OrderedNode* Src, OrderedNode* ShiftVec);
|
||||
Ref PSRLDOpImpl(OpcodeArgs, size_t ElementSize, Ref Src, Ref ShiftVec);
|
||||
|
||||
OrderedNode* SHUFOpImpl(OpcodeArgs, size_t ElementSize, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm);
|
||||
Ref SHUFOpImpl(OpcodeArgs, size_t ElementSize, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm);
|
||||
|
||||
void VMASKMOVOpImpl(OpcodeArgs, size_t ElementSize, size_t DataSize, bool IsStore, const X86Tables::DecodedOperand& MaskOp,
|
||||
const X86Tables::DecodedOperand& DataOp);
|
||||
@@ -1118,7 +1128,7 @@ private:
|
||||
void MOVScalarOpImpl(OpcodeArgs, size_t ElementSize);
|
||||
void VMOVScalarOpImpl(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
OrderedNode* VFCMPOpImpl(OpcodeArgs, size_t ElementSize, OrderedNode* Src1, OrderedNode* Src2, uint8_t CompType);
|
||||
Ref VFCMPOpImpl(OpcodeArgs, size_t ElementSize, Ref Src1, Ref Src2, uint8_t CompType);
|
||||
|
||||
void VTESTOpImpl(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
@@ -1137,92 +1147,95 @@ private:
|
||||
// - Example 32bit ADDSS Dest, Src
|
||||
// - Dest[31:0] = Dest[31:0] + Src[31:0]
|
||||
// - Dest[{256,128}:32] = (Unmodified)
|
||||
OrderedNode* VectorScalarInsertALUOpImpl(OpcodeArgs, IROps IROp, size_t DstSize, size_t ElementSize, const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op, bool ZeroUpperBits);
|
||||
Ref VectorScalarInsertALUOpImpl(OpcodeArgs, IROps IROp, size_t DstSize, size_t ElementSize, const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op, bool ZeroUpperBits);
|
||||
|
||||
OrderedNode* VectorScalarUnaryInsertALUOpImpl(OpcodeArgs, IROps IROp, size_t DstSize, size_t ElementSize, const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op, bool ZeroUpperBits);
|
||||
|
||||
OrderedNode* InsertCVTGPR_To_FPRImpl(OpcodeArgs, size_t DstSize, size_t DstElementSize, const X86Tables::DecodedOperand& Src1Op,
|
||||
Ref VectorScalarUnaryInsertALUOpImpl(OpcodeArgs, IROps IROp, size_t DstSize, size_t ElementSize, const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op, bool ZeroUpperBits);
|
||||
|
||||
OrderedNode* InsertScalar_CVT_Float_To_FloatImpl(OpcodeArgs, size_t DstSize, size_t DstElementSize, size_t SrcElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op,
|
||||
bool ZeroUpperBits);
|
||||
OrderedNode* InsertScalarRoundImpl(OpcodeArgs, size_t DstSize, size_t ElementSize, const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op, uint64_t Mode, bool ZeroUpperBits);
|
||||
Ref InsertCVTGPR_To_FPRImpl(OpcodeArgs, size_t DstSize, size_t DstElementSize, const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op, bool ZeroUpperBits);
|
||||
|
||||
OrderedNode* InsertScalarFCMPOpImpl(OpcodeArgs, size_t DstSize, size_t ElementSize, const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op, uint8_t CompType, bool ZeroUpperBits);
|
||||
Ref InsertScalar_CVT_Float_To_FloatImpl(OpcodeArgs, size_t DstSize, size_t DstElementSize, size_t SrcElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op, bool ZeroUpperBits);
|
||||
Ref InsertScalarRoundImpl(OpcodeArgs, size_t DstSize, size_t ElementSize, const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op, uint64_t Mode, bool ZeroUpperBits);
|
||||
|
||||
OrderedNode* VectorRoundImpl(OpcodeArgs, size_t ElementSize, OrderedNode* Src, uint64_t Mode);
|
||||
Ref InsertScalarFCMPOpImpl(OpcodeArgs, size_t DstSize, size_t ElementSize, const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op, uint8_t CompType, bool ZeroUpperBits);
|
||||
|
||||
OrderedNode* Scalar_CVT_Float_To_FloatImpl(OpcodeArgs, size_t DstElementSize, size_t SrcElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op);
|
||||
Ref VectorRoundImpl(OpcodeArgs, size_t ElementSize, Ref Src, uint64_t Mode);
|
||||
|
||||
Ref Scalar_CVT_Float_To_FloatImpl(OpcodeArgs, size_t DstElementSize, size_t SrcElementSize, const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op);
|
||||
|
||||
void Vector_CVT_Float_To_FloatImpl(OpcodeArgs, size_t DstElementSize, size_t SrcElementSize, bool IsAVX);
|
||||
|
||||
OrderedNode* Vector_CVT_Float_To_IntImpl(OpcodeArgs, size_t SrcElementSize, bool Narrow, bool HostRoundingMode);
|
||||
Ref Vector_CVT_Float_To_IntImpl(OpcodeArgs, size_t SrcElementSize, bool Narrow, bool HostRoundingMode);
|
||||
|
||||
OrderedNode* Vector_CVT_Int_To_FloatImpl(OpcodeArgs, size_t SrcElementSize, bool Widen);
|
||||
Ref Vector_CVT_Int_To_FloatImpl(OpcodeArgs, size_t SrcElementSize, bool Widen);
|
||||
|
||||
void XSaveOpImpl(OpcodeArgs);
|
||||
void SaveX87State(OpcodeArgs, OrderedNode* MemBase);
|
||||
void SaveSSEState(OrderedNode* MemBase);
|
||||
void SaveMXCSRState(OrderedNode* MemBase);
|
||||
void SaveAVXState(OrderedNode* MemBase);
|
||||
void SaveX87State(OpcodeArgs, Ref MemBase);
|
||||
void SaveSSEState(Ref MemBase);
|
||||
void SaveMXCSRState(Ref MemBase);
|
||||
void SaveAVXState(Ref MemBase);
|
||||
|
||||
void XRstorOpImpl(OpcodeArgs);
|
||||
void RestoreX87State(OrderedNode* MemBase);
|
||||
void RestoreSSEState(OrderedNode* MemBase);
|
||||
void RestoreMXCSRState(OrderedNode* MXCSR);
|
||||
void RestoreAVXState(OrderedNode* MemBase);
|
||||
void RestoreX87State(Ref MemBase);
|
||||
void RestoreSSEState(Ref MemBase);
|
||||
void RestoreMXCSRState(Ref MXCSR);
|
||||
void RestoreAVXState(Ref MemBase);
|
||||
void DefaultX87State(OpcodeArgs);
|
||||
void DefaultSSEState();
|
||||
void DefaultAVXState();
|
||||
|
||||
OrderedNode* GetMXCSR();
|
||||
Ref GetMXCSR();
|
||||
|
||||
#undef OpcodeArgs
|
||||
|
||||
OrderedNode* AppendSegmentOffset(OrderedNode* Value, uint32_t Flags, uint32_t DefaultPrefix = 0, bool Override = false);
|
||||
OrderedNode* GetSegment(uint32_t Flags, uint32_t DefaultPrefix = 0, bool Override = false);
|
||||
Ref AppendSegmentOffset(Ref Value, uint32_t Flags, uint32_t DefaultPrefix = 0, bool Override = false);
|
||||
Ref GetSegment(uint32_t Flags, uint32_t DefaultPrefix = 0, bool Override = false);
|
||||
|
||||
void UpdatePrefixFromSegment(OrderedNode* Segment, uint32_t SegmentReg);
|
||||
void UpdatePrefixFromSegment(Ref Segment, uint32_t SegmentReg);
|
||||
|
||||
OrderedNode* LoadGPRRegister(uint32_t GPR, int8_t Size = -1, uint8_t Offset = 0, bool AllowUpperGarbage = false);
|
||||
OrderedNode* LoadXMMRegister(uint32_t XMM);
|
||||
void StoreGPRRegister(uint32_t GPR, OrderedNode* const Src, int8_t Size = -1, uint8_t Offset = 0);
|
||||
void StoreXMMRegister(uint32_t XMM, OrderedNode* const Src);
|
||||
Ref LoadGPRRegister(uint32_t GPR, int8_t Size = -1, uint8_t Offset = 0, bool AllowUpperGarbage = false);
|
||||
Ref LoadXMMRegister(uint32_t XMM);
|
||||
void StoreGPRRegister(uint32_t GPR, const Ref Src, int8_t Size = -1, uint8_t Offset = 0);
|
||||
void StoreXMMRegister(uint32_t XMM, const Ref Src);
|
||||
|
||||
OrderedNode* GetRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset = 0);
|
||||
Ref GetRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset = 0);
|
||||
|
||||
OrderedNode* LoadSource(RegisterClassType Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags,
|
||||
const LoadSourceOptions& Options = {});
|
||||
OrderedNode* LoadSource_WithOpSize(RegisterClassType Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand,
|
||||
uint8_t OpSize, uint32_t Flags, const LoadSourceOptions& Options = {});
|
||||
AddressMode AddSegmentToAddress(AddressMode A, uint32_t Flags);
|
||||
Ref LoadEffectiveAddress(AddressMode A, bool AllowUpperGarbage = false);
|
||||
AddressMode SelectAddressMode(AddressMode A, bool AtomicTSO, bool Vector, unsigned AccessSize);
|
||||
|
||||
Ref LoadSource(RegisterClassType Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags,
|
||||
const LoadSourceOptions& Options = {});
|
||||
Ref LoadSource_WithOpSize(RegisterClassType Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand,
|
||||
uint8_t OpSize, uint32_t Flags, const LoadSourceOptions& Options = {});
|
||||
void StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op,
|
||||
const FEXCore::X86Tables::DecodedOperand& Operand, OrderedNode* const Src, uint8_t OpSize, int8_t Align,
|
||||
const FEXCore::X86Tables::DecodedOperand& Operand, const Ref Src, uint8_t OpSize, int8_t Align,
|
||||
MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
|
||||
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, const FEXCore::X86Tables::DecodedOperand& Operand,
|
||||
OrderedNode* const Src, int8_t Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
|
||||
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, OrderedNode* const Src, int8_t Align,
|
||||
const Ref Src, int8_t Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
|
||||
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, const Ref Src, int8_t Align,
|
||||
MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
|
||||
|
||||
// In several instances, it's desirable to get a base address with the segment offset
|
||||
// applied to it. This pulls all the common-case appending into a single set of functions.
|
||||
[[nodiscard]]
|
||||
OrderedNode* MakeSegmentAddress(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint8_t OpSize) {
|
||||
OrderedNode* Mem = LoadSource_WithOpSize(GPRClass, Op, Operand, OpSize, Op->Flags, {.LoadData = false});
|
||||
Ref MakeSegmentAddress(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint8_t OpSize) {
|
||||
Ref Mem = LoadSource_WithOpSize(GPRClass, Op, Operand, OpSize, Op->Flags, {.LoadData = false});
|
||||
return AppendSegmentOffset(Mem, Op->Flags);
|
||||
}
|
||||
[[nodiscard]]
|
||||
OrderedNode* MakeSegmentAddress(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand) {
|
||||
Ref MakeSegmentAddress(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand) {
|
||||
return MakeSegmentAddress(Op, Operand, GetSrcSize(Op));
|
||||
}
|
||||
[[nodiscard]]
|
||||
OrderedNode* MakeSegmentAddress(X86State::X86Reg Reg, uint32_t Flags, uint32_t DefaultPrefix = 0, bool Override = false) {
|
||||
OrderedNode* Address = LoadGPRRegister(Reg);
|
||||
Ref MakeSegmentAddress(X86State::X86Reg Reg, uint32_t Flags, uint32_t DefaultPrefix = 0, bool Override = false) {
|
||||
Ref Address = LoadGPRRegister(Reg);
|
||||
return AppendSegmentOffset(Address, Flags, DefaultPrefix, Override);
|
||||
}
|
||||
|
||||
@@ -1302,7 +1315,7 @@ private:
|
||||
HandleNZCVWrite((1u << 31) | (1u << 30));
|
||||
}
|
||||
|
||||
OrderedNode* GetNZCV() {
|
||||
Ref GetNZCV() {
|
||||
if (!CachedNZCV) {
|
||||
CachedNZCV = _LoadNZCV();
|
||||
}
|
||||
@@ -1310,7 +1323,7 @@ private:
|
||||
return CachedNZCV;
|
||||
}
|
||||
|
||||
void SetNZCV(OrderedNode* Value) {
|
||||
void SetNZCV(Ref Value) {
|
||||
CachedNZCV = Value;
|
||||
NZCVDirty = true;
|
||||
}
|
||||
@@ -1321,22 +1334,12 @@ private:
|
||||
NZCVDirty = true;
|
||||
}
|
||||
|
||||
void ZeroCV() {
|
||||
// Get old NZCV before we mess with PossiblySetNZCVBits
|
||||
auto OldNZCV = GetNZCV();
|
||||
|
||||
// Mask out the NZ bits, clearing CV. Even if the code sets CV after, this can end up faster
|
||||
// moves by allowing orlshl to be used instead of bfi.
|
||||
PossiblySetNZCVBits = (1u << IndexNZCV(FEXCore::X86State::RFLAG_SF_RAW_LOC)) | (1u << IndexNZCV(FEXCore::X86State::RFLAG_ZF_RAW_LOC));
|
||||
SetNZCV(_And(OpSize::i32Bit, OldNZCV, _Constant(PossiblySetNZCVBits)));
|
||||
}
|
||||
|
||||
void SetNZ_ZeroCV(unsigned SrcSize, OrderedNode* Res) {
|
||||
void SetNZ_ZeroCV(unsigned SrcSize, Ref Res) {
|
||||
HandleNZ00Write();
|
||||
_TestNZ(IR::SizeToOpSize(SrcSize), Res, Res);
|
||||
}
|
||||
|
||||
void InsertNZCV(unsigned BitOffset, OrderedNode* Value, signed FlagOffset, bool MustMask) {
|
||||
void InsertNZCV(unsigned BitOffset, Ref Value, signed FlagOffset, bool MustMask) {
|
||||
signed Bit = IndexNZCV(BitOffset);
|
||||
|
||||
// If NZCV is not dirty, we always want to use rmif, it's 1 instruction to
|
||||
@@ -1394,17 +1397,17 @@ private:
|
||||
}
|
||||
|
||||
template<unsigned BitOffset>
|
||||
void SetRFLAG(OrderedNode* Value, unsigned ValueOffset = 0, bool MustMask = false) {
|
||||
void SetRFLAG(Ref Value, unsigned ValueOffset = 0, bool MustMask = false) {
|
||||
SetRFLAG(Value, BitOffset, ValueOffset, MustMask);
|
||||
}
|
||||
|
||||
void SetRFLAG(OrderedNode* Value, unsigned BitOffset, unsigned ValueOffset = 0, bool MustMask = false) {
|
||||
void SetRFLAG(Ref Value, unsigned BitOffset, unsigned ValueOffset = 0, bool MustMask = false) {
|
||||
if (IsNZCV(BitOffset)) {
|
||||
InsertNZCV(BitOffset, Value, ValueOffset, MustMask);
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) {
|
||||
_StoreRegister(Value, false, offsetof(FEXCore::Core::CPUState, pf_raw), GPRClass, GPRFixedClass, CTX->GetGPRSize());
|
||||
_StoreRegister(Value, Core::CPUState::PF_AS_GREG, GPRClass, CTX->GetGPRSize());
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_AF_RAW_LOC) {
|
||||
_StoreRegister(Value, false, offsetof(FEXCore::Core::CPUState, af_raw), GPRClass, GPRFixedClass, CTX->GetGPRSize());
|
||||
_StoreRegister(Value, Core::CPUState::AF_AS_GREG, GPRClass, CTX->GetGPRSize());
|
||||
} else {
|
||||
if (ValueOffset || MustMask) {
|
||||
Value = _Bfe(OpSize::i32Bit, 1, ValueOffset, Value);
|
||||
@@ -1443,7 +1446,7 @@ private:
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode* GetRFLAG(unsigned BitOffset, bool Invert = false) {
|
||||
Ref GetRFLAG(unsigned BitOffset, bool Invert = false) {
|
||||
if (IsNZCV(BitOffset)) {
|
||||
if (!(PossiblySetNZCVBits & (1u << IndexNZCV(BitOffset)))) {
|
||||
return _Constant(Invert ? 1 : 0);
|
||||
@@ -1459,9 +1462,9 @@ private:
|
||||
return _NZCVSelect(OpSize::i32Bit, CondForNZCVBit(BitOffset, Invert), _Constant(1), _Constant(0));
|
||||
}
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) {
|
||||
return _LoadRegister(false, offsetof(FEXCore::Core::CPUState, pf_raw), GPRClass, GPRFixedClass, CTX->GetGPRSize());
|
||||
return _LoadRegister(Core::CPUState::PF_AS_GREG, GPRClass, CTX->GetGPRSize());
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_AF_RAW_LOC) {
|
||||
return _LoadRegister(false, offsetof(FEXCore::Core::CPUState, af_raw), GPRClass, GPRFixedClass, CTX->GetGPRSize());
|
||||
return _LoadRegister(Core::CPUState::AF_AS_GREG, GPRClass, CTX->GetGPRSize());
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_DF_RAW_LOC) {
|
||||
// Recover the sign bit, it is the logical DF value
|
||||
return _Lshr(OpSize::i64Bit, _LoadDF(), _Constant(63));
|
||||
@@ -1471,7 +1474,7 @@ private:
|
||||
}
|
||||
|
||||
// Returns (DF ? -Size : Size)
|
||||
OrderedNode* LoadDir(const unsigned Size) {
|
||||
Ref LoadDir(const unsigned Size) {
|
||||
auto Dir = _LoadDF();
|
||||
auto Shift = FEXCore::ilog2(Size);
|
||||
|
||||
@@ -1483,7 +1486,7 @@ private:
|
||||
}
|
||||
|
||||
// Returns DF ? (X - Size) : (X + Size)
|
||||
OrderedNode* OffsetByDir(OrderedNode* X, const unsigned Size) {
|
||||
Ref OffsetByDir(Ref X, const unsigned Size) {
|
||||
auto Shift = FEXCore::ilog2(Size);
|
||||
|
||||
return _AddShift(OpSize::i64Bit, X, _LoadDF(), ShiftType::LSL, Shift);
|
||||
@@ -1505,7 +1508,7 @@ private:
|
||||
//
|
||||
// We set PF to unordered (V), but our PF representation is inverted so we
|
||||
// actually set to !V. This is one instruction with the VC cond code.
|
||||
OrderedNode* PFInvert = _NZCVSelect(OpSize::i32Bit, CondClassType {COND_FNU}, _Constant(1), _Constant(0));
|
||||
Ref PFInvert = _NZCVSelect(OpSize::i32Bit, CondClassType {COND_FNU}, _Constant(1), _Constant(0));
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(PFInvert);
|
||||
|
||||
@@ -1518,9 +1521,9 @@ private:
|
||||
// now, add a cfinv to deal. Hopefully we delete this later.
|
||||
CarryInvert();
|
||||
} else {
|
||||
OrderedNode* Z = GetRFLAG(FEXCore::X86State::RFLAG_ZF_RAW_LOC);
|
||||
OrderedNode* C_inv = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC, true);
|
||||
OrderedNode* V = GetRFLAG(FEXCore::X86State::RFLAG_OF_RAW_LOC);
|
||||
Ref Z = GetRFLAG(FEXCore::X86State::RFLAG_ZF_RAW_LOC);
|
||||
Ref C_inv = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC, true);
|
||||
Ref V = GetRFLAG(FEXCore::X86State::RFLAG_OF_RAW_LOC);
|
||||
|
||||
// We want to zero SF/OF, and then set CF/ZF. Zeroing up front lets us do
|
||||
// this all with shifted-or's on non-flagm platforms.
|
||||
@@ -1539,7 +1542,7 @@ private:
|
||||
// Set x87 comparison flags based on the result set by Arm FCMP. Clobbers
|
||||
// NZCV on flagm2 platforms.
|
||||
void ConvertNZCVToX87() {
|
||||
OrderedNode* V = GetRFLAG(FEXCore::X86State::RFLAG_OF_RAW_LOC);
|
||||
Ref V = GetRFLAG(FEXCore::X86State::RFLAG_OF_RAW_LOC);
|
||||
|
||||
if (CTX->HostFeatures.SupportsFlagM2) {
|
||||
LOGMAN_THROW_A_FMT(!NZCVDirty, "only expected after fcmp");
|
||||
@@ -1552,8 +1555,8 @@ private:
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC, true));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(GetRFLAG(FEXCore::X86State::RFLAG_ZF_RAW_LOC));
|
||||
} else {
|
||||
OrderedNode* Z = GetRFLAG(FEXCore::X86State::RFLAG_ZF_RAW_LOC);
|
||||
OrderedNode* N = GetRFLAG(FEXCore::X86State::RFLAG_SF_RAW_LOC);
|
||||
Ref Z = GetRFLAG(FEXCore::X86State::RFLAG_ZF_RAW_LOC);
|
||||
Ref N = GetRFLAG(FEXCore::X86State::RFLAG_SF_RAW_LOC);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(_Or(OpSize::i32Bit, N, V));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(_Or(OpSize::i32Bit, Z, V));
|
||||
@@ -1565,28 +1568,35 @@ private:
|
||||
|
||||
// Helper to store a variable shift and calculate its flags for a variable
|
||||
// shift, with correct PF handling.
|
||||
void HandleShift(X86Tables::DecodedOp Op, OrderedNode* Result, OrderedNode* Dest, ShiftType Shift, OrderedNode* Src) {
|
||||
|
||||
StoreResult(GPRClass, Op, Result, -1);
|
||||
void HandleShift(X86Tables::DecodedOp Op, Ref Result, Ref Dest, ShiftType Shift, Ref Src) {
|
||||
|
||||
auto OldPF = GetRFLAG(X86State::RFLAG_PF_RAW_LOC);
|
||||
|
||||
HandleNZCV_RMW();
|
||||
CalculatePF(_ShiftFlags(OpSizeFromSrc(Op), Result, Dest, Shift, Src, OldPF));
|
||||
StoreResult(GPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
std::pair<Ref, Ref> ExtractPair(OpSize Size, Ref Pair) {
|
||||
// Extract high first. This is a hack to improve coalescing.
|
||||
Ref Hi = _ExtractElementPair(Size, Pair, 1);
|
||||
Ref Lo = _ExtractElementPair(Size, Pair, 0);
|
||||
|
||||
return std::make_pair(Lo, Hi);
|
||||
}
|
||||
|
||||
// Helper to derive Dest by a given builder-using Expression with the opcode
|
||||
// replaced with NewOp. Useful for generic building code. Not safe in general.
|
||||
// but does the right handling of ImplicitFlagClobber at least and must be
|
||||
// used instead of raw Op mutation.
|
||||
#define DeriveOp(Dest, NewOp, Expr) \
|
||||
#define DeriveOp(Dest, NewOp, Expr) \
|
||||
if (ImplicitFlagClobber(NewOp)) SaveNZCV(NewOp); \
|
||||
auto Dest = (Expr); \
|
||||
auto Dest = (Expr); \
|
||||
Dest.first->Header.Op = (NewOp)
|
||||
|
||||
// Named constant cache for the current block.
|
||||
// Different arrays for sizes 1,2,4,8,16,32.
|
||||
OrderedNode* CachedNamedVectorConstants[FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_MAX][6] {};
|
||||
Ref CachedNamedVectorConstants[FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_MAX][6] {};
|
||||
struct IndexNamedVectorMapKey {
|
||||
uint32_t Index {};
|
||||
FEXCore::IR::IndexNamedVectorConstant NamedIndexedConstant;
|
||||
@@ -1600,10 +1610,10 @@ private:
|
||||
return XXH3_64bits(&k, sizeof(k));
|
||||
}
|
||||
};
|
||||
fextl::unordered_map<IndexNamedVectorMapKey, OrderedNode*, IndexNamedVectorMapKeyHasher> CachedIndexedNamedVectorConstants;
|
||||
fextl::unordered_map<IndexNamedVectorMapKey, Ref, IndexNamedVectorMapKeyHasher> CachedIndexedNamedVectorConstants;
|
||||
|
||||
// Load and cache a named vector constant.
|
||||
OrderedNode* LoadAndCacheNamedVectorConstant(uint8_t Size, FEXCore::IR::NamedVectorConstant NamedConstant) {
|
||||
Ref LoadAndCacheNamedVectorConstant(uint8_t Size, FEXCore::IR::NamedVectorConstant NamedConstant) {
|
||||
auto log2_size_bytes = FEXCore::ilog2(Size);
|
||||
if (CachedNamedVectorConstants[NamedConstant][log2_size_bytes]) {
|
||||
return CachedNamedVectorConstants[NamedConstant][log2_size_bytes];
|
||||
@@ -1613,7 +1623,7 @@ private:
|
||||
CachedNamedVectorConstants[NamedConstant][log2_size_bytes] = Constant;
|
||||
return Constant;
|
||||
}
|
||||
OrderedNode* LoadAndCacheIndexedNamedVectorConstant(uint8_t Size, FEXCore::IR::IndexNamedVectorConstant NamedIndexedConstant, uint32_t Index) {
|
||||
Ref LoadAndCacheIndexedNamedVectorConstant(uint8_t Size, FEXCore::IR::IndexNamedVectorConstant NamedIndexedConstant, uint32_t Index) {
|
||||
IndexNamedVectorMapKey Key {
|
||||
.Index = Index,
|
||||
.NamedIndexedConstant = NamedIndexedConstant,
|
||||
@@ -1638,8 +1648,8 @@ private:
|
||||
}
|
||||
|
||||
std::pair<bool, CondClassType> DecodeNZCVCondition(uint8_t OP) const;
|
||||
OrderedNode* SelectBit(OrderedNode* Cmp, IR::OpSize ResultSize, OrderedNode* TrueValue, OrderedNode* FalseValue);
|
||||
OrderedNode* SelectCC(uint8_t OP, IR::OpSize ResultSize, OrderedNode* TrueValue, OrderedNode* FalseValue);
|
||||
Ref SelectBit(Ref Cmp, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue);
|
||||
Ref SelectCC(uint8_t OP, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue);
|
||||
|
||||
/**
|
||||
* @name Deferred RFLAG calculation and generation.
|
||||
@@ -1668,7 +1678,7 @@ private:
|
||||
uint8_t SrcSize;
|
||||
|
||||
// Every flag generation type has a result
|
||||
OrderedNode* Res {};
|
||||
Ref Res {};
|
||||
|
||||
union {
|
||||
// UMUL, BEXTR, BLSI, POPCOUNT, ZCNT, RDRAND
|
||||
@@ -1677,25 +1687,25 @@ private:
|
||||
|
||||
// MUL, BLSR, BLSMSKB, BZHI
|
||||
struct {
|
||||
OrderedNode* Src1;
|
||||
Ref Src1;
|
||||
} OneSource;
|
||||
|
||||
// Logical
|
||||
struct {
|
||||
OrderedNode* Src1;
|
||||
OrderedNode* Src2;
|
||||
Ref Src1;
|
||||
Ref Src2;
|
||||
} TwoSource;
|
||||
|
||||
// LSHLI, LSHRI, ASHRI
|
||||
struct {
|
||||
OrderedNode* Src1;
|
||||
Ref Src1;
|
||||
uint64_t Imm;
|
||||
} OneSrcImmediate;
|
||||
|
||||
// ADD, SUB
|
||||
struct {
|
||||
OrderedNode* Src1;
|
||||
OrderedNode* Src2;
|
||||
Ref Src1;
|
||||
Ref Src2;
|
||||
|
||||
bool UpdateCF;
|
||||
} TwoSrcImmediate;
|
||||
@@ -1735,7 +1745,7 @@ private:
|
||||
}
|
||||
|
||||
template<typename F>
|
||||
void Calculate_ShiftVariable(OrderedNode* Shift, F&& Calculate) {
|
||||
void Calculate_ShiftVariable(Ref Shift, F&& Calculate) {
|
||||
// RCR can call this with constants, so handle that without branching.
|
||||
uint64_t Const;
|
||||
if (IsValueConstant(WrapNode(Shift), &Const)) {
|
||||
@@ -1770,37 +1780,37 @@ private:
|
||||
/**
|
||||
* @name These functions are used by the deferred flag handling while it is calculating and storing flags in to RFLAGs.
|
||||
* @{ */
|
||||
OrderedNode* LoadPFRaw(bool Invert);
|
||||
OrderedNode* LoadAF();
|
||||
Ref LoadPFRaw(bool Invert);
|
||||
Ref LoadAF();
|
||||
void FixupAF();
|
||||
void SetAFAndFixup(OrderedNode* AF);
|
||||
OrderedNode* CalculateAFForDecimal(OrderedNode* A);
|
||||
void CalculatePF(OrderedNode* Res);
|
||||
void CalculateAF(OrderedNode* Src1, OrderedNode* Src2);
|
||||
void SetAFAndFixup(Ref AF);
|
||||
Ref CalculateAFForDecimal(Ref A);
|
||||
void CalculatePF(Ref Res);
|
||||
void CalculateAF(Ref Src1, Ref Src2);
|
||||
|
||||
void CalculateOF(uint8_t SrcSize, OrderedNode* Res, OrderedNode* Src1, OrderedNode* Src2, bool Sub);
|
||||
OrderedNode* CalculateFlags_ADC(uint8_t SrcSize, OrderedNode* Src1, OrderedNode* Src2);
|
||||
OrderedNode* CalculateFlags_SBB(uint8_t SrcSize, OrderedNode* Src1, OrderedNode* Src2);
|
||||
OrderedNode* CalculateFlags_SUB(uint8_t SrcSize, OrderedNode* Src1, OrderedNode* Src2, bool UpdateCF = true);
|
||||
OrderedNode* CalculateFlags_ADD(uint8_t SrcSize, OrderedNode* Src1, OrderedNode* Src2, bool UpdateCF = true);
|
||||
void CalculateFlags_MUL(uint8_t SrcSize, OrderedNode* Res, OrderedNode* High);
|
||||
void CalculateFlags_UMUL(OrderedNode* High);
|
||||
void CalculateFlags_Logical(uint8_t SrcSize, OrderedNode* Res, OrderedNode* Src1, OrderedNode* Src2);
|
||||
void CalculateFlags_ShiftLeft(uint8_t SrcSize, OrderedNode* Res, OrderedNode* Src1, OrderedNode* Src2);
|
||||
void CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, OrderedNode* Res, OrderedNode* Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRight(uint8_t SrcSize, OrderedNode* Res, OrderedNode* Src1, OrderedNode* Src2);
|
||||
void CalculateFlags_ShiftRightImmediate(uint8_t SrcSize, OrderedNode* Res, OrderedNode* Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRightDoubleImmediate(uint8_t SrcSize, OrderedNode* Res, OrderedNode* Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize, OrderedNode* Res, OrderedNode* Src1, uint64_t Shift);
|
||||
void CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize, OrderedNode* Res, OrderedNode* Src1, uint64_t Shift);
|
||||
void CalculateFlags_BEXTR(OrderedNode* Src);
|
||||
void CalculateFlags_BLSI(uint8_t SrcSize, OrderedNode* Src);
|
||||
void CalculateFlags_BLSMSK(uint8_t SrcSize, OrderedNode* Res, OrderedNode* Src);
|
||||
void CalculateFlags_BLSR(uint8_t SrcSize, OrderedNode* Res, OrderedNode* Src);
|
||||
void CalculateFlags_POPCOUNT(OrderedNode* Src);
|
||||
void CalculateFlags_BZHI(uint8_t SrcSize, OrderedNode* Result, OrderedNode* Src);
|
||||
void CalculateFlags_ZCNT(uint8_t SrcSize, OrderedNode* Result);
|
||||
void CalculateFlags_RDRAND(OrderedNode* Src);
|
||||
void CalculateOF(uint8_t SrcSize, Ref Res, Ref Src1, Ref Src2, bool Sub);
|
||||
Ref CalculateFlags_ADC(uint8_t SrcSize, Ref Src1, Ref Src2);
|
||||
Ref CalculateFlags_SBB(uint8_t SrcSize, Ref Src1, Ref Src2);
|
||||
Ref CalculateFlags_SUB(uint8_t SrcSize, Ref Src1, Ref Src2, bool UpdateCF = true);
|
||||
Ref CalculateFlags_ADD(uint8_t SrcSize, Ref Src1, Ref Src2, bool UpdateCF = true);
|
||||
void CalculateFlags_MUL(uint8_t SrcSize, Ref Res, Ref High);
|
||||
void CalculateFlags_UMUL(Ref High);
|
||||
void CalculateFlags_Logical(uint8_t SrcSize, Ref Res, Ref Src1, Ref Src2);
|
||||
void CalculateFlags_ShiftLeft(uint8_t SrcSize, Ref Res, Ref Src1, Ref Src2);
|
||||
void CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRight(uint8_t SrcSize, Ref Res, Ref Src1, Ref Src2);
|
||||
void CalculateFlags_ShiftRightImmediate(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRightDoubleImmediate(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
void CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
void CalculateFlags_BEXTR(Ref Src);
|
||||
void CalculateFlags_BLSI(uint8_t SrcSize, Ref Src);
|
||||
void CalculateFlags_BLSMSK(uint8_t SrcSize, Ref Res, Ref Src);
|
||||
void CalculateFlags_BLSR(uint8_t SrcSize, Ref Res, Ref Src);
|
||||
void CalculateFlags_POPCOUNT(Ref Src);
|
||||
void CalculateFlags_BZHI(uint8_t SrcSize, Ref Result, Ref Src);
|
||||
void CalculateFlags_ZCNT(uint8_t SrcSize, Ref Result);
|
||||
void CalculateFlags_RDRAND(Ref Src);
|
||||
/** @} */
|
||||
|
||||
/**
|
||||
@@ -1808,7 +1818,7 @@ private:
|
||||
*
|
||||
* Depending on the operation it may force a RFLAGs calculation before storing the new deferred state.
|
||||
* @{ */
|
||||
void GenerateFlags_SUB(FEXCore::X86Tables::DecodedOp Op, OrderedNode* Src1, OrderedNode* Src2, bool UpdateCF = true) {
|
||||
void GenerateFlags_SUB(FEXCore::X86Tables::DecodedOp Op, Ref Src1, Ref Src2, bool UpdateCF = true) {
|
||||
if (!UpdateCF) {
|
||||
// If we aren't updating CF then we need to calculate flags. Invalidation mask would make this not required.
|
||||
CalculateDeferredFlags();
|
||||
@@ -1828,7 +1838,7 @@ private:
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_MUL(FEXCore::X86Tables::DecodedOp Op, OrderedNode* Res, OrderedNode* High) {
|
||||
void GenerateFlags_MUL(FEXCore::X86Tables::DecodedOp Op, Ref Res, Ref High) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_MUL,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
@@ -1843,7 +1853,7 @@ private:
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_UMUL(FEXCore::X86Tables::DecodedOp Op, OrderedNode* High) {
|
||||
void GenerateFlags_UMUL(FEXCore::X86Tables::DecodedOp Op, Ref High) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_UMUL,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
@@ -1851,7 +1861,7 @@ private:
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_Logical(FEXCore::X86Tables::DecodedOp Op, OrderedNode* Res, OrderedNode* Src1, OrderedNode* Src2) {
|
||||
void GenerateFlags_Logical(FEXCore::X86Tables::DecodedOp Op, Ref Res, Ref Src1, Ref Src2) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_LOGICAL,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
@@ -1867,7 +1877,7 @@ private:
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_ShiftLeftImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode* Res, OrderedNode* Src1, uint64_t Shift) {
|
||||
void GenerateFlags_ShiftLeftImmediate(FEXCore::X86Tables::DecodedOp Op, Ref Res, Ref Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero.
|
||||
if (Shift == 0) {
|
||||
return;
|
||||
@@ -1888,7 +1898,7 @@ private:
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_SignShiftRightImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode* Res, OrderedNode* Src1, uint64_t Shift) {
|
||||
void GenerateFlags_SignShiftRightImmediate(FEXCore::X86Tables::DecodedOp Op, Ref Res, Ref Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero.
|
||||
if (Shift == 0) {
|
||||
return;
|
||||
@@ -1909,7 +1919,7 @@ private:
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_ShiftRightImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode* Res, OrderedNode* Src1, uint64_t Shift) {
|
||||
void GenerateFlags_ShiftRightImmediate(FEXCore::X86Tables::DecodedOp Op, Ref Res, Ref Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero.
|
||||
if (Shift == 0) {
|
||||
return;
|
||||
@@ -1930,7 +1940,7 @@ private:
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_ShiftRightDoubleImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode* Res, OrderedNode* Src1, uint64_t Shift) {
|
||||
void GenerateFlags_ShiftRightDoubleImmediate(FEXCore::X86Tables::DecodedOp Op, Ref Res, Ref Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero.
|
||||
if (Shift == 0) {
|
||||
return;
|
||||
@@ -1951,7 +1961,7 @@ private:
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_BEXTR(FEXCore::X86Tables::DecodedOp Op, OrderedNode* Src) {
|
||||
void GenerateFlags_BEXTR(FEXCore::X86Tables::DecodedOp Op, Ref Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_BEXTR,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
@@ -1959,7 +1969,7 @@ private:
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_BLSI(FEXCore::X86Tables::DecodedOp Op, OrderedNode* Src) {
|
||||
void GenerateFlags_BLSI(FEXCore::X86Tables::DecodedOp Op, Ref Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_BLSI,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
@@ -1967,7 +1977,7 @@ private:
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_BLSMSK(FEXCore::X86Tables::DecodedOp Op, OrderedNode* Res, OrderedNode* Src) {
|
||||
void GenerateFlags_BLSMSK(FEXCore::X86Tables::DecodedOp Op, Ref Res, Ref Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_BLSMSK,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
@@ -1982,7 +1992,7 @@ private:
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_BLSR(FEXCore::X86Tables::DecodedOp Op, OrderedNode* Res, OrderedNode* Src) {
|
||||
void GenerateFlags_BLSR(FEXCore::X86Tables::DecodedOp Op, Ref Res, Ref Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_BLSR,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
@@ -1997,7 +2007,7 @@ private:
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_POPCOUNT(FEXCore::X86Tables::DecodedOp Op, OrderedNode* Src) {
|
||||
void GenerateFlags_POPCOUNT(FEXCore::X86Tables::DecodedOp Op, Ref Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_POPCOUNT,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
@@ -2005,7 +2015,7 @@ private:
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_BZHI(FEXCore::X86Tables::DecodedOp Op, OrderedNode* Result, OrderedNode* Src) {
|
||||
void GenerateFlags_BZHI(FEXCore::X86Tables::DecodedOp Op, Ref Result, Ref Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_BZHI,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
@@ -2020,7 +2030,7 @@ private:
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_ZCNT(FEXCore::X86Tables::DecodedOp Op, OrderedNode* Src) {
|
||||
void GenerateFlags_ZCNT(FEXCore::X86Tables::DecodedOp Op, Ref Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_ZCNT,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
@@ -2028,7 +2038,7 @@ private:
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_RDRAND(FEXCore::X86Tables::DecodedOp Op, OrderedNode* Src) {
|
||||
void GenerateFlags_RDRAND(FEXCore::X86Tables::DecodedOp Op, Ref Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_RDRAND,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
@@ -2036,7 +2046,7 @@ private:
|
||||
};
|
||||
}
|
||||
|
||||
OrderedNode* AndConst(FEXCore::IR::OpSize Size, OrderedNode* Node, uint64_t Const) {
|
||||
Ref AndConst(FEXCore::IR::OpSize Size, Ref Node, uint64_t Const) {
|
||||
uint64_t NodeConst;
|
||||
|
||||
if (IsValueConstant(WrapNode(Node), &NodeConst)) {
|
||||
@@ -2049,14 +2059,14 @@ private:
|
||||
/** @} */
|
||||
/** @} */
|
||||
|
||||
OrderedNode* GetX87Top();
|
||||
void SetX87ValidTag(OrderedNode* Value, bool Valid);
|
||||
OrderedNode* GetX87ValidTag(OrderedNode* Value);
|
||||
OrderedNode* GetX87Tag(OrderedNode* Value, OrderedNode* AbridgedFTW);
|
||||
OrderedNode* GetX87Tag(OrderedNode* Value);
|
||||
void SetX87FTW(OrderedNode* FTW);
|
||||
OrderedNode* GetX87FTW();
|
||||
void SetX87Top(OrderedNode* Value);
|
||||
Ref GetX87Top();
|
||||
void SetX87ValidTag(Ref Value, bool Valid);
|
||||
Ref GetX87ValidTag(Ref Value);
|
||||
Ref GetX87Tag(Ref Value, Ref AbridgedFTW);
|
||||
Ref GetX87Tag(Ref Value);
|
||||
void SetX87FTW(Ref FTW);
|
||||
Ref GetX87FTW();
|
||||
void SetX87Top(Ref Value);
|
||||
|
||||
bool DestIsLockedMem(FEXCore::X86Tables::DecodedOp Op) const {
|
||||
return DestIsMem(Op) && (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK) != 0;
|
||||
@@ -2072,7 +2082,7 @@ private:
|
||||
bool Multiblock {};
|
||||
uint64_t Entry;
|
||||
|
||||
OrderedNode* _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode* Addr, OrderedNode* Value, uint8_t Align = 1) {
|
||||
Ref _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, Ref Addr, Ref Value, uint8_t Align = 1) {
|
||||
if (CTX->IsAtomicTSOEnabled()) {
|
||||
return _StoreMemTSO(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
} else {
|
||||
@@ -2080,7 +2090,7 @@ private:
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode* _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode* ssa0, uint8_t Align = 1) {
|
||||
Ref _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, Ref ssa0, uint8_t Align = 1) {
|
||||
if (CTX->IsAtomicTSOEnabled()) {
|
||||
return _LoadMemTSO(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
} else {
|
||||
@@ -2088,7 +2098,29 @@ private:
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode* Prefetch(bool ForStore, bool Stream, uint8_t CacheLevel, OrderedNode* ssa0) {
|
||||
Ref _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, AddressMode A, uint8_t Align = 1, bool ForceNonTSO = false) {
|
||||
bool AtomicTSO = CTX->IsAtomicTSOEnabled() && !ForceNonTSO;
|
||||
A = SelectAddressMode(A, AtomicTSO, Class != GPRClass, Size);
|
||||
|
||||
if (AtomicTSO) {
|
||||
return _LoadMemTSO(Class, Size, A.Base, A.Index, Align, A.IndexType, A.IndexScale);
|
||||
} else {
|
||||
return _LoadMem(Class, Size, A.Base, A.Index, Align, A.IndexType, A.IndexScale);
|
||||
}
|
||||
}
|
||||
|
||||
Ref _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, AddressMode A, Ref Value, uint8_t Align = 1, bool ForceNonTSO = false) {
|
||||
bool AtomicTSO = CTX->IsAtomicTSOEnabled() && !ForceNonTSO;
|
||||
A = SelectAddressMode(A, AtomicTSO, Class != GPRClass, Size);
|
||||
|
||||
if (AtomicTSO) {
|
||||
return _StoreMemTSO(Class, Size, Value, A.Base, A.Index, Align, A.IndexType, A.IndexScale);
|
||||
} else {
|
||||
return _StoreMem(Class, Size, Value, A.Base, A.Index, Align, A.IndexType, A.IndexScale);
|
||||
}
|
||||
}
|
||||
|
||||
Ref Prefetch(bool ForStore, bool Stream, uint8_t CacheLevel, Ref ssa0) {
|
||||
return _Prefetch(ForStore, Stream, CacheLevel, ssa0, Invalid(), MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
@@ -2096,8 +2128,8 @@ private:
|
||||
|
||||
///< Segment telemetry tracking
|
||||
uint32_t SegmentsNeedReadCheck {~0U};
|
||||
void CheckLegacySegmentWrite(OrderedNode* NewNode, uint32_t SegmentReg);
|
||||
void CheckLegacySegmentRead(OrderedNode* NewNode, uint32_t SegmentReg);
|
||||
void CheckLegacySegmentWrite(Ref NewNode, uint32_t SegmentReg);
|
||||
void CheckLegacySegmentRead(Ref NewNode, uint32_t SegmentReg);
|
||||
};
|
||||
|
||||
void InstallOpcodeHandlers(Context::OperatingMode Mode);
|
||||
|
||||
@@ -22,10 +22,10 @@ class OrderedNode;
|
||||
#define OpcodeArgs [[maybe_unused]] FEXCore::X86Tables::DecodedOp Op
|
||||
|
||||
void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
|
||||
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
OrderedNode* RotatedNode {};
|
||||
Ref RotatedNode {};
|
||||
if (CTX->HostFeatures.SupportsSHA) {
|
||||
// ARMv8 SHA1 extension provides a `SHA1H` instruction which does a fixed rotate by 30.
|
||||
// This only operates on element 0 rather than element 3. We don't have the luxury of rewriting the x86 SHA algorithm to take advantage of this.
|
||||
@@ -47,20 +47,20 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
|
||||
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
OrderedNode* NewVec = _VExtr(16, 8, Dest, Src, 1);
|
||||
Ref NewVec = _VExtr(16, 8, Dest, Src, 1);
|
||||
|
||||
// [W0, W1, W2, W3] ^ [W2, W3, W4, W5]
|
||||
OrderedNode* Result = _VXor(16, 1, Dest, NewVec);
|
||||
Ref Result = _VXor(16, 1, Dest, NewVec);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
// This instruction mostly matches ARMv8's SHA1SU1 instruction but one of the elements are flipped in an unexpected way.
|
||||
// Do all the work without it.
|
||||
@@ -92,18 +92,18 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src1 needs to be literal here to indicate function and constants");
|
||||
|
||||
using FnType = OrderedNode* (*)(OpDispatchBuilder&, OrderedNode*, OrderedNode*, OrderedNode*);
|
||||
using FnType = Ref (*)(OpDispatchBuilder&, Ref, Ref, Ref);
|
||||
|
||||
const auto f0 = [](OpDispatchBuilder& Self, OrderedNode* B, OrderedNode* C, OrderedNode* D) -> OrderedNode* {
|
||||
const auto f0 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref {
|
||||
return Self._Xor(OpSize::i32Bit, Self._And(OpSize::i32Bit, B, C), Self._Andn(OpSize::i32Bit, D, B));
|
||||
};
|
||||
const auto f1 = [](OpDispatchBuilder& Self, OrderedNode* B, OrderedNode* C, OrderedNode* D) -> OrderedNode* {
|
||||
const auto f1 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref {
|
||||
return Self._Xor(OpSize::i32Bit, Self._Xor(OpSize::i32Bit, B, C), D);
|
||||
};
|
||||
const auto f2 = [](OpDispatchBuilder& Self, OrderedNode* B, OrderedNode* C, OrderedNode* D) -> OrderedNode* {
|
||||
const auto f2 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref {
|
||||
return Self.BitwiseAtLeastTwo(B, C, D);
|
||||
};
|
||||
const auto f3 = [](OpDispatchBuilder& Self, OrderedNode* B, OrderedNode* C, OrderedNode* D) -> OrderedNode* {
|
||||
const auto f3 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref {
|
||||
return Self._Xor(OpSize::i32Bit, Self._Xor(OpSize::i32Bit, B, C), D);
|
||||
};
|
||||
|
||||
@@ -125,12 +125,12 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
const FnType Fn = fn_array[Imm8];
|
||||
auto K = _Constant(32, k_array[Imm8]);
|
||||
|
||||
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto W0E = _VExtractToGPR(16, 4, Src, 3);
|
||||
|
||||
using RoundResult = std::tuple<OrderedNode*, OrderedNode*, OrderedNode*, OrderedNode*, OrderedNode*>;
|
||||
using RoundResult = std::tuple<Ref, Ref, Ref, Ref, Ref>;
|
||||
|
||||
const auto Round0 = [&]() -> RoundResult {
|
||||
auto A = _VExtractToGPR(16, 4, Dest, 3);
|
||||
@@ -147,8 +147,7 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
|
||||
return {A1, B1, C1, D1, E1};
|
||||
};
|
||||
const auto Round1To3 = [&](OrderedNode* A, OrderedNode* B, OrderedNode* C, OrderedNode* D, OrderedNode* E, OrderedNode* Src,
|
||||
unsigned W_idx) -> RoundResult {
|
||||
const auto Round1To3 = [&](Ref A, Ref B, Ref C, Ref D, Ref E, Ref Src, unsigned W_idx) -> RoundResult {
|
||||
// Kill W and E at the beginning
|
||||
auto W = _VExtractToGPR(16, 4, Src, W_idx);
|
||||
auto Q = _Add(OpSize::i32Bit, W, E);
|
||||
@@ -177,15 +176,15 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
|
||||
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
OrderedNode* Result {};
|
||||
Ref Result {};
|
||||
|
||||
if (CTX->HostFeatures.SupportsSHA) {
|
||||
Result = _VSha256U0(Dest, Src);
|
||||
} else {
|
||||
const auto Sigma0 = [this](OrderedNode* W) -> OrderedNode* {
|
||||
const auto Sigma0 = [this](Ref W) -> Ref {
|
||||
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 7)), _Ror(OpSize::i32Bit, W, _Constant(32, 18))),
|
||||
_Lshr(OpSize::i32Bit, W, _Constant(32, 3)));
|
||||
};
|
||||
@@ -211,13 +210,13 @@ void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
const auto Sigma1 = [this](OrderedNode* W) -> OrderedNode* {
|
||||
const auto Sigma1 = [this](Ref W) -> Ref {
|
||||
return _Xor(OpSize::i32Bit, _Xor(OpSize::i32Bit, _Ror(OpSize::i32Bit, W, _Constant(32, 17)), _Ror(OpSize::i32Bit, W, _Constant(32, 19))),
|
||||
_Lshr(OpSize::i32Bit, W, _Constant(32, 10)));
|
||||
};
|
||||
|
||||
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto W14 = _VExtractToGPR(16, 4, Src, 2);
|
||||
auto W15 = _VExtractToGPR(16, 4, Src, 3);
|
||||
@@ -234,7 +233,7 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, D0, -1);
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::BitwiseAtLeastTwo(OrderedNode* A, OrderedNode* B, OrderedNode* C) {
|
||||
Ref OpDispatchBuilder::BitwiseAtLeastTwo(Ref A, Ref B, Ref C) {
|
||||
// Returns whether at least 2/3 of A/B/C is true.
|
||||
// Expressed as (A & (B | C)) | (B & C)
|
||||
//
|
||||
@@ -245,27 +244,27 @@ OrderedNode* OpDispatchBuilder::BitwiseAtLeastTwo(OrderedNode* A, OrderedNode* B
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
const auto Ch = [this](OrderedNode* E, OrderedNode* F, OrderedNode* G) -> OrderedNode* {
|
||||
const auto Ch = [this](Ref E, Ref F, Ref G) -> Ref {
|
||||
return _Xor(OpSize::i32Bit, _And(OpSize::i32Bit, E, F), _Andn(OpSize::i32Bit, G, E));
|
||||
};
|
||||
const auto Sigma0 = [this](OrderedNode* A) -> OrderedNode* {
|
||||
const auto Sigma0 = [this](Ref A) -> Ref {
|
||||
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, A, _Constant(32, 2)), A, ShiftType::ROR, 13), A,
|
||||
ShiftType::ROR, 22);
|
||||
};
|
||||
const auto Sigma1 = [this](OrderedNode* E) -> OrderedNode* {
|
||||
const auto Sigma1 = [this](Ref E) -> Ref {
|
||||
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, E, _Constant(32, 6)), E, ShiftType::ROR, 11), E,
|
||||
ShiftType::ROR, 25);
|
||||
};
|
||||
|
||||
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
// Hardcoded to XMM0
|
||||
auto XMM0 = LoadXMMRegister(0);
|
||||
|
||||
auto E0 = _VExtractToGPR(16, 4, Src, 1);
|
||||
auto F0 = _VExtractToGPR(16, 4, Src, 0);
|
||||
auto G0 = _VExtractToGPR(16, 4, Dest, 1);
|
||||
OrderedNode* Q0 = _Add(OpSize::i32Bit, Ch(E0, F0, G0), Sigma1(E0));
|
||||
Ref Q0 = _Add(OpSize::i32Bit, Ch(E0, F0, G0), Sigma1(E0));
|
||||
|
||||
auto WK0 = _VExtractToGPR(16, 4, XMM0, 0);
|
||||
Q0 = _Add(OpSize::i32Bit, Q0, WK0);
|
||||
@@ -281,7 +280,7 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
auto D0 = _VExtractToGPR(16, 4, Dest, 2);
|
||||
auto E1 = _Add(OpSize::i32Bit, Q0, D0);
|
||||
|
||||
OrderedNode* Q1 = _Add(OpSize::i32Bit, Ch(E1, E0, F0), Sigma1(E1));
|
||||
Ref Q1 = _Add(OpSize::i32Bit, Ch(E1, E0, F0), Sigma1(E1));
|
||||
|
||||
auto WK1 = _VExtractToGPR(16, 4, XMM0, 1);
|
||||
Q1 = _Add(OpSize::i32Bit, Q1, WK1);
|
||||
@@ -305,16 +304,16 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
|
||||
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode* Result = _VAESImc(Src);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESImc(Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
|
||||
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode* Result = _VAESEnc(16, Dest, Src, ZeroRegister);
|
||||
Ref Result = _VAESEnc(16, Dest, Src, ZeroRegister);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
@@ -325,19 +324,19 @@ void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESENC.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENC unimplemented");
|
||||
|
||||
OrderedNode* State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode* Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode* Result = _VAESEnc(DstSize, State, Key, ZeroRegister);
|
||||
Ref Result = _VAESEnc(DstSize, State, Key, ZeroRegister);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
|
||||
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode* Result = _VAESEncLast(16, Dest, Src, ZeroRegister);
|
||||
Ref Result = _VAESEncLast(16, Dest, Src, ZeroRegister);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
@@ -348,19 +347,19 @@ void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESENCLAST.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENCLAST unimplemented");
|
||||
|
||||
OrderedNode* State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode* Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode* Result = _VAESEncLast(DstSize, State, Key, ZeroRegister);
|
||||
Ref Result = _VAESEncLast(DstSize, State, Key, ZeroRegister);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
|
||||
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode* Result = _VAESDec(16, Dest, Src, ZeroRegister);
|
||||
Ref Result = _VAESDec(16, Dest, Src, ZeroRegister);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
@@ -371,19 +370,19 @@ void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESDEC.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDEC unimplemented");
|
||||
|
||||
OrderedNode* State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode* Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode* Result = _VAESDec(DstSize, State, Key, ZeroRegister);
|
||||
Ref Result = _VAESDec(DstSize, State, Key, ZeroRegister);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
|
||||
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(16, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode* Result = _VAESDecLast(16, Dest, Src, ZeroRegister);
|
||||
Ref Result = _VAESDecLast(16, Dest, Src, ZeroRegister);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
@@ -394,16 +393,16 @@ void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESDECLAST.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDECLAST unimplemented");
|
||||
|
||||
OrderedNode* State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode* Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto ZeroRegister = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
OrderedNode* Result = _VAESDecLast(DstSize, State, Key, ZeroRegister);
|
||||
Ref Result = _VAESDecLast(DstSize, State, Key, ZeroRegister);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) {
|
||||
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) {
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src1 needs to be literal here");
|
||||
const uint64_t RCON = Op->Src[1].Data.Literal.Value;
|
||||
|
||||
@@ -413,15 +412,15 @@ OrderedNode* OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) {
|
||||
OrderedNode* Result = AESKeyGenAssistImpl(Op);
|
||||
Ref Result = AESKeyGenAssistImpl(Op);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Selector needs to be literal here");
|
||||
|
||||
OrderedNode* Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
OrderedNode* Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[1].Data.Literal.Value);
|
||||
|
||||
auto Res = _PCLMUL(16, Dest, Src, Selector);
|
||||
@@ -433,11 +432,11 @@ void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) {
|
||||
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
|
||||
OrderedNode* Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
OrderedNode* Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[2].Data.Literal.Value);
|
||||
|
||||
OrderedNode* Res = _PCLMUL(DstSize, Src1, Src2, Selector);
|
||||
Ref Res = _PCLMUL(DstSize, Src1, Src2, Selector);
|
||||
StoreResult(FPRClass, Op, Res, -1);
|
||||
}
|
||||
|
||||
|
||||
@@ -33,7 +33,7 @@ void OpDispatchBuilder::ZeroPF_AF() {
|
||||
SetAF(0);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, OrderedNode* Src) {
|
||||
void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, Ref Src) {
|
||||
size_t NumFlags = FlagOffsets.size();
|
||||
if (Lower8) {
|
||||
// Calculate flags early.
|
||||
@@ -61,7 +61,7 @@ void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, OrderedNode* Src) {
|
||||
SetRFLAG(Src, FEXCore::X86State::RFLAG_AF_RAW_LOC);
|
||||
} else if (FlagOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) {
|
||||
// PF is stored parity flipped
|
||||
OrderedNode* Tmp = _Bfe(OpSize::i32Bit, 1, FlagOffset, Src);
|
||||
Ref Tmp = _Bfe(OpSize::i32Bit, 1, FlagOffset, Src);
|
||||
Tmp = _Xor(OpSize::i32Bit, Tmp, _Constant(1));
|
||||
SetRFLAG(Tmp, FlagOffset);
|
||||
} else {
|
||||
@@ -70,11 +70,11 @@ void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, OrderedNode* Src) {
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
|
||||
Ref OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
|
||||
// Calculate flags early.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
OrderedNode* Original = _Constant(0);
|
||||
Ref Original = _Constant(0);
|
||||
|
||||
// SF/ZF and N/Z are together on both arm64 and x86_64, so we special case that.
|
||||
bool GetNZ = (FlagsMask & (1 << FEXCore::X86State::RFLAG_SF_RAW_LOC)) && (FlagsMask & (1 << FEXCore::X86State::RFLAG_ZF_RAW_LOC));
|
||||
@@ -99,7 +99,7 @@ OrderedNode* OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
|
||||
|
||||
// Note that the Bfi only considers the bottom bit of the flag, the rest of
|
||||
// the byte is allowed to be garbage.
|
||||
OrderedNode* Flag;
|
||||
Ref Flag;
|
||||
if (FlagOffset == FEXCore::X86State::RFLAG_AF_RAW_LOC) {
|
||||
Flag = LoadAF();
|
||||
} else {
|
||||
@@ -139,10 +139,10 @@ OrderedNode* OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
|
||||
return Original;
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateOF(uint8_t SrcSize, OrderedNode* Res, OrderedNode* Src1, OrderedNode* Src2, bool Sub) {
|
||||
void OpDispatchBuilder::CalculateOF(uint8_t SrcSize, Ref Res, Ref Src1, Ref Src2, bool Sub) {
|
||||
auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
uint64_t SignBit = (SrcSize * 8) - 1;
|
||||
OrderedNode* Anded = nullptr;
|
||||
Ref Anded = nullptr;
|
||||
|
||||
// For add, OF is set iff the sources have the same sign but the destination
|
||||
// sign differs. If we know a source sign, we can simplify the expression: if
|
||||
@@ -175,7 +175,7 @@ void OpDispatchBuilder::CalculateOF(uint8_t SrcSize, OrderedNode* Res, OrderedNo
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Anded, SrcSize * 8 - 1, true);
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::LoadPFRaw(bool Invert) {
|
||||
Ref OpDispatchBuilder::LoadPFRaw(bool Invert) {
|
||||
// Read the stored byte. This is the original result (up to 64-bits), it needs
|
||||
// parity calculated.
|
||||
auto Result = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
|
||||
@@ -193,7 +193,7 @@ OrderedNode* OpDispatchBuilder::LoadPFRaw(bool Invert) {
|
||||
return Result;
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::LoadAF() {
|
||||
Ref OpDispatchBuilder::LoadAF() {
|
||||
// Read the stored value. This is the XOR of the arguments.
|
||||
auto AFWord = GetRFLAG(FEXCore::X86State::RFLAG_AF_RAW_LOC);
|
||||
|
||||
@@ -214,11 +214,11 @@ void OpDispatchBuilder::FixupAF() {
|
||||
auto PFRaw = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
|
||||
auto AFRaw = GetRFLAG(FEXCore::X86State::RFLAG_AF_RAW_LOC);
|
||||
|
||||
OrderedNode* XorRes = _Xor(OpSize::i32Bit, AFRaw, PFRaw);
|
||||
Ref XorRes = _Xor(OpSize::i32Bit, AFRaw, PFRaw);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SetAFAndFixup(OrderedNode* AF) {
|
||||
void OpDispatchBuilder::SetAFAndFixup(Ref AF) {
|
||||
// We have a value of AF, we shift into AF[4]. We need to fixup AF[4] so that
|
||||
// we get the right value when we XOR in PF[4] later. The easiest solution is
|
||||
// to XOR by PF[4], since:
|
||||
@@ -227,16 +227,16 @@ void OpDispatchBuilder::SetAFAndFixup(OrderedNode* AF) {
|
||||
|
||||
auto PFRaw = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
|
||||
|
||||
OrderedNode* XorRes = _XorShift(OpSize::i32Bit, PFRaw, AF, ShiftType::LSL, 4);
|
||||
Ref XorRes = _XorShift(OpSize::i32Bit, PFRaw, AF, ShiftType::LSL, 4);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculatePF(OrderedNode* Res) {
|
||||
void OpDispatchBuilder::CalculatePF(Ref Res) {
|
||||
// Calculation is entirely deferred until load, just store the 8-bit result.
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(Res);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateAF(OrderedNode* Src1, OrderedNode* Src2) {
|
||||
void OpDispatchBuilder::CalculateAF(Ref Src1, Ref Src2) {
|
||||
// We only care about bit 4 in the subsequent XOR. If we'll XOR with 0,
|
||||
// there's no sense XOR'ing at all. If we'll XOR with 1, that's just
|
||||
// inverting.
|
||||
@@ -254,7 +254,7 @@ void OpDispatchBuilder::CalculateAF(OrderedNode* Src1, OrderedNode* Src2) {
|
||||
// We store the XOR of the arguments. At read time, we XOR with the
|
||||
// appropriate bit of the result (available as the PF flag) and extract the
|
||||
// appropriate bit.
|
||||
OrderedNode* XorRes = _Xor(OpSize::i32Bit, Src1, Src2);
|
||||
Ref XorRes = _Xor(OpSize::i32Bit, Src1, Src2);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
|
||||
}
|
||||
|
||||
@@ -328,11 +328,11 @@ void OpDispatchBuilder::CalculateDeferredFlags(uint32_t FlagsToCalculateMask) {
|
||||
NZCVDirty = false;
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::CalculateFlags_ADC(uint8_t SrcSize, OrderedNode* Src1, OrderedNode* Src2) {
|
||||
Ref OpDispatchBuilder::CalculateFlags_ADC(uint8_t SrcSize, Ref Src1, Ref Src2) {
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
OrderedNode* Res;
|
||||
Ref Res;
|
||||
|
||||
CalculateAF(Src1, Src2);
|
||||
|
||||
@@ -345,7 +345,7 @@ OrderedNode* OpDispatchBuilder::CalculateFlags_ADC(uint8_t SrcSize, OrderedNode*
|
||||
|
||||
// Note that we do not extend Src2PlusCF, since we depend on proper
|
||||
// 32-bit arithmetic to correctly handle the Src2 = 0xffff case.
|
||||
OrderedNode* Src2PlusCF = _Adc(OpSize, _Constant(0), Src2);
|
||||
Ref Src2PlusCF = _Adc(OpSize, _Constant(0), Src2);
|
||||
|
||||
// Need to zero-extend for the comparison.
|
||||
Res = _Add(OpSize, Src1, Src2PlusCF);
|
||||
@@ -363,14 +363,14 @@ OrderedNode* OpDispatchBuilder::CalculateFlags_ADC(uint8_t SrcSize, OrderedNode*
|
||||
return Res;
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::CalculateFlags_SBB(uint8_t SrcSize, OrderedNode* Src1, OrderedNode* Src2) {
|
||||
Ref OpDispatchBuilder::CalculateFlags_SBB(uint8_t SrcSize, Ref Src1, Ref Src2) {
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
auto OpSize = SrcSize == 8 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
|
||||
CalculateAF(Src1, Src2);
|
||||
|
||||
OrderedNode* Res;
|
||||
Ref Res;
|
||||
if (SrcSize >= 4) {
|
||||
// Rectify input carry
|
||||
CarryInvert();
|
||||
@@ -402,7 +402,7 @@ OrderedNode* OpDispatchBuilder::CalculateFlags_SBB(uint8_t SrcSize, OrderedNode*
|
||||
return Res;
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, OrderedNode* Src1, OrderedNode* Src2, bool UpdateCF) {
|
||||
Ref OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, Ref Src1, Ref Src2, bool UpdateCF) {
|
||||
// Stash CF before stomping over it
|
||||
auto OldCF = UpdateCF ? nullptr : GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
|
||||
@@ -410,7 +410,7 @@ OrderedNode* OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, OrderedNode*
|
||||
|
||||
CalculateAF(Src1, Src2);
|
||||
|
||||
OrderedNode* Res;
|
||||
Ref Res;
|
||||
if (SrcSize >= 4) {
|
||||
Res = _SubWithFlags(IR::SizeToOpSize(SrcSize), Src1, Src2);
|
||||
} else {
|
||||
@@ -431,7 +431,7 @@ OrderedNode* OpDispatchBuilder::CalculateFlags_SUB(uint8_t SrcSize, OrderedNode*
|
||||
return Res;
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, OrderedNode* Src1, OrderedNode* Src2, bool UpdateCF) {
|
||||
Ref OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, Ref Src1, Ref Src2, bool UpdateCF) {
|
||||
// Stash CF before stomping over it
|
||||
auto OldCF = UpdateCF ? nullptr : GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
|
||||
@@ -439,7 +439,7 @@ OrderedNode* OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, OrderedNode*
|
||||
|
||||
CalculateAF(Src1, Src2);
|
||||
|
||||
OrderedNode* Res;
|
||||
Ref Res;
|
||||
if (SrcSize >= 4) {
|
||||
Res = _AddWithFlags(IR::SizeToOpSize(SrcSize), Src1, Src2);
|
||||
} else {
|
||||
@@ -457,7 +457,7 @@ OrderedNode* OpDispatchBuilder::CalculateFlags_ADD(uint8_t SrcSize, OrderedNode*
|
||||
return Res;
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_MUL(uint8_t SrcSize, OrderedNode* Res, OrderedNode* High) {
|
||||
void OpDispatchBuilder::CalculateFlags_MUL(uint8_t SrcSize, Ref Res, Ref High) {
|
||||
HandleNZCVWrite();
|
||||
|
||||
// PF/AF/ZF/SF
|
||||
@@ -481,7 +481,7 @@ void OpDispatchBuilder::CalculateFlags_MUL(uint8_t SrcSize, OrderedNode* Res, Or
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_UMUL(OrderedNode* High) {
|
||||
void OpDispatchBuilder::CalculateFlags_UMUL(Ref High) {
|
||||
HandleNZCVWrite();
|
||||
|
||||
auto Zero = _Constant(0);
|
||||
@@ -506,18 +506,21 @@ void OpDispatchBuilder::CalculateFlags_UMUL(OrderedNode* High) {
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_Logical(uint8_t SrcSize, OrderedNode* Res, OrderedNode* Src1, OrderedNode* Src2) {
|
||||
void OpDispatchBuilder::CalculateFlags_Logical(uint8_t SrcSize, Ref Res, Ref Src1, Ref Src2) {
|
||||
// AF
|
||||
// Undefined
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
|
||||
CalculatePF(Res);
|
||||
|
||||
// SF/ZF/CF/OF
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
if (SrcSize >= 4) {
|
||||
HandleNZ00Write();
|
||||
CalculatePF(_AndWithFlags(IR::SizeToOpSize(SrcSize), Res, Res));
|
||||
} else {
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
CalculatePF(Res);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, OrderedNode* UnmaskedRes, OrderedNode* Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, Ref UnmaskedRes, Ref Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero
|
||||
if (Shift == 0) {
|
||||
return;
|
||||
@@ -553,7 +556,7 @@ void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, Order
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize, OrderedNode* Res, OrderedNode* Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero
|
||||
if (Shift == 0) {
|
||||
return;
|
||||
@@ -579,7 +582,7 @@ void OpDispatchBuilder::CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize,
|
||||
// already zeroed there's nothing to do here.
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize, OrderedNode* Res, OrderedNode* Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift) {
|
||||
// Set SF and PF. Clobbers OF, but OF only defined for Shift = 1 where it is
|
||||
// set below.
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
@@ -597,7 +600,7 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize
|
||||
_InvalidateFlags(1 << X86State::RFLAG_AF_RAW_LOC);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftRightImmediate(uint8_t SrcSize, OrderedNode* Res, OrderedNode* Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftRightImmediate(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero
|
||||
if (Shift == 0) {
|
||||
return;
|
||||
@@ -615,7 +618,7 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightImmediate(uint8_t SrcSize, Orde
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftRightDoubleImmediate(uint8_t SrcSize, OrderedNode* Res, OrderedNode* Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftRightDoubleImmediate(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero
|
||||
if (Shift == 0) {
|
||||
return;
|
||||
@@ -636,7 +639,7 @@ void OpDispatchBuilder::CalculateFlags_ShiftRightDoubleImmediate(uint8_t SrcSize
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BEXTR(OrderedNode* Src) {
|
||||
void OpDispatchBuilder::CalculateFlags_BEXTR(Ref Src) {
|
||||
// ZF is set properly. CF and OF are defined as being set to zero. SF, PF, and
|
||||
// AF are undefined.
|
||||
SetNZ_ZeroCV(GetOpSize(Src), Src);
|
||||
@@ -644,7 +647,7 @@ void OpDispatchBuilder::CalculateFlags_BEXTR(OrderedNode* Src) {
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) | (1UL << X86State::RFLAG_AF_RAW_LOC));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BLSI(uint8_t SrcSize, OrderedNode* Result) {
|
||||
void OpDispatchBuilder::CalculateFlags_BLSI(uint8_t SrcSize, Ref Result) {
|
||||
// CF is cleared if Src is zero, otherwise it's set. However, Src is zero iff
|
||||
// Result is zero, so we can test the result instead. So, CF is just the
|
||||
// inverted ZF.
|
||||
@@ -659,7 +662,7 @@ void OpDispatchBuilder::CalculateFlags_BLSI(uint8_t SrcSize, OrderedNode* Result
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) | (1UL << X86State::RFLAG_AF_RAW_LOC));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BLSMSK(uint8_t SrcSize, OrderedNode* Result, OrderedNode* Src) {
|
||||
void OpDispatchBuilder::CalculateFlags_BLSMSK(uint8_t SrcSize, Ref Result, Ref Src) {
|
||||
// PF/AF undefined
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) | (1UL << X86State::RFLAG_AF_RAW_LOC));
|
||||
|
||||
@@ -674,7 +677,7 @@ void OpDispatchBuilder::CalculateFlags_BLSMSK(uint8_t SrcSize, OrderedNode* Resu
|
||||
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(CFOp);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BLSR(uint8_t SrcSize, OrderedNode* Result, OrderedNode* Src) {
|
||||
void OpDispatchBuilder::CalculateFlags_BLSR(uint8_t SrcSize, Ref Result, Ref Src) {
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
auto CFOp = _Select(IR::COND_EQ, Src, Zero, One, Zero);
|
||||
@@ -686,7 +689,7 @@ void OpDispatchBuilder::CalculateFlags_BLSR(uint8_t SrcSize, OrderedNode* Result
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) | (1UL << X86State::RFLAG_AF_RAW_LOC));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_POPCOUNT(OrderedNode* Result) {
|
||||
void OpDispatchBuilder::CalculateFlags_POPCOUNT(Ref Result) {
|
||||
// We need to set ZF while clearing the rest of NZCV. The result of a popcount
|
||||
// is in the range [0, 63]. In particular, it is always positive. So a
|
||||
// combined NZ test will correctly zero SF/CF/OF while setting ZF.
|
||||
@@ -694,7 +697,7 @@ void OpDispatchBuilder::CalculateFlags_POPCOUNT(OrderedNode* Result) {
|
||||
ZeroPF_AF();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_BZHI(uint8_t SrcSize, OrderedNode* Result, OrderedNode* Src) {
|
||||
void OpDispatchBuilder::CalculateFlags_BZHI(uint8_t SrcSize, Ref Result, Ref Src) {
|
||||
// PF/AF undefined
|
||||
_InvalidateFlags((1UL << X86State::RFLAG_PF_RAW_LOC) | (1UL << X86State::RFLAG_AF_RAW_LOC));
|
||||
|
||||
@@ -702,7 +705,7 @@ void OpDispatchBuilder::CalculateFlags_BZHI(uint8_t SrcSize, OrderedNode* Result
|
||||
SetRFLAG<X86State::RFLAG_CF_RAW_LOC>(Src);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ZCNT(uint8_t SrcSize, OrderedNode* Result) {
|
||||
void OpDispatchBuilder::CalculateFlags_ZCNT(uint8_t SrcSize, Ref Result) {
|
||||
// OF, SF, AF, PF all undefined
|
||||
// Test ZF of result, SF is undefined so this is ok.
|
||||
SetNZ_ZeroCV(SrcSize, Result);
|
||||
@@ -714,7 +717,7 @@ void OpDispatchBuilder::CalculateFlags_ZCNT(uint8_t SrcSize, OrderedNode* Result
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_RAW_LOC>(Result, CarryBit);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_RDRAND(OrderedNode* Src) {
|
||||
void OpDispatchBuilder::CalculateFlags_RDRAND(Ref Src) {
|
||||
// OF, SF, ZF, AF, PF all zero
|
||||
ZeroNZCV();
|
||||
ZeroPF_AF();
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -23,45 +23,45 @@ class OrderedNode;
|
||||
|
||||
#define OpcodeArgs [[maybe_unused]] FEXCore::X86Tables::DecodedOp Op
|
||||
|
||||
OrderedNode* OpDispatchBuilder::GetX87Top() {
|
||||
Ref OpDispatchBuilder::GetX87Top() {
|
||||
// Yes, we are storing 3 bits in a single flag register.
|
||||
// Deal with it
|
||||
return _LoadContext(1, GPRClass, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SetX87ValidTag(OrderedNode* Value, bool Valid) {
|
||||
void OpDispatchBuilder::SetX87ValidTag(Ref Value, bool Valid) {
|
||||
// if we are popping then we must first mark this location as empty
|
||||
OrderedNode* AbridgedFTW = _LoadContext(1, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
OrderedNode* RegMask = _Lshl(OpSize::i32Bit, _Constant(1), Value);
|
||||
OrderedNode* NewAbridgedFTW = Valid ? _Or(OpSize::i32Bit, AbridgedFTW, RegMask) : _Andn(OpSize::i32Bit, AbridgedFTW, RegMask);
|
||||
Ref AbridgedFTW = _LoadContext(1, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
Ref RegMask = _Lshl(OpSize::i32Bit, _Constant(1), Value);
|
||||
Ref NewAbridgedFTW = Valid ? _Or(OpSize::i32Bit, AbridgedFTW, RegMask) : _Andn(OpSize::i32Bit, AbridgedFTW, RegMask);
|
||||
_StoreContext(1, GPRClass, NewAbridgedFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::GetX87ValidTag(OrderedNode* Value) {
|
||||
OrderedNode* AbridgedFTW = _LoadContext(1, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
Ref OpDispatchBuilder::GetX87ValidTag(Ref Value) {
|
||||
Ref AbridgedFTW = _LoadContext(1, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
return _And(OpSize::i32Bit, _Lshr(OpSize::i32Bit, AbridgedFTW, Value), _Constant(1));
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::GetX87Tag(OrderedNode* Value, OrderedNode* AbridgedFTW) {
|
||||
OrderedNode* RegValid = _And(OpSize::i32Bit, _Lshr(OpSize::i32Bit, AbridgedFTW, Value), _Constant(1));
|
||||
OrderedNode* X87Empty = _Constant(static_cast<uint8_t>(FPState::X87Tag::Empty));
|
||||
OrderedNode* X87Valid = _Constant(static_cast<uint8_t>(FPState::X87Tag::Valid));
|
||||
Ref OpDispatchBuilder::GetX87Tag(Ref Value, Ref AbridgedFTW) {
|
||||
Ref RegValid = _And(OpSize::i32Bit, _Lshr(OpSize::i32Bit, AbridgedFTW, Value), _Constant(1));
|
||||
Ref X87Empty = _Constant(static_cast<uint8_t>(FPState::X87Tag::Empty));
|
||||
Ref X87Valid = _Constant(static_cast<uint8_t>(FPState::X87Tag::Valid));
|
||||
|
||||
return _Select(FEXCore::IR::COND_EQ, RegValid, _Constant(0), X87Empty, X87Valid);
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::GetX87Tag(OrderedNode* Value) {
|
||||
OrderedNode* AbridgedFTW = _LoadContext(1, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
Ref OpDispatchBuilder::GetX87Tag(Ref Value) {
|
||||
Ref AbridgedFTW = _LoadContext(1, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
return GetX87Tag(Value, AbridgedFTW);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SetX87FTW(OrderedNode* FTW) {
|
||||
OrderedNode* X87Empty = _Constant(static_cast<uint8_t>(FPState::X87Tag::Empty));
|
||||
OrderedNode* NewAbridgedFTW;
|
||||
void OpDispatchBuilder::SetX87FTW(Ref FTW) {
|
||||
Ref X87Empty = _Constant(static_cast<uint8_t>(FPState::X87Tag::Empty));
|
||||
Ref NewAbridgedFTW;
|
||||
|
||||
for (int i = 0; i < 8; i++) {
|
||||
OrderedNode* RegTag = _Bfe(OpSize::i32Bit, 2, i * 2, FTW);
|
||||
OrderedNode* RegValid = _Select(FEXCore::IR::COND_NEQ, RegTag, X87Empty, _Constant(1), _Constant(0));
|
||||
Ref RegTag = _Bfe(OpSize::i32Bit, 2, i * 2, FTW);
|
||||
Ref RegValid = _Select(FEXCore::IR::COND_NEQ, RegTag, X87Empty, _Constant(1), _Constant(0));
|
||||
|
||||
if (i) {
|
||||
NewAbridgedFTW = _Orlshl(OpSize::i32Bit, NewAbridgedFTW, RegValid, i);
|
||||
@@ -73,9 +73,9 @@ void OpDispatchBuilder::SetX87FTW(OrderedNode* FTW) {
|
||||
_StoreContext(1, GPRClass, NewAbridgedFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::GetX87FTW() {
|
||||
OrderedNode* AbridgedFTW = _LoadContext(1, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
OrderedNode* FTW = _Constant(0);
|
||||
Ref OpDispatchBuilder::GetX87FTW() {
|
||||
Ref AbridgedFTW = _LoadContext(1, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
Ref FTW = _Constant(0);
|
||||
|
||||
for (int i = 0; i < 8; i++) {
|
||||
const auto RegTag = GetX87Tag(_Constant(i), AbridgedFTW);
|
||||
@@ -85,13 +85,13 @@ OrderedNode* OpDispatchBuilder::GetX87FTW() {
|
||||
return FTW;
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SetX87Top(OrderedNode* Value) {
|
||||
void OpDispatchBuilder::SetX87Top(Ref Value) {
|
||||
_StoreContext(1, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::ReconstructFSW() {
|
||||
Ref OpDispatchBuilder::ReconstructFSW() {
|
||||
// We must construct the FSW from our various bits
|
||||
OrderedNode* FSW = _Constant(0);
|
||||
Ref FSW = _Constant(0);
|
||||
auto Top = GetX87Top();
|
||||
FSW = _Bfi(OpSize::i64Bit, 3, 11, FSW, Top);
|
||||
|
||||
@@ -107,7 +107,7 @@ OrderedNode* OpDispatchBuilder::ReconstructFSW() {
|
||||
return FSW;
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::ReconstructX87StateFromFSW(OrderedNode* FSW) {
|
||||
Ref OpDispatchBuilder::ReconstructX87StateFromFSW(Ref FSW) {
|
||||
auto Top = _Bfe(OpSize::i32Bit, 3, 11, FSW);
|
||||
SetX87Top(Top);
|
||||
|
||||
@@ -131,7 +131,7 @@ void OpDispatchBuilder::FLD(OpcodeArgs) {
|
||||
|
||||
size_t read_width = (width == 80) ? 16 : width / 8;
|
||||
|
||||
OrderedNode* data {};
|
||||
Ref data {};
|
||||
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Read from memory
|
||||
@@ -142,7 +142,7 @@ void OpDispatchBuilder::FLD(OpcodeArgs) {
|
||||
data = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, orig_top, offset), mask);
|
||||
data = _LoadContextIndexed(data, 16, MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
OrderedNode* converted = data;
|
||||
Ref converted = data;
|
||||
|
||||
// Convert to 80bit float
|
||||
if constexpr (width == 32 || width == 64) {
|
||||
@@ -170,8 +170,8 @@ void OpDispatchBuilder::FBLD(OpcodeArgs) {
|
||||
SetX87Top(top);
|
||||
|
||||
// Read from memory
|
||||
OrderedNode* data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 16, Op->Flags);
|
||||
OrderedNode* converted = _F80BCDLoad(data);
|
||||
Ref data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 16, Op->Flags);
|
||||
Ref converted = _F80BCDLoad(data);
|
||||
_StoreContextIndexed(converted, top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
|
||||
@@ -179,7 +179,7 @@ void OpDispatchBuilder::FBSTP(OpcodeArgs) {
|
||||
auto orig_top = GetX87Top();
|
||||
auto data = _LoadContextIndexed(orig_top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
OrderedNode* converted = _F80BCDStore(data);
|
||||
Ref converted = _F80BCDStore(data);
|
||||
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, converted, 10, 1);
|
||||
|
||||
@@ -197,7 +197,7 @@ void OpDispatchBuilder::FLD_Const(OpcodeArgs) {
|
||||
SetX87ValidTag(top, true);
|
||||
SetX87Top(top);
|
||||
|
||||
OrderedNode* data = LoadAndCacheNamedVectorConstant(16, constant);
|
||||
Ref data = LoadAndCacheNamedVectorConstant(16, constant);
|
||||
|
||||
// Write to ST[TOP]
|
||||
_StoreContextIndexed(data, top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
@@ -247,7 +247,7 @@ void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
auto upper = _Or(OpSize::i64Bit, sign, zeroed_exponent);
|
||||
|
||||
|
||||
OrderedNode* converted = _VCastFromGPR(16, 8, shifted);
|
||||
Ref converted = _VCastFromGPR(16, 8, shifted);
|
||||
converted = _VInsElement(16, 8, 1, 0, converted, _VCastFromGPR(16, 8, upper));
|
||||
|
||||
// Write to ST[TOP]
|
||||
@@ -283,7 +283,7 @@ void OpDispatchBuilder::FIST(OpcodeArgs) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
|
||||
auto orig_top = GetX87Top();
|
||||
OrderedNode* data = _LoadContextIndexed(orig_top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
Ref data = _LoadContextIndexed(orig_top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
data = _F80CVTInt(Size, data, Truncate);
|
||||
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, data, Size, 1);
|
||||
@@ -303,10 +303,10 @@ template void OpDispatchBuilder::FIST<true>(OpcodeArgs);
|
||||
template<size_t width, bool Integer, OpDispatchBuilder::OpResult ResInST0>
|
||||
void OpDispatchBuilder::FADD(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
OrderedNode* StackLocation = top;
|
||||
Ref StackLocation = top;
|
||||
|
||||
OrderedNode* arg {};
|
||||
OrderedNode* b {};
|
||||
Ref arg {};
|
||||
Ref b {};
|
||||
|
||||
auto mask = _Constant(7);
|
||||
|
||||
@@ -357,9 +357,9 @@ template void OpDispatchBuilder::FADD<32, true, OpDispatchBuilder::OpResult::RES
|
||||
template<size_t width, bool Integer, OpDispatchBuilder::OpResult ResInST0>
|
||||
void OpDispatchBuilder::FMUL(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
OrderedNode* StackLocation = top;
|
||||
OrderedNode* arg {};
|
||||
OrderedNode* b {};
|
||||
Ref StackLocation = top;
|
||||
Ref arg {};
|
||||
Ref b {};
|
||||
|
||||
auto mask = _Constant(7);
|
||||
|
||||
@@ -413,9 +413,9 @@ template void OpDispatchBuilder::FMUL<32, true, OpDispatchBuilder::OpResult::RES
|
||||
template<size_t width, bool Integer, bool reverse, OpDispatchBuilder::OpResult ResInST0>
|
||||
void OpDispatchBuilder::FDIV(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
OrderedNode* StackLocation = top;
|
||||
OrderedNode* arg {};
|
||||
OrderedNode* b {};
|
||||
Ref StackLocation = top;
|
||||
Ref arg {};
|
||||
Ref b {};
|
||||
|
||||
auto mask = _Constant(7);
|
||||
|
||||
@@ -444,7 +444,7 @@ void OpDispatchBuilder::FDIV(OpcodeArgs) {
|
||||
|
||||
auto a = _LoadContextIndexed(top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
OrderedNode* result {};
|
||||
Ref result {};
|
||||
if constexpr (reverse) {
|
||||
result = _F80Div(b, a);
|
||||
} else {
|
||||
@@ -484,9 +484,9 @@ template void OpDispatchBuilder::FDIV<32, true, true, OpDispatchBuilder::OpResul
|
||||
template<size_t width, bool Integer, bool reverse, OpDispatchBuilder::OpResult ResInST0>
|
||||
void OpDispatchBuilder::FSUB(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
OrderedNode* StackLocation = top;
|
||||
OrderedNode* arg {};
|
||||
OrderedNode* b {};
|
||||
Ref StackLocation = top;
|
||||
Ref arg {};
|
||||
Ref b {};
|
||||
|
||||
auto mask = _Constant(7);
|
||||
|
||||
@@ -514,7 +514,7 @@ void OpDispatchBuilder::FSUB(OpcodeArgs) {
|
||||
|
||||
auto a = _LoadContextIndexed(top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
OrderedNode* result {};
|
||||
Ref result {};
|
||||
if constexpr (reverse) {
|
||||
result = _F80Sub(b, a);
|
||||
} else {
|
||||
@@ -558,7 +558,7 @@ void OpDispatchBuilder::FCHS(OpcodeArgs) {
|
||||
|
||||
auto low = _Constant(0);
|
||||
auto high = _Constant(0b1'000'0000'0000'0000ULL);
|
||||
OrderedNode* data = _VCastFromGPR(16, 8, low);
|
||||
Ref data = _VCastFromGPR(16, 8, low);
|
||||
data = _VInsGPR(16, 8, 1, data, high);
|
||||
|
||||
auto result = _VXor(16, 1, a, data);
|
||||
@@ -573,7 +573,7 @@ void OpDispatchBuilder::FABS(OpcodeArgs) {
|
||||
|
||||
auto low = _Constant(~0ULL);
|
||||
auto high = _Constant(0b0'111'1111'1111'1111ULL);
|
||||
OrderedNode* data = _VCastFromGPR(16, 8, low);
|
||||
Ref data = _VCastFromGPR(16, 8, low);
|
||||
data = _VInsGPR(16, 8, 1, data, high);
|
||||
|
||||
auto result = _VAnd(16, 1, a, data);
|
||||
@@ -587,13 +587,13 @@ void OpDispatchBuilder::FTST(OpcodeArgs) {
|
||||
auto a = _LoadContextIndexed(top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
auto low = _Constant(0);
|
||||
OrderedNode* data = _VCastFromGPR(16, 8, low);
|
||||
Ref data = _VCastFromGPR(16, 8, low);
|
||||
|
||||
OrderedNode* Res = _F80Cmp(a, data, (1 << FCMP_FLAG_EQ) | (1 << FCMP_FLAG_LT) | (1 << FCMP_FLAG_UNORDERED));
|
||||
Ref Res = _F80Cmp(a, data, (1 << FCMP_FLAG_EQ) | (1 << FCMP_FLAG_LT) | (1 << FCMP_FLAG_UNORDERED));
|
||||
|
||||
OrderedNode* HostFlag_CF = _GetHostFlag(Res, FCMP_FLAG_LT);
|
||||
OrderedNode* HostFlag_ZF = _GetHostFlag(Res, FCMP_FLAG_EQ);
|
||||
OrderedNode* HostFlag_Unordered = _GetHostFlag(Res, FCMP_FLAG_UNORDERED);
|
||||
Ref HostFlag_CF = _GetHostFlag(Res, FCMP_FLAG_LT);
|
||||
Ref HostFlag_ZF = _GetHostFlag(Res, FCMP_FLAG_EQ);
|
||||
Ref HostFlag_Unordered = _GetHostFlag(Res, FCMP_FLAG_UNORDERED);
|
||||
HostFlag_CF = _Or(OpSize::i32Bit, HostFlag_CF, HostFlag_Unordered);
|
||||
HostFlag_ZF = _Or(OpSize::i32Bit, HostFlag_ZF, HostFlag_Unordered);
|
||||
|
||||
@@ -652,8 +652,8 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
auto mask = _Constant(7);
|
||||
|
||||
OrderedNode* arg {};
|
||||
OrderedNode* b {};
|
||||
Ref arg {};
|
||||
Ref b {};
|
||||
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Memory arg
|
||||
@@ -675,11 +675,11 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs) {
|
||||
|
||||
auto a = _LoadContextIndexed(top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
OrderedNode* Res = _F80Cmp(a, b, (1 << FCMP_FLAG_EQ) | (1 << FCMP_FLAG_LT) | (1 << FCMP_FLAG_UNORDERED));
|
||||
Ref Res = _F80Cmp(a, b, (1 << FCMP_FLAG_EQ) | (1 << FCMP_FLAG_LT) | (1 << FCMP_FLAG_UNORDERED));
|
||||
|
||||
OrderedNode* HostFlag_CF = _GetHostFlag(Res, FCMP_FLAG_LT);
|
||||
OrderedNode* HostFlag_ZF = _GetHostFlag(Res, FCMP_FLAG_EQ);
|
||||
OrderedNode* HostFlag_Unordered = _GetHostFlag(Res, FCMP_FLAG_UNORDERED);
|
||||
Ref HostFlag_CF = _GetHostFlag(Res, FCMP_FLAG_LT);
|
||||
Ref HostFlag_ZF = _GetHostFlag(Res, FCMP_FLAG_EQ);
|
||||
Ref HostFlag_Unordered = _GetHostFlag(Res, FCMP_FLAG_UNORDERED);
|
||||
HostFlag_CF = _Or(OpSize::i32Bit, HostFlag_CF, HostFlag_Unordered);
|
||||
HostFlag_ZF = _Or(OpSize::i32Bit, HostFlag_ZF, HostFlag_Unordered);
|
||||
|
||||
@@ -734,7 +734,7 @@ template void OpDispatchBuilder::FCOMI<32, true, OpDispatchBuilder::FCOMIFlags::
|
||||
|
||||
void OpDispatchBuilder::FXCH(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
OrderedNode* arg;
|
||||
Ref arg;
|
||||
|
||||
auto mask = _Constant(7);
|
||||
|
||||
@@ -752,7 +752,7 @@ void OpDispatchBuilder::FXCH(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::FST(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
OrderedNode* arg;
|
||||
Ref arg;
|
||||
|
||||
auto mask = _Constant(7);
|
||||
|
||||
@@ -799,7 +799,7 @@ void OpDispatchBuilder::X87BinaryOp(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
|
||||
auto mask = _Constant(7);
|
||||
OrderedNode* st1 = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, top, _Constant(1)), mask);
|
||||
Ref st1 = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, top, _Constant(1)), mask);
|
||||
|
||||
auto a = _LoadContextIndexed(top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
st1 = _LoadContextIndexed(st1, 16, MMBaseOffset(), 16, FPRClass);
|
||||
@@ -862,11 +862,11 @@ void OpDispatchBuilder::X87FYL2X(OpcodeArgs) {
|
||||
auto top = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, orig_top, _Constant(1)), _Constant(7));
|
||||
SetX87Top(top);
|
||||
|
||||
OrderedNode* st0 = _LoadContextIndexed(orig_top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
OrderedNode* st1 = _LoadContextIndexed(top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
Ref st0 = _LoadContextIndexed(orig_top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
Ref st1 = _LoadContextIndexed(top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
if (Plus1) {
|
||||
OrderedNode* data = LoadAndCacheNamedVectorConstant(16, NamedVectorConstant::NAMED_VECTOR_X87_ONE);
|
||||
Ref data = LoadAndCacheNamedVectorConstant(16, NamedVectorConstant::NAMED_VECTOR_X87_ONE);
|
||||
st0 = _F80Add(st0, data);
|
||||
}
|
||||
|
||||
@@ -886,7 +886,7 @@ void OpDispatchBuilder::X87TAN(OpcodeArgs) {
|
||||
|
||||
auto result = _F80TAN(a);
|
||||
|
||||
OrderedNode* data = LoadAndCacheNamedVectorConstant(16, NamedVectorConstant::NAMED_VECTOR_X87_ONE);
|
||||
Ref data = LoadAndCacheNamedVectorConstant(16, NamedVectorConstant::NAMED_VECTOR_X87_ONE);
|
||||
|
||||
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
@@ -904,7 +904,7 @@ void OpDispatchBuilder::X87ATAN(OpcodeArgs) {
|
||||
SetX87Top(top);
|
||||
|
||||
auto a = _LoadContextIndexed(orig_top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
OrderedNode* st1 = _LoadContextIndexed(top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
Ref st1 = _LoadContextIndexed(top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
auto result = _F80ATAN(st1, a);
|
||||
|
||||
@@ -914,19 +914,17 @@ void OpDispatchBuilder::X87ATAN(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::X87LDENV(OpcodeArgs) {
|
||||
const auto Size = GetSrcSize(Op);
|
||||
OrderedNode* Mem = MakeSegmentAddress(Op, Op->Src[0]);
|
||||
Ref Mem = MakeSegmentAddress(Op, Op->Src[0]);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 1));
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, MemLocation, Size);
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, Mem, _Constant(Size * 1), Size, MEM_OFFSET_SXTX, 1);
|
||||
ReconstructX87StateFromFSW(NewFSW);
|
||||
|
||||
{
|
||||
// FTW
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 2));
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, MemLocation, Size));
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, Mem, _Constant(Size * 2), Size, MEM_OFFSET_SXTX, 1));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -951,53 +949,45 @@ void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) {
|
||||
// 4 bytes : data pointer selector
|
||||
|
||||
const auto Size = GetDstSize(Op);
|
||||
OrderedNode* Mem = MakeSegmentAddress(Op, Op->Dest);
|
||||
Ref Mem = MakeSegmentAddress(Op, Op->Dest);
|
||||
|
||||
{
|
||||
auto FCW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreMem(GPRClass, Size, Mem, FCW, Size);
|
||||
}
|
||||
|
||||
{
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 1));
|
||||
_StoreMem(GPRClass, Size, MemLocation, ReconstructFSW(), Size);
|
||||
}
|
||||
{ _StoreMem(GPRClass, Size, ReconstructFSW(), Mem, _Constant(Size * 1), Size, MEM_OFFSET_SXTX, 1); }
|
||||
|
||||
auto ZeroConst = _Constant(0);
|
||||
|
||||
{
|
||||
// FTW
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 2));
|
||||
_StoreMem(GPRClass, Size, MemLocation, GetX87FTW(), Size);
|
||||
_StoreMem(GPRClass, Size, GetX87FTW(), Mem, _Constant(Size * 2), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction Offset
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 3));
|
||||
_StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size);
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 3), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction CS selector (+ Opcode)
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 4));
|
||||
_StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size);
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 4), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer offset
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 5));
|
||||
_StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size);
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 5), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer selector
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 6));
|
||||
_StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size);
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 6), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FLDCW(OpcodeArgs) {
|
||||
OrderedNode* NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
}
|
||||
|
||||
@@ -1008,7 +998,7 @@ void OpDispatchBuilder::X87FSTCW(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87LDSW(OpcodeArgs) {
|
||||
OrderedNode* NewFSW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref NewFSW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
ReconstructX87StateFromFSW(NewFSW);
|
||||
}
|
||||
|
||||
@@ -1037,59 +1027,47 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
// 4 bytes : data pointer selector
|
||||
|
||||
const auto Size = GetDstSize(Op);
|
||||
OrderedNode* Mem = MakeSegmentAddress(Op, Op->Dest);
|
||||
OrderedNode* Top = GetX87Top();
|
||||
Ref Mem = MakeSegmentAddress(Op, Op->Dest);
|
||||
Ref Top = GetX87Top();
|
||||
{
|
||||
auto FCW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreMem(GPRClass, Size, Mem, FCW, Size);
|
||||
}
|
||||
|
||||
{
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 1));
|
||||
_StoreMem(GPRClass, Size, MemLocation, ReconstructFSW(), Size);
|
||||
}
|
||||
{ _StoreMem(GPRClass, Size, ReconstructFSW(), Mem, _Constant(Size * 1), Size, MEM_OFFSET_SXTX, 1); }
|
||||
|
||||
auto ZeroConst = _Constant(0);
|
||||
|
||||
{
|
||||
// FTW
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 2));
|
||||
_StoreMem(GPRClass, Size, MemLocation, GetX87FTW(), Size);
|
||||
_StoreMem(GPRClass, Size, GetX87FTW(), Mem, _Constant(Size * 2), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction Offset
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 3));
|
||||
_StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size);
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 3), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction CS selector (+ Opcode)
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 4));
|
||||
_StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size);
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 4), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer offset
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 5));
|
||||
_StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size);
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 5), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer selector
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 6));
|
||||
_StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size);
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 6), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
OrderedNode* ST0Location = _Add(OpSize::i64Bit, Mem, _Constant(Size * 7));
|
||||
|
||||
auto OneConst = _Constant(1);
|
||||
auto SevenConst = _Constant(7);
|
||||
auto TenConst = _Constant(10);
|
||||
for (int i = 0; i < 7; ++i) {
|
||||
auto data = _LoadContextIndexed(Top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
_StoreMem(FPRClass, 16, ST0Location, data, 1);
|
||||
ST0Location = _Add(OpSize::i64Bit, ST0Location, TenConst);
|
||||
_StoreMem(FPRClass, 16, data, Mem, _Constant((Size * 7) + (10 * i)), 1, MEM_OFFSET_SXTX, 1);
|
||||
Top = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, Top, OneConst), SevenConst);
|
||||
}
|
||||
|
||||
@@ -1098,10 +1076,9 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
// ST7 broken in to two parts
|
||||
// Lower 64bits [63:0]
|
||||
// upper 16 bits [79:64]
|
||||
_StoreMem(FPRClass, 8, ST0Location, data, 1);
|
||||
ST0Location = _Add(OpSize::i64Bit, ST0Location, _Constant(8));
|
||||
_StoreMem(FPRClass, 8, data, Mem, _Constant((Size * 7) + (7 * 10)), 1, MEM_OFFSET_SXTX, 1);
|
||||
auto topBytes = _VDupElement(16, 2, data, 4);
|
||||
_StoreMem(FPRClass, 2, ST0Location, topBytes, 1);
|
||||
_StoreMem(FPRClass, 2, topBytes, Mem, _Constant((Size * 7) + (7 * 10) + 8), 1, MEM_OFFSET_SXTX, 1);
|
||||
|
||||
// reset to default
|
||||
FNINIT(Op);
|
||||
@@ -1109,39 +1086,33 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
const auto Size = GetSrcSize(Op);
|
||||
OrderedNode* Mem = MakeSegmentAddress(Op, Op->Src[0]);
|
||||
Ref Mem = MakeSegmentAddress(Op, Op->Src[0]);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 1));
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, MemLocation, Size);
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, Mem, _Constant(Size * 1), Size, MEM_OFFSET_SXTX, 1);
|
||||
auto Top = ReconstructX87StateFromFSW(NewFSW);
|
||||
{
|
||||
// FTW
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 2));
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, MemLocation, Size));
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, Mem, _Constant(Size * 2), Size, MEM_OFFSET_SXTX, 1));
|
||||
}
|
||||
|
||||
OrderedNode* ST0Location = _Add(OpSize::i64Bit, Mem, _Constant(Size * 7));
|
||||
|
||||
auto OneConst = _Constant(1);
|
||||
auto SevenConst = _Constant(7);
|
||||
auto TenConst = _Constant(10);
|
||||
|
||||
auto low = _Constant(~0ULL);
|
||||
auto high = _Constant(0xFFFF);
|
||||
OrderedNode* Mask = _VCastFromGPR(16, 8, low);
|
||||
Ref Mask = _VCastFromGPR(16, 8, low);
|
||||
Mask = _VInsGPR(16, 8, 1, Mask, high);
|
||||
|
||||
for (int i = 0; i < 7; ++i) {
|
||||
OrderedNode* Reg = _LoadMem(FPRClass, 16, ST0Location, 1);
|
||||
Ref Reg = _LoadMem(FPRClass, 16, Mem, _Constant((Size * 7) + (10 * i)), 1, MEM_OFFSET_SXTX, 1);
|
||||
// Mask off the top bits
|
||||
Reg = _VAnd(16, 16, Reg, Mask);
|
||||
|
||||
_StoreContextIndexed(Reg, Top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
ST0Location = _Add(OpSize::i64Bit, ST0Location, TenConst);
|
||||
Top = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, Top, OneConst), SevenConst);
|
||||
}
|
||||
|
||||
@@ -1150,9 +1121,8 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
// Lower 64bits [63:0]
|
||||
// upper 16 bits [79:64]
|
||||
|
||||
OrderedNode* Reg = _LoadMem(FPRClass, 8, ST0Location, 1);
|
||||
ST0Location = _Add(OpSize::i64Bit, ST0Location, _Constant(8));
|
||||
OrderedNode* RegHigh = _LoadMem(FPRClass, 2, ST0Location, 1);
|
||||
Ref Reg = _LoadMem(FPRClass, 8, Mem, _Constant((Size * 7) + (10 * 7)), 1, MEM_OFFSET_SXTX, 1);
|
||||
Ref RegHigh = _LoadMem(FPRClass, 2, Mem, _Constant((Size * 7) + (10 * 7) + 8), 1, MEM_OFFSET_SXTX, 1);
|
||||
Reg = _VInsElement(16, 2, 4, 0, Reg, RegHigh);
|
||||
_StoreContextIndexed(Reg, Top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
@@ -1160,7 +1130,7 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
void OpDispatchBuilder::X87FXAM(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
auto a = _LoadContextIndexed(top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
OrderedNode* Result = _VExtractToGPR(16, 8, a, 1);
|
||||
Ref Result = _VExtractToGPR(16, 8, a, 1);
|
||||
|
||||
// Extract the sign bit
|
||||
Result = _Bfe(OpSize::i64Bit, 1, 15, Result);
|
||||
@@ -1219,11 +1189,11 @@ void OpDispatchBuilder::X87FCMOV(OpcodeArgs) {
|
||||
auto ZeroConst = _Constant(0);
|
||||
auto AllOneConst = _Constant(0xffff'ffff'ffff'ffffull);
|
||||
|
||||
OrderedNode* SrcCond = SelectCC(CC, OpSize::i64Bit, AllOneConst, ZeroConst);
|
||||
OrderedNode* VecCond = _VDupFromGPR(16, 8, SrcCond);
|
||||
Ref SrcCond = SelectCC(CC, OpSize::i64Bit, AllOneConst, ZeroConst);
|
||||
Ref VecCond = _VDupFromGPR(16, 8, SrcCond);
|
||||
|
||||
auto top = GetX87Top();
|
||||
OrderedNode* arg;
|
||||
Ref arg;
|
||||
|
||||
auto mask = _Constant(7);
|
||||
|
||||
@@ -1246,7 +1216,7 @@ void OpDispatchBuilder::X87EMMS(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::X87FFREE(OpcodeArgs) {
|
||||
// Only sets the selected stack register's tag bits to EMPTY
|
||||
OrderedNode* top = GetX87Top();
|
||||
Ref top = GetX87Top();
|
||||
|
||||
// Implicit arg
|
||||
auto offset = _Constant(Op->OP & 7);
|
||||
|
||||
@@ -65,32 +65,30 @@ void OpDispatchBuilder::FNINITF64(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::X87LDENVF64(OpcodeArgs) {
|
||||
const auto Size = GetSrcSize(Op);
|
||||
OrderedNode* Mem = MakeSegmentAddress(Op, Op->Src[0]);
|
||||
Ref Mem = MakeSegmentAddress(Op, Op->Src[0]);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
// ignore the rounding precision, we're always 64-bit in F64.
|
||||
// extract rounding mode
|
||||
OrderedNode* roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
|
||||
Ref roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
|
||||
_SetRoundingMode(roundingMode);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 1));
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, MemLocation, Size);
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, Mem, _Constant(Size * 1), Size, MEM_OFFSET_SXTX, 1);
|
||||
ReconstructX87StateFromFSW(NewFSW);
|
||||
|
||||
{
|
||||
// FTW
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 2));
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, MemLocation, Size));
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, Mem, _Constant(Size * 2), Size, MEM_OFFSET_SXTX, 1));
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
void OpDispatchBuilder::X87FLDCWF64(OpcodeArgs) {
|
||||
OrderedNode* NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
// ignore the rounding precision, we're always 64-bit in F64.
|
||||
// extract rounding mode
|
||||
OrderedNode* roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
|
||||
Ref roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
|
||||
_SetRoundingMode(roundingMode);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
}
|
||||
@@ -105,8 +103,8 @@ void OpDispatchBuilder::FLDF64(OpcodeArgs) {
|
||||
|
||||
size_t read_width = (width == 80) ? 16 : width / 8;
|
||||
|
||||
OrderedNode* data {};
|
||||
OrderedNode* converted {};
|
||||
Ref data {};
|
||||
Ref converted {};
|
||||
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Read from memory
|
||||
@@ -147,8 +145,8 @@ void OpDispatchBuilder::FBLDF64(OpcodeArgs) {
|
||||
SetX87Top(top);
|
||||
|
||||
// Read from memory
|
||||
OrderedNode* data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 16, Op->Flags);
|
||||
OrderedNode* converted = _F80BCDLoad(data);
|
||||
Ref data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 16, Op->Flags);
|
||||
Ref converted = _F80BCDLoad(data);
|
||||
converted = _F80CVT(8, converted);
|
||||
_StoreContextIndexed(converted, top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
@@ -157,7 +155,7 @@ void OpDispatchBuilder::FBSTPF64(OpcodeArgs) {
|
||||
auto orig_top = GetX87Top();
|
||||
auto data = _LoadContextIndexed(orig_top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
OrderedNode* converted = _F80CVTTo(data, 8);
|
||||
Ref converted = _F80CVTTo(data, 8);
|
||||
converted = _F80BCDStore(converted);
|
||||
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, converted, 10, 1);
|
||||
@@ -241,7 +239,7 @@ void OpDispatchBuilder::FISTF64(OpcodeArgs) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
|
||||
auto orig_top = GetX87Top();
|
||||
OrderedNode* data = _LoadContextIndexed(orig_top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
Ref data = _LoadContextIndexed(orig_top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
if constexpr (Truncate) {
|
||||
data = _Float_ToGPR_ZS(Size == 4 ? 4 : 8, 8, data);
|
||||
} else {
|
||||
@@ -264,10 +262,10 @@ template void OpDispatchBuilder::FISTF64<true>(OpcodeArgs);
|
||||
template<size_t width, bool Integer, OpDispatchBuilder::OpResult ResInST0>
|
||||
void OpDispatchBuilder::FADDF64(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
OrderedNode* StackLocation = top;
|
||||
Ref StackLocation = top;
|
||||
|
||||
OrderedNode* arg {};
|
||||
OrderedNode* b {};
|
||||
Ref arg {};
|
||||
Ref b {};
|
||||
|
||||
auto mask = _Constant(7);
|
||||
|
||||
@@ -320,9 +318,9 @@ template void OpDispatchBuilder::FADDF64<32, true, OpDispatchBuilder::OpResult::
|
||||
template<size_t width, bool Integer, OpDispatchBuilder::OpResult ResInST0>
|
||||
void OpDispatchBuilder::FMULF64(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
OrderedNode* StackLocation = top;
|
||||
OrderedNode* arg {};
|
||||
OrderedNode* b {};
|
||||
Ref StackLocation = top;
|
||||
Ref arg {};
|
||||
Ref b {};
|
||||
|
||||
auto mask = _Constant(7);
|
||||
|
||||
@@ -378,9 +376,9 @@ template void OpDispatchBuilder::FMULF64<32, true, OpDispatchBuilder::OpResult::
|
||||
template<size_t width, bool Integer, bool reverse, OpDispatchBuilder::OpResult ResInST0>
|
||||
void OpDispatchBuilder::FDIVF64(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
OrderedNode* StackLocation = top;
|
||||
OrderedNode* arg {};
|
||||
OrderedNode* b {};
|
||||
Ref StackLocation = top;
|
||||
Ref arg {};
|
||||
Ref b {};
|
||||
|
||||
auto mask = _Constant(7);
|
||||
|
||||
@@ -411,7 +409,7 @@ void OpDispatchBuilder::FDIVF64(OpcodeArgs) {
|
||||
|
||||
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
OrderedNode* result {};
|
||||
Ref result {};
|
||||
if constexpr (reverse) {
|
||||
result = _VFDiv(8, 8, b, a);
|
||||
} else {
|
||||
@@ -451,9 +449,9 @@ template void OpDispatchBuilder::FDIVF64<32, true, true, OpDispatchBuilder::OpRe
|
||||
template<size_t width, bool Integer, bool reverse, OpDispatchBuilder::OpResult ResInST0>
|
||||
void OpDispatchBuilder::FSUBF64(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
OrderedNode* StackLocation = top;
|
||||
OrderedNode* arg {};
|
||||
OrderedNode* b {};
|
||||
Ref StackLocation = top;
|
||||
Ref arg {};
|
||||
Ref b {};
|
||||
|
||||
auto mask = _Constant(7);
|
||||
|
||||
@@ -484,7 +482,7 @@ void OpDispatchBuilder::FSUBF64(OpcodeArgs) {
|
||||
|
||||
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
OrderedNode* result {};
|
||||
Ref result {};
|
||||
if constexpr (reverse) {
|
||||
result = _VFSub(8, 8, b, a);
|
||||
} else {
|
||||
@@ -543,7 +541,7 @@ void OpDispatchBuilder::FTSTF64(OpcodeArgs) {
|
||||
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
auto low = _Constant(0);
|
||||
OrderedNode* data = _VCastFromGPR(8, 8, low);
|
||||
Ref data = _VCastFromGPR(8, 8, low);
|
||||
|
||||
// We are going to clobber NZCV, make sure it's in a GPR first.
|
||||
GetNZCV();
|
||||
@@ -573,11 +571,11 @@ void OpDispatchBuilder::FXTRACTF64(OpcodeArgs) {
|
||||
|
||||
auto a = _LoadContextIndexed(orig_top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
auto gpr = _VExtractToGPR(8, 8, a, 0);
|
||||
OrderedNode* exp = _And(OpSize::i64Bit, gpr, _Constant(0x7ff0000000000000LL));
|
||||
Ref exp = _And(OpSize::i64Bit, gpr, _Constant(0x7ff0000000000000LL));
|
||||
exp = _Lshr(OpSize::i64Bit, exp, _Constant(52));
|
||||
exp = _Sub(OpSize::i64Bit, exp, _Constant(1023));
|
||||
exp = _Float_FromGPR_S(8, 8, exp);
|
||||
OrderedNode* sig = _And(OpSize::i64Bit, gpr, _Constant(0x800fffffffffffffLL));
|
||||
Ref sig = _And(OpSize::i64Bit, gpr, _Constant(0x800fffffffffffffLL));
|
||||
sig = _Or(OpSize::i64Bit, sig, _Constant(0x3ff0000000000000LL));
|
||||
sig = _VCastFromGPR(8, 8, sig);
|
||||
// Write to ST[TOP]
|
||||
@@ -591,8 +589,8 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
auto mask = _Constant(7);
|
||||
|
||||
OrderedNode* arg {};
|
||||
OrderedNode* b {};
|
||||
Ref arg {};
|
||||
Ref b {};
|
||||
|
||||
if (!Op->Src[0].IsNone()) {
|
||||
// Memory arg
|
||||
@@ -701,7 +699,7 @@ void OpDispatchBuilder::X87BinaryOpF64(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
|
||||
auto mask = _Constant(7);
|
||||
OrderedNode* st1 = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, top, _Constant(1)), mask);
|
||||
Ref st1 = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, top, _Constant(1)), mask);
|
||||
|
||||
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
st1 = _LoadContextIndexed(st1, 8, MMBaseOffset(), 16, FPRClass);
|
||||
@@ -749,8 +747,8 @@ void OpDispatchBuilder::X87FYL2XF64(OpcodeArgs) {
|
||||
auto top = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, orig_top, _Constant(1)), _Constant(7));
|
||||
SetX87Top(top);
|
||||
|
||||
OrderedNode* st0 = _LoadContextIndexed(orig_top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
OrderedNode* st1 = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
Ref st0 = _LoadContextIndexed(orig_top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
Ref st1 = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
if (Plus1) {
|
||||
auto one = _VCastFromGPR(8, 8, _Constant(0x3FF0000000000000));
|
||||
@@ -791,7 +789,7 @@ void OpDispatchBuilder::X87ATANF64(OpcodeArgs) {
|
||||
SetX87Top(top);
|
||||
|
||||
auto a = _LoadContextIndexed(orig_top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
OrderedNode* st1 = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
Ref st1 = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
auto result = _F64ATAN(st1, a);
|
||||
|
||||
@@ -822,73 +820,60 @@ void OpDispatchBuilder::X87FNSAVEF64(OpcodeArgs) {
|
||||
// 4 bytes : data pointer selector
|
||||
|
||||
const auto Size = GetDstSize(Op);
|
||||
OrderedNode* Mem = MakeSegmentAddress(Op, Op->Dest);
|
||||
OrderedNode* Top = GetX87Top();
|
||||
Ref Mem = MakeSegmentAddress(Op, Op->Dest);
|
||||
Ref Top = GetX87Top();
|
||||
{
|
||||
auto FCW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreMem(GPRClass, Size, Mem, FCW, Size);
|
||||
}
|
||||
|
||||
{
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 1));
|
||||
_StoreMem(GPRClass, Size, MemLocation, ReconstructFSW(), Size);
|
||||
}
|
||||
{ _StoreMem(GPRClass, Size, ReconstructFSW(), Mem, _Constant(Size * 1), Size, MEM_OFFSET_SXTX, 1); }
|
||||
|
||||
auto ZeroConst = _Constant(0);
|
||||
|
||||
{
|
||||
// FTW
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 2));
|
||||
_StoreMem(GPRClass, Size, MemLocation, GetX87FTW(), Size);
|
||||
_StoreMem(GPRClass, Size, GetX87FTW(), Mem, _Constant(Size * 2), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction Offset
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 3));
|
||||
_StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size);
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 3), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction CS selector (+ Opcode)
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 4));
|
||||
_StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size);
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 4), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer offset
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 5));
|
||||
_StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size);
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 5), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer selector
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 6));
|
||||
_StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size);
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, _Constant(Size * 6), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
OrderedNode* ST0Location = _Add(OpSize::i64Bit, Mem, _Constant(Size * 7));
|
||||
|
||||
auto OneConst = _Constant(1);
|
||||
auto SevenConst = _Constant(7);
|
||||
auto TenConst = _Constant(10);
|
||||
for (int i = 0; i < 7; ++i) {
|
||||
OrderedNode* data = _LoadContextIndexed(Top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
Ref data = _LoadContextIndexed(Top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
data = _F80CVTTo(data, 8);
|
||||
_StoreMem(FPRClass, 16, ST0Location, data, 1);
|
||||
ST0Location = _Add(OpSize::i64Bit, ST0Location, TenConst);
|
||||
_StoreMem(FPRClass, 16, data, Mem, _Constant((Size * 7) + (i * 10)), 1, MEM_OFFSET_SXTX, 1);
|
||||
Top = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, Top, OneConst), SevenConst);
|
||||
}
|
||||
|
||||
// The final st(7) needs a bit of special handling here
|
||||
OrderedNode* data = _LoadContextIndexed(Top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
Ref data = _LoadContextIndexed(Top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
data = _F80CVTTo(data, 8);
|
||||
// ST7 broken in to two parts
|
||||
// Lower 64bits [63:0]
|
||||
// upper 16 bits [79:64]
|
||||
_StoreMem(FPRClass, 8, ST0Location, data, 1);
|
||||
ST0Location = _Add(OpSize::i64Bit, ST0Location, _Constant(8));
|
||||
_StoreMem(FPRClass, 8, data, Mem, _Constant((Size * 7) + (7 * 10)), 1, MEM_OFFSET_SXTX, 1);
|
||||
auto topBytes = _VDupElement(16, 2, data, 4);
|
||||
_StoreMem(FPRClass, 2, ST0Location, topBytes, 1);
|
||||
_StoreMem(FPRClass, 2, topBytes, Mem, _Constant((Size * 7) + (7 * 10) + 8), 1, MEM_OFFSET_SXTX, 1);
|
||||
|
||||
// reset to default
|
||||
FNINIT(Op);
|
||||
@@ -898,12 +883,12 @@ void OpDispatchBuilder::X87FNSAVEF64(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::X87FRSTORF64(OpcodeArgs) {
|
||||
const auto Size = GetSrcSize(Op);
|
||||
OrderedNode* Mem = MakeSegmentAddress(Op, Op->Src[0]);
|
||||
Ref Mem = MakeSegmentAddress(Op, Op->Src[0]);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
// ignore the rounding precision, we're always 64-bit in F64.
|
||||
// extract rounding mode
|
||||
OrderedNode* roundingMode = NewFCW;
|
||||
Ref roundingMode = NewFCW;
|
||||
auto roundShift = _Constant(10);
|
||||
auto roundMask = _Constant(3);
|
||||
roundingMode = _Lshr(OpSize::i32Bit, roundingMode, roundShift);
|
||||
@@ -912,36 +897,30 @@ void OpDispatchBuilder::X87FRSTORF64(OpcodeArgs) {
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 1));
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, MemLocation, Size);
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, Mem, _Constant(Size * 1), Size, MEM_OFFSET_SXTX, 1);
|
||||
auto Top = ReconstructX87StateFromFSW(NewFSW);
|
||||
|
||||
{
|
||||
// FTW
|
||||
OrderedNode* MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 2));
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, MemLocation, Size));
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, Mem, _Constant(Size * 2), Size, MEM_OFFSET_SXTX, 1));
|
||||
}
|
||||
|
||||
OrderedNode* ST0Location = _Add(OpSize::i64Bit, Mem, _Constant(Size * 7));
|
||||
|
||||
auto OneConst = _Constant(1);
|
||||
auto SevenConst = _Constant(7);
|
||||
auto TenConst = _Constant(10);
|
||||
|
||||
auto low = _Constant(~0ULL);
|
||||
auto high = _Constant(0xFFFF);
|
||||
OrderedNode* Mask = _VCastFromGPR(16, 8, low);
|
||||
Ref Mask = _VCastFromGPR(16, 8, low);
|
||||
Mask = _VInsGPR(16, 8, 1, Mask, high);
|
||||
|
||||
for (int i = 0; i < 7; ++i) {
|
||||
OrderedNode* Reg = _LoadMem(FPRClass, 16, ST0Location, 1);
|
||||
Ref Reg = _LoadMem(FPRClass, 16, Mem, _Constant((Size * 7) + (i * 10)), 1, MEM_OFFSET_SXTX, 1);
|
||||
// Mask off the top bits
|
||||
Reg = _VAnd(16, 16, Reg, Mask);
|
||||
// Convert to double precision
|
||||
Reg = _F80CVT(8, Reg);
|
||||
_StoreContextIndexed(Reg, Top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
|
||||
ST0Location = _Add(OpSize::i64Bit, ST0Location, TenConst);
|
||||
Top = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, Top, OneConst), SevenConst);
|
||||
}
|
||||
|
||||
@@ -950,9 +929,8 @@ void OpDispatchBuilder::X87FRSTORF64(OpcodeArgs) {
|
||||
// Lower 64bits [63:0]
|
||||
// upper 16 bits [79:64]
|
||||
|
||||
OrderedNode* Reg = _LoadMem(FPRClass, 8, ST0Location, 1);
|
||||
ST0Location = _Add(OpSize::i64Bit, ST0Location, _Constant(8));
|
||||
OrderedNode* RegHigh = _LoadMem(FPRClass, 2, ST0Location, 1);
|
||||
Ref Reg = _LoadMem(FPRClass, 8, Mem, _Constant((Size * 7) + (7 * 10)), 1, MEM_OFFSET_SXTX, 1);
|
||||
Ref RegHigh = _LoadMem(FPRClass, 2, Mem, _Constant((Size * 7) + (7 * 10) + 8), 1, MEM_OFFSET_SXTX, 1);
|
||||
Reg = _VInsElement(16, 2, 4, 0, Reg, RegHigh);
|
||||
Reg = _F80CVT(8, Reg); // Convert to double precision
|
||||
_StoreContextIndexed(Reg, Top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
@@ -963,7 +941,7 @@ void OpDispatchBuilder::X87FRSTORF64(OpcodeArgs) {
|
||||
void OpDispatchBuilder::X87FXAMF64(OpcodeArgs) {
|
||||
auto top = GetX87Top();
|
||||
auto a = _LoadContextIndexed(top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
OrderedNode* Result = _VExtractToGPR(8, 8, a, 0);
|
||||
Ref Result = _VExtractToGPR(8, 8, a, 0);
|
||||
|
||||
// Extract the sign bit
|
||||
Result = _Bfe(OpSize::i64Bit, 1, 63, Result);
|
||||
|
||||
@@ -102,10 +102,10 @@ std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
{0x6B, 1, X86InstInfo{"IMUL", TYPE_INST, FLAGS_MODRM | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
|
||||
// This should just throw a GP
|
||||
{0x6C, 1, X86InstInfo{"INSB", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x6D, 1, X86InstInfo{"INSW", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x6E, 1, X86InstInfo{"OUTS", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x6F, 1, X86InstInfo{"OUTS", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x6C, 1, X86InstInfo{"INSB", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x6D, 1, X86InstInfo{"INSW", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x6E, 1, X86InstInfo{"OUTS", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x6F, 1, X86InstInfo{"OUTS", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0x70, 1, X86InstInfo{"JO", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
{0x71, 1, X86InstInfo{"JNO", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
@@ -183,24 +183,24 @@ std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
{0xE3, 1, X86InstInfo{"JrCXZ", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
|
||||
// Should just throw GP
|
||||
{0xE4, 2, X86InstInfo{"IN", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0xE6, 2, X86InstInfo{"OUT", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0xE4, 2, X86InstInfo{"IN", TYPE_INST, FLAGS_NONE, 1, nullptr}},
|
||||
{0xE6, 2, X86InstInfo{"OUT", TYPE_INST, FLAGS_NONE, 1, nullptr}},
|
||||
|
||||
{0xE8, 1, X86InstInfo{"CALL", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_BLOCK_END , 4, nullptr}},
|
||||
{0xE9, 1, X86InstInfo{"JMP", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_BLOCK_END , 4, nullptr}},
|
||||
{0xEB, 1, X86InstInfo{"JMP", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT | FLAGS_BLOCK_END , 1, nullptr}},
|
||||
|
||||
// Should just throw GP
|
||||
{0xEC, 2, X86InstInfo{"IN", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0xEE, 2, X86InstInfo{"OUT", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0xEC, 2, X86InstInfo{"IN", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xEE, 2, X86InstInfo{"OUT", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0xF1, 1, X86InstInfo{"INT1", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xF4, 1, X86InstInfo{"HLT", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0xF5, 1, X86InstInfo{"CMC", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xF8, 1, X86InstInfo{"CLC", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xF9, 1, X86InstInfo{"STC", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xFA, 1, X86InstInfo{"CLI", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{0xFB, 1, X86InstInfo{"STI", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{0xFA, 1, X86InstInfo{"CLI", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xFB, 1, X86InstInfo{"STI", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xFC, 1, X86InstInfo{"CLD", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xFD, 1, X86InstInfo{"STD", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
|
||||
@@ -32,7 +32,7 @@ std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGroupOps = [
|
||||
{OPD(TYPE_GROUP_6, PF_NONE, 0), 1, X86InstInfo{"SLDT", TYPE_UNDEC, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_6, PF_NONE, 1), 1, X86InstInfo{"STR", TYPE_PRIV, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_6, PF_NONE, 2), 1, X86InstInfo{"LLDT", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_6, PF_NONE, 3), 1, X86InstInfo{"LTR", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_6, PF_NONE, 3), 1, X86InstInfo{"LTR", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_6, PF_NONE, 4), 1, X86InstInfo{"VERR", TYPE_UNDEC, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_6, PF_NONE, 5), 1, X86InstInfo{"VERW", TYPE_UNDEC, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_6, PF_NONE, 6), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -41,7 +41,7 @@ std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGroupOps = [
|
||||
{OPD(TYPE_GROUP_6, PF_F3, 0), 1, X86InstInfo{"SLDT", TYPE_UNDEC, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_6, PF_F3, 1), 1, X86InstInfo{"STR", TYPE_PRIV, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_6, PF_F3, 2), 1, X86InstInfo{"LLDT", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_6, PF_F3, 3), 1, X86InstInfo{"LTR", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_6, PF_F3, 3), 1, X86InstInfo{"LTR", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_6, PF_F3, 4), 1, X86InstInfo{"VERR", TYPE_UNDEC, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_6, PF_F3, 5), 1, X86InstInfo{"VERW", TYPE_UNDEC, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_6, PF_F3, 6), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -50,7 +50,7 @@ std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGroupOps = [
|
||||
{OPD(TYPE_GROUP_6, PF_66, 0), 1, X86InstInfo{"SLDT", TYPE_UNDEC, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_6, PF_66, 1), 1, X86InstInfo{"STR", TYPE_PRIV, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_6, PF_66, 2), 1, X86InstInfo{"LLDT", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_6, PF_66, 3), 1, X86InstInfo{"LTR", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_6, PF_66, 3), 1, X86InstInfo{"LTR", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_6, PF_66, 4), 1, X86InstInfo{"VERR", TYPE_UNDEC, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_6, PF_66, 5), 1, X86InstInfo{"VERW", TYPE_UNDEC, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_6, PF_66, 6), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -59,7 +59,7 @@ std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGroupOps = [
|
||||
{OPD(TYPE_GROUP_6, PF_F2, 0), 1, X86InstInfo{"SLDT", TYPE_UNDEC, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_6, PF_F2, 1), 1, X86InstInfo{"STR", TYPE_PRIV, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_6, PF_F2, 2), 1, X86InstInfo{"LLDT", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_6, PF_F2, 3), 1, X86InstInfo{"LTR", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_6, PF_F2, 3), 1, X86InstInfo{"LTR", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_6, PF_F2, 4), 1, X86InstInfo{"VERR", TYPE_UNDEC, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_6, PF_F2, 5), 1, X86InstInfo{"VERW", TYPE_UNDEC, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_6, PF_F2, 6), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -72,7 +72,7 @@ std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGroupOps = [
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 3), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 4), 1, X86InstInfo{"SMSW", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 6), 1, X86InstInfo{"LMSW", TYPE_PRIV, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 6), 1, X86InstInfo{"LMSW", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 7), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 0), 1, X86InstInfo{"SGDT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
@@ -81,7 +81,7 @@ std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGroupOps = [
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 3), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 4), 1, X86InstInfo{"SMSW", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 6), 1, X86InstInfo{"LMSW", TYPE_PRIV, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 6), 1, X86InstInfo{"LMSW", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 7), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_7, PF_66, 0), 1, X86InstInfo{"SGDT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
@@ -90,7 +90,7 @@ std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGroupOps = [
|
||||
{OPD(TYPE_GROUP_7, PF_66, 3), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 4), 1, X86InstInfo{"SMSW", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 6), 1, X86InstInfo{"LMSW", TYPE_PRIV, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 6), 1, X86InstInfo{"LMSW", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 7), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 0), 1, X86InstInfo{"SGDT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
@@ -99,7 +99,7 @@ std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGroupOps = [
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 3), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 4), 1, X86InstInfo{"SMSW", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 6), 1, X86InstInfo{"LMSW", TYPE_PRIV, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 6), 1, X86InstInfo{"LMSW", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 7), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
// GROUP 8
|
||||
|
||||
@@ -15,8 +15,8 @@ std::array<X86InstInfo, MAX_SECOND_MODRM_TABLE_SIZE> SecondModRMTableOps = []()
|
||||
std::array<X86InstInfo, MAX_SECOND_MODRM_TABLE_SIZE> Table{};
|
||||
constexpr U8U8InfoStruct SecondaryModRMExtensionOpTable[] = {
|
||||
// REG /1
|
||||
{((0 << 3) | 0), 1, X86InstInfo{"MONITOR", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{((0 << 3) | 1), 1, X86InstInfo{"MWAIT", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{((0 << 3) | 0), 1, X86InstInfo{"MONITOR", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{((0 << 3) | 1), 1, X86InstInfo{"MWAIT", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{((0 << 3) | 2), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{((0 << 3) | 3), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{((0 << 3) | 4), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -42,10 +42,10 @@ std::array<X86InstInfo, MAX_SECOND_MODRM_TABLE_SIZE> SecondModRMTableOps = []()
|
||||
{((2 << 3) | 4), 1, X86InstInfo{"STGI", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{((2 << 3) | 5), 1, X86InstInfo{"CLGI", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{((2 << 3) | 6), 1, X86InstInfo{"SKINIT", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{((2 << 3) | 7), 1, X86InstInfo{"INVLPGA", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{((2 << 3) | 7), 1, X86InstInfo{"INVLPGA", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
// REG /7
|
||||
{((3 << 3) | 0), 1, X86InstInfo{"SWAPGS", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{((3 << 3) | 0), 1, X86InstInfo{"SWAPGS", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{((3 << 3) | 1), 1, X86InstInfo{"RDTSCP", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{((3 << 3) | 2), 1, X86InstInfo{"MONITORX", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{((3 << 3) | 3), 1, X86InstInfo{"MWAITX", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
@@ -25,8 +25,8 @@ auto BaseOpsLambda = []() consteval {
|
||||
{0x03, 1, X86InstInfo{"LSL", TYPE_UNDEC, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x04, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x05, 1, X86InstInfo{"SYSCALL", TYPE_INST, DEFAULT_SYSCALL_FLAGS, 0, nullptr}},
|
||||
{0x06, 1, X86InstInfo{"CLTS", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x07, 1, X86InstInfo{"SYSRET", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x06, 1, X86InstInfo{"CLTS", TYPE_INST, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x07, 1, X86InstInfo{"SYSRET", TYPE_INST, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x08, 1, X86InstInfo{"INVD", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x09, 1, X86InstInfo{"WBINVD", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x0A, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
@@ -47,8 +47,8 @@ auto BaseOpsLambda = []() consteval {
|
||||
{0x18, 1, X86InstInfo{"", TYPE_GROUP_16, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x19, 7, X86InstInfo{"NOP", TYPE_INST, FLAGS_MODRM | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
|
||||
{0x20, 2, X86InstInfo{"MOV", TYPE_PRIV, GenFlagsSameSize(SIZE_64BIT) | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x22, 2, X86InstInfo{"MOV", TYPE_PRIV, GenFlagsSameSize(SIZE_64BIT) | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x20, 2, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x22, 2, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x24, 4, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x28, 1, X86InstInfo{"MOVAPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x29, 1, X86InstInfo{"MOVAPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
@@ -59,12 +59,12 @@ auto BaseOpsLambda = []() consteval {
|
||||
{0x2E, 1, X86InstInfo{"UCOMISS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{0x2F, 1, X86InstInfo{"COMISS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{0x30, 1, X86InstInfo{"WRMSR", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x30, 1, X86InstInfo{"WRMSR", TYPE_INST, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x31, 1, X86InstInfo{"RDTSC", TYPE_INST, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x32, 1, X86InstInfo{"RDMSR", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x33, 1, X86InstInfo{"RDPMC", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x34, 1, X86InstInfo{"SYSENTER", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x35, 1, X86InstInfo{"SYSEXIT", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x32, 1, X86InstInfo{"RDMSR", TYPE_INST, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x33, 1, X86InstInfo{"RDPMC", TYPE_INST, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x34, 1, X86InstInfo{"SYSENTER", TYPE_INST, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x35, 1, X86InstInfo{"SYSEXIT", TYPE_INST, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x36, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x38, 1, X86InstInfo{"", TYPE_0F38_TABLE, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x39, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
|
||||
@@ -227,8 +227,7 @@ struct ThunkHandler_impl final : public ThunkHandler {
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
|
||||
if (GPRSize == 8) {
|
||||
emit->_StoreRegister(emit->_Constant(Entrypoint), false, offsetof(Core::CPUState, gregs[X86State::REG_R11]), IR::GPRClass,
|
||||
IR::GPRFixedClass, GPRSize);
|
||||
emit->_StoreRegister(emit->_Constant(Entrypoint), X86State::REG_R11, IR::GPRClass, GPRSize);
|
||||
} else {
|
||||
emit->_StoreContext(GPRSize, IR::FPRClass, emit->_VCastFromGPR(8, 8, emit->_Constant(Entrypoint)), offsetof(Core::CPUState, mm[0][0]));
|
||||
}
|
||||
|
||||
@@ -59,20 +59,20 @@ IR::IRListView* AOTIRInlineEntry::GetIRData() {
|
||||
}
|
||||
|
||||
void AOTIRCaptureCacheEntry::AppendAOTIRCaptureCache(uint64_t GuestRIP, uint64_t Start, uint64_t Length, uint64_t Hash,
|
||||
FEXCore::IR::IRListView* IRList, FEXCore::IR::RegisterAllocationData* RAData) {
|
||||
const FEXCore::IR::IRListView& IRList, const FEXCore::IR::RegisterAllocationData* RAData) {
|
||||
auto Inserted = Index.emplace(GuestRIP, Stream->Offset());
|
||||
|
||||
if (Inserted.second) {
|
||||
// GuestHash
|
||||
Stream->Write((const char*)&Hash, sizeof(Hash));
|
||||
|
||||
// GuestLength
|
||||
Stream->Write((const char*)&Length, sizeof(Length));
|
||||
AOTIRInlineEntry entry {
|
||||
.GuestHash = Hash,
|
||||
.GuestLength = Length,
|
||||
};
|
||||
Stream->Write((const char*)&entry, sizeof(entry));
|
||||
|
||||
RAData->Serialize(*Stream);
|
||||
|
||||
// IRData (inline)
|
||||
IRList->Serialize(*Stream);
|
||||
IRList.Serialize(*Stream);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -170,25 +170,23 @@ void AOTIRCaptureCache::FinalizeAOTIRCache() {
|
||||
stream->Write(&Zero, 1);
|
||||
}
|
||||
|
||||
// AOTIRInlineIndex
|
||||
const auto FnCount = Entry.Index.size();
|
||||
const size_t DataBase = -stream->Offset();
|
||||
|
||||
stream->Write((const char*)&FnCount, sizeof(FnCount));
|
||||
stream->Write((const char*)&DataBase, sizeof(DataBase));
|
||||
AOTIRInlineIndex index {
|
||||
.Count = Entry.Index.size(),
|
||||
.DataBase = -stream->Offset(),
|
||||
};
|
||||
stream->Write((const char*)&index, sizeof(index));
|
||||
|
||||
for (const auto& [GuestStart, DataOffset] : Entry.Index) {
|
||||
// AOTIRInlineIndexEntry
|
||||
AOTIRInlineIndexEntry entry {
|
||||
.GuestStart = GuestStart,
|
||||
.DataOffset = DataOffset,
|
||||
};
|
||||
|
||||
// GuestStart
|
||||
stream->Write((const char*)&GuestStart, sizeof(GuestStart));
|
||||
|
||||
// DataOffset
|
||||
stream->Write((const char*)&DataOffset, sizeof(DataOffset));
|
||||
stream->Write((const char*)&entry, sizeof(entry));
|
||||
}
|
||||
|
||||
// End of file header
|
||||
const auto IndexSize = FnCount * sizeof(FEXCore::IR::AOTIRInlineIndexEntry) + sizeof(DataBase) + sizeof(FnCount);
|
||||
const auto IndexSize = sizeof(AOTIRInlineIndex) + index.Count * sizeof(FEXCore::IR::AOTIRInlineIndexEntry);
|
||||
stream->Write((const char*)&IndexSize, sizeof(IndexSize));
|
||||
stream->Write(String.c_str(), ModSize);
|
||||
stream->Write((const char*)&ModSize, sizeof(ModSize));
|
||||
@@ -261,8 +259,23 @@ void AOTIRCaptureCache::WriteFilesWithCode(const Context::AOTIRCodeFileWriterFn&
|
||||
}
|
||||
}
|
||||
|
||||
AOTIRCaptureCache::PreGenerateIRFetchResult
|
||||
AOTIRCaptureCache::PreGenerateIRFetch(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, FEXCore::IR::IRListView* IRList) {
|
||||
// IRStorageBase with memory owned by IR cache
|
||||
class IRInlineStorage : public IRStorageBase {
|
||||
AOTIRInlineEntry& entry;
|
||||
public:
|
||||
IRInlineStorage(AOTIRInlineEntry& entry)
|
||||
: entry(entry) {}
|
||||
const RegisterAllocationData* RAData() override {
|
||||
return entry.GetRAData();
|
||||
}
|
||||
IRListView GetIRView() override {
|
||||
return entry.GetIRData();
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
std::optional<AOTIRCaptureCache::PreGenerateIRFetchResult>
|
||||
AOTIRCaptureCache::PreGenerateIRFetch(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) {
|
||||
auto AOTIRCacheEntry = CTX->SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
|
||||
|
||||
PreGenerateIRFetchResult Result {};
|
||||
@@ -270,7 +283,7 @@ AOTIRCaptureCache::PreGenerateIRFetch(FEXCore::Core::InternalThreadState* Thread
|
||||
if (AOTIRCacheEntry.Entry) {
|
||||
AOTIRCacheEntry.Entry->ContainsCode = true;
|
||||
|
||||
if (IRList == nullptr && CTX->Config.AOTIRLoad()) {
|
||||
if (CTX->Config.AOTIRLoad()) {
|
||||
auto Mod = AOTIRCacheEntry.Entry->Array;
|
||||
|
||||
if (Mod != nullptr) {
|
||||
@@ -281,14 +294,12 @@ AOTIRCaptureCache::PreGenerateIRFetch(FEXCore::Core::InternalThreadState* Thread
|
||||
auto MappedStart = GuestRIP;
|
||||
auto hash = XXH3_64bits((void*)MappedStart, AOTEntry->GuestLength);
|
||||
if (hash == AOTEntry->GuestHash) {
|
||||
Result.IRList = AOTEntry->GetIRData();
|
||||
Result.IR = fextl::make_unique<IRInlineStorage>(*AOTEntry);
|
||||
// LogMan::Msg::DFmt("using {} + {:x} -> {:x}\n", file->second.fileid, AOTEntry->first, GuestRIP);
|
||||
|
||||
Result.RAData = AOTEntry->GetRAData()->CreateCopy();
|
||||
Result.DebugData = new FEXCore::Core::DebugData();
|
||||
Result.StartAddr = MappedStart;
|
||||
Result.Length = AOTEntry->GuestLength;
|
||||
Result.GeneratedIR = true;
|
||||
return Result;
|
||||
} else {
|
||||
LogMan::Msg::IFmt("AOTIR: hash check failed {:x}\n", MappedStart);
|
||||
}
|
||||
@@ -299,12 +310,12 @@ AOTIRCaptureCache::PreGenerateIRFetch(FEXCore::Core::InternalThreadState* Thread
|
||||
}
|
||||
}
|
||||
|
||||
return Result;
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
bool AOTIRCaptureCache::PostCompileCode(FEXCore::Core::InternalThreadState* Thread, void* CodePtr, uint64_t GuestRIP, uint64_t StartAddr,
|
||||
uint64_t Length, FEXCore::IR::RegisterAllocationData::UniquePtr RAData,
|
||||
FEXCore::IR::IRListView* IRList, FEXCore::Core::DebugData* DebugData, bool GeneratedIR) {
|
||||
uint64_t Length, fextl::unique_ptr<FEXCore::IR::IRStorageBase> IR,
|
||||
FEXCore::Core::DebugData* DebugData, bool GeneratedIR) {
|
||||
|
||||
// Both generated ir and LibraryJITName need a named region lookup
|
||||
if (GeneratedIR || CTX->Config.LibraryJITNaming() || CTX->Config.GDBSymbols()) {
|
||||
@@ -321,24 +332,20 @@ bool AOTIRCaptureCache::PostCompileCode(FEXCore::Core::InternalThreadState* Thre
|
||||
}
|
||||
|
||||
// Add to AOT cache if aot generation is enabled
|
||||
if (GeneratedIR && RAData && (CTX->Config.AOTIRCapture() || CTX->Config.AOTIRGenerate())) {
|
||||
if (GeneratedIR && IR->RAData() && (CTX->Config.AOTIRCapture() || CTX->Config.AOTIRGenerate())) {
|
||||
|
||||
auto hash = XXH3_64bits((void*)StartAddr, Length);
|
||||
|
||||
auto LocalRIP = GuestRIP - AOTIRCacheEntry.VAFileStart;
|
||||
auto LocalStartAddr = StartAddr - AOTIRCacheEntry.VAFileStart;
|
||||
auto FileId = AOTIRCacheEntry.Entry->FileId;
|
||||
// The underlying pointer and the unique_ptr deleter for RAData must
|
||||
// be marshalled separately to the lambda below. Otherwise, the
|
||||
// lambda can't be used as an std::function due to being non-copyable
|
||||
auto RADataCopy = RAData->CreateCopy();
|
||||
auto RADataCopyDeleter = RADataCopy.get_deleter();
|
||||
auto IRListCopy = IRList->CreateCopy();
|
||||
|
||||
// The lambda is converted to std::function. This is tricky to refactor so it doesn't allocate memory through glibc.
|
||||
// NOTE: unique_ptr must be passed as a raw pointer since std::function requires lambda captures to be copyable
|
||||
FEXCore::Allocator::YesIKnowImNotSupposedToUseTheGlibcAllocator glibc;
|
||||
AOTIRCaptureCacheWriteoutQueue_Append(
|
||||
[this, LocalRIP, LocalStartAddr, Length, hash, IRListCopy, RADataCopy = RADataCopy.release(), RADataCopyDeleter, FileId]() {
|
||||
AOTIRCaptureCacheWriteoutQueue_Append([this, LocalRIP, LocalStartAddr, Length, hash, IRRaw = IR.release(), FileId]() {
|
||||
fextl::unique_ptr<FEXCore::IR::IRStorageBase> IR(IRRaw);
|
||||
|
||||
// It is guaranteed via AOTIRCaptureCacheWriteoutLock and AOTIRCaptureCacheWriteoutFlusing that this will not run concurrently
|
||||
// Memory coherency is guaranteed via AOTIRCaptureCacheWriteoutLock
|
||||
|
||||
@@ -349,9 +356,7 @@ bool AOTIRCaptureCache::PostCompileCode(FEXCore::Core::InternalThreadState* Thre
|
||||
uint64_t tag = FEXCore::IR::AOTIR_COOKIE;
|
||||
AotFile->Stream->Write(&tag, sizeof(tag));
|
||||
}
|
||||
AotFile->AppendAOTIRCaptureCache(LocalRIP, LocalStartAddr, Length, hash, IRListCopy, RADataCopy);
|
||||
RADataCopyDeleter(RADataCopy);
|
||||
delete IRListCopy;
|
||||
AotFile->AppendAOTIRCaptureCache(LocalRIP, LocalStartAddr, Length, hash, IR->GetIRView(), IR->RAData());
|
||||
});
|
||||
|
||||
if (CTX->Config.AOTIRGenerate()) {
|
||||
@@ -366,9 +371,6 @@ bool AOTIRCaptureCache::PostCompileCode(FEXCore::Core::InternalThreadState* Thre
|
||||
if (GeneratedIR) {
|
||||
// If the IR doesn't need to be retained then we can just delete it now
|
||||
delete DebugData;
|
||||
if (IRList->IsCopy()) {
|
||||
delete IRList;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
#pragma once
|
||||
|
||||
#include "Interface/IR/RegisterAllocationData.h"
|
||||
#include "Interface/IR/IntrusiveIRList.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
@@ -74,8 +75,8 @@ struct AOTIRCaptureCacheEntry {
|
||||
fextl::unique_ptr<FEXCore::Context::AOTIRWriter> Stream;
|
||||
fextl::map<uint64_t, uint64_t> Index;
|
||||
|
||||
void AppendAOTIRCaptureCache(uint64_t GuestRIP, uint64_t Start, uint64_t Length, uint64_t Hash, FEXCore::IR::IRListView* IRList,
|
||||
FEXCore::IR::RegisterAllocationData* RAData);
|
||||
void AppendAOTIRCaptureCache(uint64_t GuestRIP, uint64_t Start, uint64_t Length, uint64_t Hash, const FEXCore::IR::IRListView& IRList,
|
||||
const FEXCore::IR::RegisterAllocationData* RAData);
|
||||
};
|
||||
|
||||
struct AOTIRCacheEntry {
|
||||
@@ -103,19 +104,16 @@ public:
|
||||
void WriteFilesWithCode(const Context::AOTIRCodeFileWriterFn& Writer);
|
||||
|
||||
struct PreGenerateIRFetchResult {
|
||||
FEXCore::IR::IRListView* IRList {};
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData {};
|
||||
fextl::unique_ptr<IRStorageBase> IR;
|
||||
FEXCore::Core::DebugData* DebugData {};
|
||||
uint64_t StartAddr {};
|
||||
uint64_t Length {};
|
||||
bool GeneratedIR {};
|
||||
};
|
||||
[[nodiscard]]
|
||||
PreGenerateIRFetchResult PreGenerateIRFetch(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, FEXCore::IR::IRListView* IRList);
|
||||
std::optional<PreGenerateIRFetchResult> PreGenerateIRFetch(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP);
|
||||
|
||||
bool PostCompileCode(FEXCore::Core::InternalThreadState* Thread, void* CodePtr, uint64_t GuestRIP, uint64_t StartAddr, uint64_t Length,
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData, FEXCore::IR::IRListView* IRList,
|
||||
FEXCore::Core::DebugData* DebugData, bool GeneratedIR);
|
||||
fextl::unique_ptr<FEXCore::IR::IRStorageBase> IR, FEXCore::Core::DebugData* DebugData, bool GeneratedIR);
|
||||
|
||||
AOTIRCacheEntry* LoadAOTIRCacheEntry(const fextl::string& filename);
|
||||
void UnloadAOTIRCacheEntry(AOTIRCacheEntry* Entry);
|
||||
|
||||
@@ -360,6 +360,17 @@ static_assert(std::is_trivially_copyable_v<OrderedNode>);
|
||||
static_assert(offsetof(OrderedNode, Header) == 0);
|
||||
static_assert(sizeof(OrderedNode) == (sizeof(OrderedNodeHeader) + sizeof(uint32_t)));
|
||||
|
||||
// This is temporary. We are transitioning away from OrderedNode's in favour of
|
||||
// flat Ref words. To ease porting, we have this typedef. Eventually OrderedNode
|
||||
// will be removed and this typedef will be replaced by something like:
|
||||
//
|
||||
// struct Ref {
|
||||
// uint Flags : 1;
|
||||
// uint ID : 23;
|
||||
// uint Reg : 8;
|
||||
// };
|
||||
using Ref = OrderedNode*;
|
||||
|
||||
struct RegisterClassType final {
|
||||
using value_type = uint32_t;
|
||||
|
||||
@@ -656,7 +667,6 @@ bool IsFragmentExit(FEXCore::IR::IROps Op);
|
||||
bool IsBlockExit(FEXCore::IR::IROps Op);
|
||||
|
||||
void Dump(fextl::stringstream* out, const IRListView* IR, IR::RegisterAllocationData* RAData);
|
||||
fextl::unique_ptr<IREmitter> Parse(FEXCore::Utils::IntrusivePooledAllocator& ThreadAllocator, fextl::stringstream& MapsStream);
|
||||
} // namespace FEXCore::IR
|
||||
|
||||
template<>
|
||||
|
||||
@@ -340,23 +340,38 @@
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i64Bit || Size == FEXCore::IR::OpSize::i128Bit"
|
||||
]
|
||||
},
|
||||
|
||||
"GPR = Copy GPR:$Source": {
|
||||
"Desc": ["GPR copy, generated by RA to split live ranges"],
|
||||
"DestSize": "8"
|
||||
},
|
||||
|
||||
"GPR = Swap1 GPR:$A, GPR:$B": {
|
||||
"Desc": ["GPR swap part 1, generated by RA. Returns value of first source.",
|
||||
"Destination must be second GPR."],
|
||||
"DestSize": "8"
|
||||
},
|
||||
|
||||
"GPR = Swap2": {
|
||||
"Desc": ["GPR swap part 2, generated by RA. Returns source source.",
|
||||
"Must immediately succeed Swap1 with no intervening instructions",
|
||||
"Kludge to workaround single destination restriction on IR",
|
||||
"Hopefully temporary"],
|
||||
"DestSize": "8"
|
||||
}
|
||||
},
|
||||
"StaticRA": {
|
||||
"SSA = LoadRegister i1:$IsAlias, u32:$Offset, RegisterClass:$Class, RegisterClass:$StaticClass, u8:#Size": {
|
||||
"Desc": ["Loads a value from the static-ra context with offset",
|
||||
"Dest = Ctx[Offset]"
|
||||
],
|
||||
"SSA = LoadRegister u32:$Reg, RegisterClass:$Class, u8:#Size": {
|
||||
"Desc": ["Loads a value from the given register",
|
||||
"Size must match the execution mode."],
|
||||
"DestSize": "Size"
|
||||
},
|
||||
|
||||
"StoreRegister SSA:$Value, i1:$IsPrewrite, u32:$Offset, RegisterClass:$Class, RegisterClass:$StaticClass, u8:#Size": {
|
||||
"StoreRegister SSA:$Value, u32:$Reg, RegisterClass:$Class, u8:#Size": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Stores a value to the static-ra context with offset",
|
||||
"Ctx[Offset] = Value",
|
||||
"Zero Extends if value's type is too small",
|
||||
"Truncates if value's type is too large"
|
||||
],
|
||||
"Desc": ["Stores a value to a given register.",
|
||||
"Size must match the execution mode."],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value) == $Class"
|
||||
@@ -534,6 +549,7 @@
|
||||
"Desc": ["Does a memory load to a single element of a vector.",
|
||||
"Leaves the rest of the vector's data intact.",
|
||||
"Matches arm64 ld1 semantics"],
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -555,6 +571,7 @@
|
||||
"The address is decremented by the value size while.",
|
||||
"The return value size is the size of the current operating mode"
|
||||
],
|
||||
"TiedSource": 1,
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size"
|
||||
},
|
||||
@@ -565,7 +582,7 @@
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "8"
|
||||
},
|
||||
"GPRPair = MemCpy i1:$IsAtomic, u8:$Size, GPR:$PrefixDest, GPR:$PrefixSrc, GPR:$AddrDest, GPR:$AddrSrc, GPR:$Length, GPR:$Direction": {
|
||||
"GPRPair = MemCpy i1:$IsAtomic, u8:$Size, GPR:$Dest, GPR:$Src, GPR:$Length, GPR:$Direction": {
|
||||
"Desc": ["Duplicates behaviour of x86 MOVS repeat",
|
||||
"Returns the final addresses of destination and src addresses after they have been incremented or decremented"
|
||||
],
|
||||
@@ -986,7 +1003,8 @@
|
||||
},
|
||||
"GPR = AddShift OpSize:#Size, GPR:$Src1, GPR:$Src2, ShiftType:$Shift{ShiftType::LSL}, u8:$ShiftAmount{0}": {
|
||||
"Desc": [ "Integer Add with shifted register",
|
||||
"Will truncate to 64 or 32bits"
|
||||
"Will truncate to 64 or 32bits",
|
||||
"Dest = Src1 + (Src2 << ShiftAmount)"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
@@ -1180,6 +1198,7 @@
|
||||
"Desc": ["Integer binary and"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"TiedSource": 0,
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"GPR = Andn OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
@@ -1320,6 +1339,7 @@
|
||||
"The bitfield is copied in to Dest[(Width + lsb):lsb]"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"TiedSource": 0,
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
@@ -1331,6 +1351,7 @@
|
||||
"The bitfield is copied in to Dest[Width:0]"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"TiedSource": 0,
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
@@ -1761,29 +1782,35 @@
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VShlI u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, u8:$BitShift": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VUShrI u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, u8:$BitShift": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VUShraI u8:#RegisterSize, u8:#ElementSize, FPR:$DestVector, FPR:$Vector, u8:$BitShift": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VSShrI u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, u8:$BitShift": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VUShrNI u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, u8:$BitShift": {
|
||||
"TiedSource": 0,
|
||||
"Desc": "Unsigned shifts right each element and then narrows to the next lower element size",
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize >> 1)"
|
||||
},
|
||||
|
||||
"FPR = VUShrNI2 u8:#RegisterSize, u8:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper, u8:$BitShift": {
|
||||
"TiedSource": 0,
|
||||
"Desc": ["Unsigned shifts right each element and then narrows to the next lower element size",
|
||||
"Inserts results in to the high elements of the first argument"
|
||||
],
|
||||
@@ -1815,10 +1842,12 @@
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
},
|
||||
"FPR = VSQXTN u8:#RegisterSize, u8:#ElementSize, FPR:$Vector": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize >> 1)"
|
||||
},
|
||||
"FPR = VSQXTN2 u8:#RegisterSize, u8:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize >> 1)"
|
||||
},
|
||||
@@ -1846,6 +1875,7 @@
|
||||
"Desc": ["Signed rounding shift right by immediate",
|
||||
"Exactly matching Arm64 srshr semantics"
|
||||
],
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -1853,6 +1883,7 @@
|
||||
"Desc": ["Signed satuating shift left by immediate",
|
||||
"Exactly matching Arm64 sqshl semantics"
|
||||
],
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -2061,42 +2092,52 @@
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
},
|
||||
"FPR = VUShl u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, FPR:$ShiftVector, i1:$RangeCheck": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VUShr u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, FPR:$ShiftVector, i1:$RangeCheck": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VSShr u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, FPR:$ShiftVector, i1:$RangeCheck": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VUShlS u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, FPR:$ShiftScalar": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VUShrS u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, FPR:$ShiftScalar": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VSShrS u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, FPR:$ShiftScalar": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VUShrSWide u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, FPR:$ShiftScalar": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VSShrSWide u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, FPR:$ShiftScalar": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VUShlSWide u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, FPR:$ShiftScalar": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VInsElement u8:#RegisterSize, u8:#ElementSize, u8:$DestIdx, u8:$SrcIdx, FPR:$DestVector, FPR:$SrcVector": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -2183,6 +2224,7 @@
|
||||
"Table is always treated as a 128bit register",
|
||||
"Indices matches destination size. Either 64bit or 128bit"
|
||||
],
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize"
|
||||
},
|
||||
"FPR = VBSL u8:#RegisterSize, FPR:$VectorMask, FPR:$VectorTrue, FPR:$VectorFalse": {
|
||||
|
||||
@@ -34,7 +34,7 @@ bool IsBlockExit(FEXCore::IR::IROps Op) {
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::IR::RegisterClassType IREmitter::WalkFindRegClass(OrderedNode* Node) {
|
||||
FEXCore::IR::RegisterClassType IREmitter::WalkFindRegClass(Ref Node) {
|
||||
auto Class = GetOpRegClass(Node);
|
||||
switch (Class) {
|
||||
case GPRClass:
|
||||
@@ -92,12 +92,12 @@ void IREmitter::ResetWorkingList() {
|
||||
CodeBlocks.clear();
|
||||
CurrentWriteCursor = nullptr;
|
||||
// This is necessary since we do "null" pointer checks
|
||||
InvalidNode = reinterpret_cast<OrderedNode*>(DualListData.ListAllocate(sizeof(OrderedNode)));
|
||||
InvalidNode = reinterpret_cast<Ref>(DualListData.ListAllocate(sizeof(OrderedNode)));
|
||||
memset(InvalidNode, 0, sizeof(OrderedNode));
|
||||
CurrentCodeBlock = nullptr;
|
||||
}
|
||||
|
||||
void IREmitter::ReplaceAllUsesWithRange(OrderedNode* Node, OrderedNode* NewNode, AllNodesIterator Begin, AllNodesIterator End) {
|
||||
void IREmitter::ReplaceAllUsesWithRange(Ref Node, Ref NewNode, AllNodesIterator Begin, AllNodesIterator End) {
|
||||
uintptr_t ListBegin = DualListData.ListBegin();
|
||||
auto NodeId = Node->Wrapped(ListBegin).ID();
|
||||
|
||||
@@ -122,19 +122,19 @@ void IREmitter::ReplaceAllUsesWithRange(OrderedNode* Node, OrderedNode* NewNode,
|
||||
}
|
||||
}
|
||||
|
||||
void IREmitter::ReplaceNodeArgument(OrderedNode* Node, uint8_t Arg, OrderedNode* NewArg) {
|
||||
void IREmitter::ReplaceNodeArgument(Ref Node, uint8_t Arg, Ref NewArg) {
|
||||
uintptr_t ListBegin = DualListData.ListBegin();
|
||||
uintptr_t DataBegin = DualListData.DataBegin();
|
||||
|
||||
FEXCore::IR::IROp_Header* IROp = Node->Op(DataBegin);
|
||||
OrderedNodeWrapper OldArgWrapper = IROp->Args[Arg];
|
||||
OrderedNode* OldArg = OldArgWrapper.GetNode(ListBegin);
|
||||
Ref OldArg = OldArgWrapper.GetNode(ListBegin);
|
||||
OldArg->RemoveUse();
|
||||
NewArg->AddUse();
|
||||
IROp->Args[Arg].NodeOffset = NewArg->Wrapped(ListBegin).NodeOffset;
|
||||
}
|
||||
|
||||
void IREmitter::RemoveArgUses(OrderedNode* Node) {
|
||||
void IREmitter::RemoveArgUses(Ref Node) {
|
||||
uintptr_t ListBegin = DualListData.ListBegin();
|
||||
uintptr_t DataBegin = DualListData.DataBegin();
|
||||
|
||||
@@ -147,13 +147,13 @@ void IREmitter::RemoveArgUses(OrderedNode* Node) {
|
||||
}
|
||||
}
|
||||
|
||||
void IREmitter::Remove(OrderedNode* Node) {
|
||||
void IREmitter::Remove(Ref Node) {
|
||||
RemoveArgUses(Node);
|
||||
|
||||
Node->Unlink(DualListData.ListBegin());
|
||||
}
|
||||
|
||||
IREmitter::IRPair<IROp_CodeBlock> IREmitter::CreateNewCodeBlockAfter(OrderedNode* insertAfter) {
|
||||
IREmitter::IRPair<IROp_CodeBlock> IREmitter::CreateNewCodeBlockAfter(Ref insertAfter) {
|
||||
auto OldCursor = GetWriteCursor();
|
||||
|
||||
auto CodeNode = CreateCodeNode();
|
||||
@@ -179,14 +179,14 @@ IREmitter::IRPair<IROp_CodeBlock> IREmitter::CreateNewCodeBlockAfter(OrderedNode
|
||||
return CodeNode;
|
||||
}
|
||||
|
||||
void IREmitter::SetCurrentCodeBlock(OrderedNode* Node) {
|
||||
void IREmitter::SetCurrentCodeBlock(Ref Node) {
|
||||
CurrentCodeBlock = Node;
|
||||
LOGMAN_THROW_A_FMT(Node->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK, "Node wasn't codeblock. It was '{}'",
|
||||
IR::GetName(Node->Op(DualListData.DataBegin())->Op));
|
||||
SetWriteCursor(Node->Op(DualListData.DataBegin())->CW<IROp_CodeBlock>()->Begin.GetNode(DualListData.ListBegin()));
|
||||
}
|
||||
|
||||
void IREmitter::ReplaceWithConstant(OrderedNode* Node, uint64_t Value) {
|
||||
void IREmitter::ReplaceWithConstant(Ref Node, uint64_t Value) {
|
||||
auto Header = Node->Op(DualListData.DataBegin());
|
||||
|
||||
if (IRSizes[Header->Op] >= sizeof(IROp_Constant)) {
|
||||
|
||||
@@ -41,10 +41,7 @@ public:
|
||||
}
|
||||
|
||||
IRListView ViewIR() {
|
||||
return IRListView(&DualListData, false);
|
||||
}
|
||||
IRListView* CreateIRCopy() {
|
||||
return new IRListView(&DualListData, true);
|
||||
return IRListView(&DualListData);
|
||||
}
|
||||
void ResetWorkingList();
|
||||
|
||||
@@ -53,7 +50,7 @@ public:
|
||||
*
|
||||
* @{ */
|
||||
|
||||
FEXCore::IR::RegisterClassType WalkFindRegClass(OrderedNode* Node);
|
||||
FEXCore::IR::RegisterClassType WalkFindRegClass(Ref Node);
|
||||
|
||||
// These handlers add cost to the constructor and destructor
|
||||
// If it becomes an issue then blow them away
|
||||
@@ -73,14 +70,14 @@ public:
|
||||
IRPair<IROp_Jump> _Jump() {
|
||||
return _Jump(InvalidNode);
|
||||
}
|
||||
IRPair<IROp_CondJump> _CondJump(OrderedNode* ssa0, CondClassType cond = {COND_NEQ}) {
|
||||
IRPair<IROp_CondJump> _CondJump(Ref ssa0, CondClassType cond = {COND_NEQ}) {
|
||||
return _CondJump(ssa0, _Constant(0), InvalidNode, InvalidNode, cond, GetOpSize(ssa0));
|
||||
}
|
||||
IRPair<IROp_CondJump> _CondJump(OrderedNode* ssa0, OrderedNode* ssa1, OrderedNode* ssa2, CondClassType cond = {COND_NEQ}) {
|
||||
IRPair<IROp_CondJump> _CondJump(Ref ssa0, Ref ssa1, Ref ssa2, CondClassType cond = {COND_NEQ}) {
|
||||
return _CondJump(ssa0, _Constant(0), ssa1, ssa2, cond, GetOpSize(ssa0));
|
||||
}
|
||||
// TODO: Work to remove this implicit sized Select implementation.
|
||||
IRPair<IROp_Select> _Select(uint8_t Cond, OrderedNode* ssa0, OrderedNode* ssa1, OrderedNode* ssa2, OrderedNode* ssa3, uint8_t CompareSize = 0) {
|
||||
IRPair<IROp_Select> _Select(uint8_t Cond, Ref ssa0, Ref ssa1, Ref ssa2, Ref ssa3, uint8_t CompareSize = 0) {
|
||||
if (CompareSize == 0) {
|
||||
CompareSize = std::max<uint8_t>(4, std::max<uint8_t>(GetOpSize(ssa0), GetOpSize(ssa1)));
|
||||
}
|
||||
@@ -88,54 +85,53 @@ public:
|
||||
return _Select(IR::SizeToOpSize(std::max<uint8_t>(4, std::max<uint8_t>(GetOpSize(ssa2), GetOpSize(ssa3)))),
|
||||
IR::SizeToOpSize(CompareSize), CondClassType {Cond}, ssa0, ssa1, ssa2, ssa3);
|
||||
}
|
||||
IRPair<IROp_LoadMem> _LoadMem(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode* ssa0, uint8_t Align = 1) {
|
||||
IRPair<IROp_LoadMem> _LoadMem(FEXCore::IR::RegisterClassType Class, uint8_t Size, Ref ssa0, uint8_t Align = 1) {
|
||||
return _LoadMem(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
IRPair<IROp_LoadMemTSO> _LoadMemTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode* ssa0, uint8_t Align = 1) {
|
||||
IRPair<IROp_LoadMemTSO> _LoadMemTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, Ref ssa0, uint8_t Align = 1) {
|
||||
return _LoadMemTSO(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
IRPair<IROp_StoreMem> _StoreMem(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode* Addr, OrderedNode* Value, uint8_t Align = 1) {
|
||||
IRPair<IROp_StoreMem> _StoreMem(FEXCore::IR::RegisterClassType Class, uint8_t Size, Ref Addr, Ref Value, uint8_t Align = 1) {
|
||||
return _StoreMem(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
IRPair<IROp_StoreMemTSO>
|
||||
_StoreMemTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode* Addr, OrderedNode* Value, uint8_t Align = 1) {
|
||||
IRPair<IROp_StoreMemTSO> _StoreMemTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, Ref Addr, Ref Value, uint8_t Align = 1) {
|
||||
return _StoreMemTSO(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
OrderedNode* Invalid() {
|
||||
Ref Invalid() {
|
||||
return InvalidNode;
|
||||
}
|
||||
|
||||
void SetJumpTarget(IR::IROp_Jump* Op, OrderedNode* Target) {
|
||||
void SetJumpTarget(IR::IROp_Jump* Op, Ref Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK, "Tried setting Jump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(), IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
|
||||
Op->Header.Args[0].NodeOffset = Target->Wrapped(DualListData.ListBegin()).NodeOffset;
|
||||
}
|
||||
void SetTrueJumpTarget(IR::IROp_CondJump* Op, OrderedNode* Target) {
|
||||
void SetTrueJumpTarget(IR::IROp_CondJump* Op, Ref Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK, "Tried setting CondJump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(), IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
|
||||
Op->TrueBlock.NodeOffset = Target->Wrapped(DualListData.ListBegin()).NodeOffset;
|
||||
}
|
||||
void SetFalseJumpTarget(IR::IROp_CondJump* Op, OrderedNode* Target) {
|
||||
void SetFalseJumpTarget(IR::IROp_CondJump* Op, Ref Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK, "Tried setting CondJump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(), IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
|
||||
Op->FalseBlock.NodeOffset = Target->Wrapped(DualListData.ListBegin()).NodeOffset;
|
||||
}
|
||||
|
||||
void SetJumpTarget(IRPair<IROp_Jump> Op, OrderedNode* Target) {
|
||||
void SetJumpTarget(IRPair<IROp_Jump> Op, Ref Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK, "Tried setting Jump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(), IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
|
||||
Op.first->Header.Args[0].NodeOffset = Target->Wrapped(DualListData.ListBegin()).NodeOffset;
|
||||
}
|
||||
void SetTrueJumpTarget(IRPair<IROp_CondJump> Op, OrderedNode* Target) {
|
||||
void SetTrueJumpTarget(IRPair<IROp_CondJump> Op, Ref Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK, "Tried setting CondJump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(), IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
Op.first->TrueBlock.NodeOffset = Target->Wrapped(DualListData.ListBegin()).NodeOffset;
|
||||
}
|
||||
void SetFalseJumpTarget(IRPair<IROp_CondJump> Op, OrderedNode* Target) {
|
||||
void SetFalseJumpTarget(IRPair<IROp_CondJump> Op, Ref Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK, "Tried setting CondJump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(), IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
Op.first->FalseBlock.NodeOffset = Target->Wrapped(DualListData.ListBegin()).NodeOffset;
|
||||
@@ -143,12 +139,12 @@ public:
|
||||
|
||||
/** @} */
|
||||
FEXCore::IR::RegisterClassType WalkFindRegClass(OrderedNodeWrapper ssa) {
|
||||
OrderedNode* RealNode = ssa.GetNode(DualListData.ListBegin());
|
||||
Ref RealNode = ssa.GetNode(DualListData.ListBegin());
|
||||
return WalkFindRegClass(RealNode);
|
||||
}
|
||||
|
||||
bool IsValueConstant(OrderedNodeWrapper ssa, uint64_t* Constant = nullptr) {
|
||||
OrderedNode* RealNode = ssa.GetNode(DualListData.ListBegin());
|
||||
Ref RealNode = ssa.GetNode(DualListData.ListBegin());
|
||||
FEXCore::IR::IROp_Header* IROp = RealNode->Op(DualListData.DataBegin());
|
||||
if (IROp->Op == OP_CONSTANT) {
|
||||
auto Op = IROp->C<IR::IROp_Constant>();
|
||||
@@ -161,7 +157,7 @@ public:
|
||||
}
|
||||
|
||||
bool IsValueInlineConstant(OrderedNodeWrapper ssa) {
|
||||
OrderedNode* RealNode = ssa.GetNode(DualListData.ListBegin());
|
||||
Ref RealNode = ssa.GetNode(DualListData.ListBegin());
|
||||
FEXCore::IR::IROp_Header* IROp = RealNode->Op(DualListData.DataBegin());
|
||||
if (IROp->Op == OP_INLINECONSTANT) {
|
||||
return true;
|
||||
@@ -170,15 +166,15 @@ public:
|
||||
}
|
||||
|
||||
FEXCore::IR::IROp_Header* GetOpHeader(OrderedNodeWrapper ssa) {
|
||||
OrderedNode* RealNode = ssa.GetNode(DualListData.ListBegin());
|
||||
Ref RealNode = ssa.GetNode(DualListData.ListBegin());
|
||||
return RealNode->Op(DualListData.DataBegin());
|
||||
}
|
||||
|
||||
OrderedNode* UnwrapNode(OrderedNodeWrapper ssa) {
|
||||
Ref UnwrapNode(OrderedNodeWrapper ssa) {
|
||||
return ssa.GetNode(DualListData.ListBegin());
|
||||
}
|
||||
|
||||
OrderedNodeWrapper WrapNode(OrderedNode* node) {
|
||||
OrderedNodeWrapper WrapNode(Ref node) {
|
||||
return node->Wrapped(DualListData.ListBegin());
|
||||
}
|
||||
|
||||
@@ -189,23 +185,23 @@ public:
|
||||
// Overwrite a node with a constant
|
||||
// Depending on what node has been overwritten, there might be some unallocated space around the node
|
||||
// Because we are overwriting the node, we don't have to worry about update all the arguments which use it
|
||||
void ReplaceWithConstant(OrderedNode* Node, uint64_t Value);
|
||||
void ReplaceWithConstant(Ref Node, uint64_t Value);
|
||||
|
||||
void ReplaceAllUsesWithRange(OrderedNode* Node, OrderedNode* NewNode, AllNodesIterator Begin, AllNodesIterator End);
|
||||
void ReplaceAllUsesWithRange(Ref Node, Ref NewNode, AllNodesIterator Begin, AllNodesIterator End);
|
||||
|
||||
void ReplaceUsesWithAfter(OrderedNode* Node, OrderedNode* NewNode, AllNodesIterator After) {
|
||||
void ReplaceUsesWithAfter(Ref Node, Ref NewNode, AllNodesIterator After) {
|
||||
++After;
|
||||
ReplaceAllUsesWithRange(Node, NewNode, After, AllNodesIterator(DualListData.ListBegin(), DualListData.DataBegin()));
|
||||
}
|
||||
|
||||
void ReplaceUsesWithAfter(OrderedNode* Node, OrderedNode* NewNode, OrderedNode* After) {
|
||||
void ReplaceUsesWithAfter(Ref Node, Ref NewNode, Ref After) {
|
||||
auto Wrapped = After->Wrapped(DualListData.ListBegin());
|
||||
AllNodesIterator It = AllNodesIterator(DualListData.ListBegin(), DualListData.DataBegin(), Wrapped);
|
||||
|
||||
ReplaceUsesWithAfter(Node, NewNode, It);
|
||||
}
|
||||
|
||||
void ReplaceAllUsesWith(OrderedNode* Node, OrderedNode* NewNode) {
|
||||
void ReplaceAllUsesWith(Ref Node, Ref NewNode) {
|
||||
auto Start = AllNodesIterator(DualListData.ListBegin(), DualListData.DataBegin(), Node->Wrapped(DualListData.ListBegin()));
|
||||
|
||||
ReplaceAllUsesWithRange(Node, NewNode, Start, AllNodesIterator(DualListData.ListBegin(), DualListData.DataBegin()));
|
||||
@@ -220,12 +216,12 @@ public:
|
||||
}
|
||||
}
|
||||
|
||||
void ReplaceNodeArgument(OrderedNode* Node, uint8_t Arg, OrderedNode* NewArg);
|
||||
void ReplaceNodeArgument(Ref Node, uint8_t Arg, Ref NewArg);
|
||||
|
||||
void Remove(OrderedNode* Node);
|
||||
void Remove(Ref Node);
|
||||
|
||||
void SetPackedRFLAG(bool Lower8, OrderedNode* Src);
|
||||
OrderedNode* GetPackedRFLAG(bool Lower8);
|
||||
void SetPackedRFLAG(bool Lower8, Ref Src);
|
||||
Ref GetPackedRFLAG(bool Lower8);
|
||||
|
||||
void CopyData(const IREmitter& rhs) {
|
||||
LOGMAN_THROW_A_FMT(rhs.DualListData.DataBackingSize() <= DualListData.DataBackingSize(), "Trying to take ownership of data that is too "
|
||||
@@ -241,12 +237,12 @@ public:
|
||||
}
|
||||
}
|
||||
|
||||
void SetWriteCursor(OrderedNode* Node) {
|
||||
void SetWriteCursor(Ref Node) {
|
||||
CurrentWriteCursor = Node;
|
||||
}
|
||||
|
||||
// Set cursor to write before Node
|
||||
void SetWriteCursorBefore(OrderedNode* Node) {
|
||||
void SetWriteCursorBefore(Ref Node) {
|
||||
auto IR = ViewIR();
|
||||
auto Before = IR.at(Node);
|
||||
--Before;
|
||||
@@ -254,11 +250,11 @@ public:
|
||||
SetWriteCursor(std::get<0>(*Before));
|
||||
}
|
||||
|
||||
OrderedNode* GetWriteCursor() {
|
||||
Ref GetWriteCursor() {
|
||||
return CurrentWriteCursor;
|
||||
}
|
||||
|
||||
OrderedNode* GetCurrentBlock() {
|
||||
Ref GetCurrentBlock() {
|
||||
return CurrentCodeBlock;
|
||||
}
|
||||
|
||||
@@ -300,7 +296,7 @@ public:
|
||||
*
|
||||
* @{ */
|
||||
/** @} */
|
||||
void LinkCodeBlocks(OrderedNode* CodeNode, OrderedNode* Next) {
|
||||
void LinkCodeBlocks(Ref CodeNode, Ref Next) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
FEXCore::IR::IROp_CodeBlock* CurrentIROp =
|
||||
#endif
|
||||
@@ -314,17 +310,17 @@ public:
|
||||
IRPair<IROp_CodeBlock> CreateNewCodeBlockAtEnd() {
|
||||
return CreateNewCodeBlockAfter(nullptr);
|
||||
}
|
||||
IRPair<IROp_CodeBlock> CreateNewCodeBlockAfter(OrderedNode* insertAfter);
|
||||
void SetCurrentCodeBlock(OrderedNode* Node);
|
||||
IRPair<IROp_CodeBlock> CreateNewCodeBlockAfter(Ref insertAfter);
|
||||
void SetCurrentCodeBlock(Ref Node);
|
||||
|
||||
protected:
|
||||
void RemoveArgUses(OrderedNode* Node);
|
||||
void RemoveArgUses(Ref Node);
|
||||
|
||||
OrderedNode* CreateNode(IROp_Header* Op) {
|
||||
Ref CreateNode(IROp_Header* Op) {
|
||||
uintptr_t ListBegin = DualListData.ListBegin();
|
||||
size_t Size = sizeof(OrderedNode);
|
||||
void* Ptr = DualListData.ListAllocate(Size);
|
||||
OrderedNode* Node = new (Ptr) OrderedNode();
|
||||
Ref Node = new (Ptr) OrderedNode();
|
||||
Node->Header.Value.SetOffset(DualListData.DataBegin(), reinterpret_cast<uintptr_t>(Op));
|
||||
|
||||
if (CurrentWriteCursor) {
|
||||
@@ -334,15 +330,15 @@ protected:
|
||||
return Node;
|
||||
}
|
||||
|
||||
OrderedNode* GetNode(uint32_t SSANode) {
|
||||
Ref GetNode(uint32_t SSANode) {
|
||||
uintptr_t ListBegin = DualListData.ListBegin();
|
||||
OrderedNode* Node = reinterpret_cast<OrderedNode*>(ListBegin + SSANode * sizeof(OrderedNode));
|
||||
Ref Node = reinterpret_cast<Ref>(ListBegin + SSANode * sizeof(OrderedNode));
|
||||
return Node;
|
||||
}
|
||||
|
||||
OrderedNode* EmplaceOrphanedNode(OrderedNode* OldNode) {
|
||||
Ref EmplaceOrphanedNode(Ref OldNode) {
|
||||
size_t Size = sizeof(OrderedNode);
|
||||
OrderedNode* Ptr = reinterpret_cast<OrderedNode*>(DualListData.ListAllocate(Size));
|
||||
Ref Ptr = reinterpret_cast<Ref>(DualListData.ListAllocate(Size));
|
||||
memcpy(Ptr, OldNode, Size);
|
||||
return Ptr;
|
||||
}
|
||||
@@ -351,14 +347,14 @@ protected:
|
||||
// Overriden by dispatcher, stubbed for IR tests
|
||||
}
|
||||
|
||||
OrderedNode* CurrentWriteCursor = nullptr;
|
||||
Ref CurrentWriteCursor = nullptr;
|
||||
|
||||
// These could be combined with a little bit of work to be more efficient with memory usage. Isn't a big deal
|
||||
DualIntrusiveAllocatorThreadPool DualListData;
|
||||
|
||||
OrderedNode* InvalidNode;
|
||||
OrderedNode* CurrentCodeBlock {};
|
||||
fextl::vector<OrderedNode*> CodeBlocks;
|
||||
Ref InvalidNode;
|
||||
Ref CurrentCodeBlock {};
|
||||
fextl::vector<Ref> CodeBlocks;
|
||||
uint64_t Entry;
|
||||
};
|
||||
|
||||
|
||||
@@ -1,724 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
meta: ir|parser ~ Text -> IR
|
||||
tags: ir|parser
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/StringUtils.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <errno.h>
|
||||
#include <memory>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string_view>
|
||||
#include <utility>
|
||||
#include <istream>
|
||||
#include <unordered_map>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
namespace {
|
||||
|
||||
enum class DecodeFailure {
|
||||
DECODE_OKAY,
|
||||
DECODE_UNKNOWN_TYPE,
|
||||
DECODE_INVALID,
|
||||
DECODE_INVALIDCHAR,
|
||||
DECODE_INVALIDRANGE,
|
||||
DECODE_INVALIDREGISTERCLASS,
|
||||
DECODE_UNKNOWN_SSA,
|
||||
DECODE_INVALID_CONDFLAG,
|
||||
DECODE_INVALID_MEMOFFSETTYPE,
|
||||
DECODE_INVALID_FENCETYPE,
|
||||
DECODE_INVALID_BREAKTYPE,
|
||||
DECODE_INVALID_OPSIZE,
|
||||
};
|
||||
|
||||
fextl::string DecodeErrorToString(DecodeFailure Failure) {
|
||||
switch (Failure) {
|
||||
case DecodeFailure::DECODE_OKAY: return "Okay";
|
||||
case DecodeFailure::DECODE_UNKNOWN_TYPE: return "Unknown Type";
|
||||
case DecodeFailure::DECODE_INVALID: return "Invalid";
|
||||
case DecodeFailure::DECODE_INVALIDCHAR: return "Invalid starting char";
|
||||
case DecodeFailure::DECODE_INVALIDRANGE: return "Invalid integer range";
|
||||
case DecodeFailure::DECODE_INVALIDREGISTERCLASS: return "Invalid register class";
|
||||
case DecodeFailure::DECODE_UNKNOWN_SSA: return "Unknown SSA value";
|
||||
case DecodeFailure::DECODE_INVALID_CONDFLAG: return "Invalid Conditional name";
|
||||
case DecodeFailure::DECODE_INVALID_MEMOFFSETTYPE: return "Invalid Memory Offset Type";
|
||||
case DecodeFailure::DECODE_INVALID_FENCETYPE: return "Invalid Fence Type";
|
||||
case DecodeFailure::DECODE_INVALID_BREAKTYPE: return "Invalid Break Reason Type";
|
||||
case DecodeFailure::DECODE_INVALID_OPSIZE: return "Invalid Operation size name";
|
||||
}
|
||||
return "Unknown Error";
|
||||
}
|
||||
|
||||
class IRParser : public FEXCore::IR::IREmitter {
|
||||
public:
|
||||
template<typename Type>
|
||||
std::pair<DecodeFailure, Type> DecodeValue(const fextl::string& Arg) {
|
||||
return {DecodeFailure::DECODE_UNKNOWN_TYPE, {}};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, uint8_t> DecodeValue(const fextl::string& Arg) {
|
||||
if (Arg.at(0) != '#') {
|
||||
return {DecodeFailure::DECODE_INVALIDCHAR, 0};
|
||||
}
|
||||
|
||||
uint8_t Result = strtoul(&Arg.at(1), nullptr, 0);
|
||||
if (errno == ERANGE) {
|
||||
return {DecodeFailure::DECODE_INVALIDRANGE, 0};
|
||||
}
|
||||
return {DecodeFailure::DECODE_OKAY, Result};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, bool> DecodeValue(const fextl::string& Arg) {
|
||||
if (Arg.at(0) != '#') {
|
||||
return {DecodeFailure::DECODE_INVALIDCHAR, 0};
|
||||
}
|
||||
|
||||
uint8_t Result = strtoul(&Arg.at(1), nullptr, 0);
|
||||
if (errno == ERANGE || Result > 1) {
|
||||
return {DecodeFailure::DECODE_INVALIDRANGE, 0};
|
||||
}
|
||||
return {DecodeFailure::DECODE_OKAY, Result != 0};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, uint16_t> DecodeValue(const fextl::string& Arg) {
|
||||
if (Arg.at(0) != '#') {
|
||||
return {DecodeFailure::DECODE_INVALIDCHAR, 0};
|
||||
}
|
||||
|
||||
uint16_t Result = strtoul(&Arg.at(1), nullptr, 0);
|
||||
if (errno == ERANGE) {
|
||||
return {DecodeFailure::DECODE_INVALIDRANGE, 0};
|
||||
}
|
||||
return {DecodeFailure::DECODE_OKAY, Result};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, uint32_t> DecodeValue(const fextl::string& Arg) {
|
||||
if (Arg.at(0) != '#') {
|
||||
return {DecodeFailure::DECODE_INVALIDCHAR, 0};
|
||||
}
|
||||
|
||||
uint32_t Result = strtoul(&Arg.at(1), nullptr, 0);
|
||||
if (errno == ERANGE) {
|
||||
return {DecodeFailure::DECODE_INVALIDRANGE, 0};
|
||||
}
|
||||
return {DecodeFailure::DECODE_OKAY, Result};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, uint64_t> DecodeValue(const fextl::string& Arg) {
|
||||
if (Arg.at(0) != '#') {
|
||||
return {DecodeFailure::DECODE_INVALIDCHAR, 0};
|
||||
}
|
||||
|
||||
uint64_t Result = strtoull(&Arg.at(1), nullptr, 0);
|
||||
if (errno == ERANGE) {
|
||||
return {DecodeFailure::DECODE_INVALIDRANGE, 0};
|
||||
}
|
||||
return {DecodeFailure::DECODE_OKAY, Result};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, int64_t> DecodeValue(const fextl::string& Arg) {
|
||||
if (Arg.at(0) != '#') {
|
||||
return {DecodeFailure::DECODE_INVALIDCHAR, 0};
|
||||
}
|
||||
|
||||
int64_t Result = (int64_t)strtoull(&Arg.at(1), nullptr, 0);
|
||||
if (errno == ERANGE) {
|
||||
return {DecodeFailure::DECODE_INVALIDRANGE, 0};
|
||||
}
|
||||
return {DecodeFailure::DECODE_OKAY, Result};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, IR::SHA256Sum> DecodeValue(const fextl::string& Arg) {
|
||||
IR::SHA256Sum Result;
|
||||
|
||||
if (Arg.at(0) != 's' || Arg.at(1) != 'h' || Arg.at(2) != 'a' || Arg.at(3) != '2' || Arg.at(4) != '5' || Arg.at(5) != '6' || Arg.at(6) != ':') {
|
||||
return {DecodeFailure::DECODE_INVALIDCHAR, Result};
|
||||
}
|
||||
|
||||
auto GetDigit = [](const fextl::string& Arg, int pos, uint8_t* val) {
|
||||
auto chr = Arg.at(pos);
|
||||
if (chr >= '0' && chr <= '9') {
|
||||
*val = chr - '0';
|
||||
return true;
|
||||
} else if (chr >= 'a' && chr <= 'f') {
|
||||
*val = 10 + chr - 'a';
|
||||
return true;
|
||||
} else {
|
||||
return false;
|
||||
}
|
||||
};
|
||||
|
||||
for (size_t i = 0; i < sizeof(Result.data); i++) {
|
||||
uint8_t high, low;
|
||||
if (!GetDigit(Arg, 7 + 2 * i + 0, &high) || !GetDigit(Arg, 7 + 2 * i + 1, &low)) {
|
||||
return {DecodeFailure::DECODE_INVALIDRANGE, Result};
|
||||
}
|
||||
Result.data[i] = high * 16 + low;
|
||||
}
|
||||
|
||||
return {DecodeFailure::DECODE_OKAY, Result};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, FEXCore::IR::RegisterClassType> DecodeValue(const fextl::string& Arg) {
|
||||
if (Arg == "GPR") {
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::GPRClass};
|
||||
} else if (Arg == "FPR") {
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::FPRClass};
|
||||
} else if (Arg == "GPRFixed") {
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::GPRFixedClass};
|
||||
} else if (Arg == "FPRFixed") {
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::FPRFixedClass};
|
||||
} else if (Arg == "GPRPair") {
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::GPRPairClass};
|
||||
} else if (Arg == "Complex") {
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::ComplexClass};
|
||||
}
|
||||
|
||||
return {DecodeFailure::DECODE_INVALIDREGISTERCLASS, FEXCore::IR::InvalidClass};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, FEXCore::IR::TypeDefinition> DecodeValue(const fextl::string& Arg) {
|
||||
uint8_t Size {}, Elements {1};
|
||||
int NumArgs = sscanf(Arg.c_str(), "i%hhdv%hhd", &Size, &Elements);
|
||||
|
||||
if (NumArgs != 1 && NumArgs != 2) {
|
||||
return {DecodeFailure::DECODE_INVALID, {}};
|
||||
}
|
||||
|
||||
return {DecodeFailure::DECODE_OKAY, FEXCore::IR::TypeDefinition::Create(Size / 8, Elements)};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, FEXCore::IR::CondClassType> DecodeValue(const fextl::string& Arg) {
|
||||
static constexpr std::array<std::string_view, 22> CondNames = {"EQ", "NEQ", "UGE", "ULT", "MI", "PL", "VS", "VC",
|
||||
"UGT", "ULE", "SGE", "SLT", "SGT", "SLE", "ANDZ", "ANDNZ",
|
||||
"FLU", "FGE", "FLEU", "FGT", "FU", "FNU"};
|
||||
|
||||
for (size_t i = 0; i < CondNames.size(); ++i) {
|
||||
if (CondNames[i] == Arg) {
|
||||
return {DecodeFailure::DECODE_OKAY, CondClassType {static_cast<uint8_t>(i)}};
|
||||
}
|
||||
}
|
||||
return {DecodeFailure::DECODE_INVALID_CONDFLAG, {}};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, FEXCore::IR::MemOffsetType> DecodeValue(const fextl::string& Arg) {
|
||||
static constexpr std::array<std::string_view, 3> Names = {
|
||||
"SXTX",
|
||||
"UXTW",
|
||||
"SXTW",
|
||||
};
|
||||
|
||||
for (size_t i = 0; i < Names.size(); ++i) {
|
||||
if (Names[i] == Arg) {
|
||||
return {DecodeFailure::DECODE_OKAY, MemOffsetType {static_cast<uint8_t>(i)}};
|
||||
}
|
||||
}
|
||||
return {DecodeFailure::DECODE_INVALID_MEMOFFSETTYPE, {}};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, FEXCore::IR::FenceType> DecodeValue(const fextl::string& Arg) {
|
||||
static constexpr std::array<std::string_view, 3> Names = {
|
||||
"Loads",
|
||||
"Stores",
|
||||
"LoadStores",
|
||||
};
|
||||
|
||||
for (size_t i = 0; i < Names.size(); ++i) {
|
||||
if (Names[i] == Arg) {
|
||||
return {DecodeFailure::DECODE_OKAY, FenceType {static_cast<uint8_t>(i)}};
|
||||
}
|
||||
}
|
||||
return {DecodeFailure::DECODE_INVALID_FENCETYPE, {}};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, FEXCore::IR::BreakDefinition> DecodeValue(const fextl::string& Arg) {
|
||||
uint32_t tmp {};
|
||||
fextl::stringstream ss {Arg};
|
||||
BreakDefinition Reason {};
|
||||
|
||||
// Seek past '{'
|
||||
ss.seekg(1, std::ios::cur);
|
||||
ss >> Reason.ErrorRegister;
|
||||
|
||||
// Seek past '.'
|
||||
ss.seekg(1, std::ios::cur);
|
||||
ss >> tmp;
|
||||
Reason.Signal = tmp;
|
||||
|
||||
// Seek past '.'
|
||||
ss.seekg(1, std::ios::cur);
|
||||
ss >> tmp;
|
||||
Reason.TrapNumber = tmp;
|
||||
|
||||
// Seek past '.'
|
||||
ss.seekg(1, std::ios::cur);
|
||||
ss >> tmp;
|
||||
Reason.si_code = tmp;
|
||||
|
||||
if (ss.fail()) {
|
||||
return {DecodeFailure::DECODE_INVALIDCHAR, {}};
|
||||
} else {
|
||||
return {DecodeFailure::DECODE_OKAY, Reason};
|
||||
}
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, FEXCore::IR::OpSize> DecodeValue(const fextl::string& Arg) {
|
||||
static constexpr std::array<std::pair<std::string_view, FEXCore::IR::OpSize>, 6> Names = {{
|
||||
{"i8", OpSize::i8Bit},
|
||||
{"i16", OpSize::i16Bit},
|
||||
{"i32", OpSize::i32Bit},
|
||||
{"i64", OpSize::i64Bit},
|
||||
{"i128", OpSize::i128Bit},
|
||||
{"i256", OpSize::i256Bit},
|
||||
}};
|
||||
|
||||
for (size_t i = 0; i < Names.size(); ++i) {
|
||||
if (Names[i].first == Arg) {
|
||||
return {DecodeFailure::DECODE_OKAY, Names[i].second};
|
||||
}
|
||||
}
|
||||
return {DecodeFailure::DECODE_INVALID_OPSIZE, {}};
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, OrderedNode*> DecodeValue(const fextl::string& Arg) {
|
||||
if (Arg.at(0) != '%') {
|
||||
return {DecodeFailure::DECODE_INVALIDCHAR, 0};
|
||||
}
|
||||
|
||||
// Strip off the type qualifier from the ssa value
|
||||
fextl::string SSAName = FEXCore::StringUtils::Trim(Arg);
|
||||
const size_t ArgEnd = SSAName.find_first_of(' ');
|
||||
|
||||
if (ArgEnd != fextl::string::npos) {
|
||||
SSAName = SSAName.substr(0, ArgEnd);
|
||||
}
|
||||
|
||||
// Forward declarations may make this not succed
|
||||
auto Op = SSANameMapper.find(SSAName);
|
||||
if (Op == SSANameMapper.end()) {
|
||||
return {DecodeFailure::DECODE_UNKNOWN_SSA, nullptr};
|
||||
}
|
||||
|
||||
return {DecodeFailure::DECODE_OKAY, Op->second};
|
||||
}
|
||||
|
||||
struct LineDefinition {
|
||||
size_t LineNumber;
|
||||
bool HasDefinition {};
|
||||
fextl::string Definition {};
|
||||
FEXCore::IR::TypeDefinition Size {};
|
||||
fextl::string IROp {};
|
||||
FEXCore::IR::IROps OpEnum;
|
||||
bool HasArgs {};
|
||||
fextl::vector<fextl::string> Args;
|
||||
OrderedNode* Node {};
|
||||
};
|
||||
|
||||
fextl::vector<fextl::string> Lines;
|
||||
fextl::unordered_map<fextl::string, OrderedNode*> SSANameMapper;
|
||||
fextl::vector<LineDefinition> Defs;
|
||||
LineDefinition* CurrentDef {};
|
||||
fextl::unordered_map<std::string_view, FEXCore::IR::IROps> NameToOpMap;
|
||||
|
||||
IRParser(FEXCore::Utils::IntrusivePooledAllocator& ThreadAllocator, fextl::stringstream& MapsStream)
|
||||
: IREmitter {ThreadAllocator} {
|
||||
InitializeNameMap();
|
||||
|
||||
fextl::string Line;
|
||||
while (std::getline(MapsStream, Line)) {
|
||||
if (MapsStream.eof()) {
|
||||
break;
|
||||
}
|
||||
if (MapsStream.fail()) {
|
||||
LogMan::Msg::EFmt("Failed to getline on line: {}", Lines.size());
|
||||
return;
|
||||
}
|
||||
Lines.emplace_back(Line);
|
||||
}
|
||||
|
||||
ResetWorkingList();
|
||||
Loaded = Parse();
|
||||
}
|
||||
|
||||
bool Loaded = false;
|
||||
|
||||
bool Parse() {
|
||||
const auto CheckPrintError = [&](const LineDefinition& Def, DecodeFailure Failure) -> bool {
|
||||
if (Failure != DecodeFailure::DECODE_OKAY) {
|
||||
LogMan::Msg::EFmt("Error on Line: {}", Def.LineNumber);
|
||||
LogMan::Msg::EFmt("{}", Lines[Def.LineNumber]);
|
||||
LogMan::Msg::EFmt("Value Couldn't be decoded due to {}", DecodeErrorToString(Failure));
|
||||
return false;
|
||||
}
|
||||
|
||||
return true;
|
||||
};
|
||||
|
||||
const auto CheckPrintErrorArg = [&](const LineDefinition& Def, DecodeFailure Failure, size_t Arg) -> bool {
|
||||
if (Failure != DecodeFailure::DECODE_OKAY) {
|
||||
LogMan::Msg::EFmt("Error on Line: {}", Def.LineNumber);
|
||||
LogMan::Msg::EFmt("{}", Lines[Def.LineNumber]);
|
||||
LogMan::Msg::EFmt("Argument Number {}: {}", Arg + 1, Def.Args[Arg]);
|
||||
LogMan::Msg::EFmt("Value Couldn't be decoded due to {}", DecodeErrorToString(Failure));
|
||||
return false;
|
||||
}
|
||||
|
||||
return true;
|
||||
};
|
||||
|
||||
// String parse every line for our definitions
|
||||
for (size_t i = 0; i < Lines.size(); ++i) {
|
||||
fextl::string Line = Lines[i];
|
||||
LineDefinition Def {};
|
||||
CurrentDef = &Def;
|
||||
Def.LineNumber = i;
|
||||
|
||||
Line = FEXCore::StringUtils::Trim(Line);
|
||||
|
||||
// Skip empty lines
|
||||
if (Line.empty()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if (Line[0] == ';') {
|
||||
// This is a comment line
|
||||
// Skip it
|
||||
continue;
|
||||
}
|
||||
|
||||
size_t CurrentPos {};
|
||||
// Let's see if this node is assigning something first
|
||||
if (Line[0] == '%') {
|
||||
size_t DefinitionEnd = fextl::string::npos;
|
||||
if ((DefinitionEnd = Line.find_first_of('=', CurrentPos)) != fextl::string::npos) {
|
||||
Def.Definition = Line.substr(0, DefinitionEnd);
|
||||
Def.Definition = FEXCore::StringUtils::Trim(Def.Definition);
|
||||
Def.HasDefinition = true;
|
||||
CurrentPos = DefinitionEnd + 1; // +1 to ensure we go past then assignment
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Error on Line: {}", i);
|
||||
LogMan::Msg::EFmt("{}", Lines[i]);
|
||||
LogMan::Msg::EFmt("SSA declaration without assignment");
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// Check if we are pulling in some IR from the IR Printer
|
||||
// Prints (%%d) at the start of lines without a definition
|
||||
if (Line[0] == '(') {
|
||||
size_t DefinitionEnd = fextl::string::npos;
|
||||
if ((DefinitionEnd = Line.find_first_of(')', CurrentPos)) != fextl::string::npos) {
|
||||
size_t SSAEnd = fextl::string::npos;
|
||||
if ((SSAEnd = Line.find_last_of(' ', DefinitionEnd)) != fextl::string::npos) {
|
||||
fextl::string Type = Line.substr(SSAEnd + 1, DefinitionEnd - SSAEnd - 1);
|
||||
Type = FEXCore::StringUtils::Trim(Type);
|
||||
|
||||
auto DefinitionSize = DecodeValue<FEXCore::IR::TypeDefinition>(Type);
|
||||
if (!CheckPrintError(Def, DefinitionSize.first)) {
|
||||
return false;
|
||||
}
|
||||
Def.Size = DefinitionSize.second;
|
||||
}
|
||||
|
||||
Def.Definition = FEXCore::StringUtils::Trim(Line.substr(1, std::min(DefinitionEnd, SSAEnd) - 1));
|
||||
|
||||
CurrentPos = DefinitionEnd + 1;
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Error on Line: {}", i);
|
||||
LogMan::Msg::EFmt("{}", Lines[i]);
|
||||
LogMan::Msg::EFmt("SSA value with numbered SSA provided but no closing parentheses");
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
if (Def.HasDefinition) {
|
||||
// Let's check if we have a size declared with this variable
|
||||
size_t NameEnd = fextl::string::npos;
|
||||
if ((NameEnd = Def.Definition.find_first_of(' ')) != fextl::string::npos) {
|
||||
fextl::string Type = Def.Definition.substr(NameEnd + 1);
|
||||
Type = FEXCore::StringUtils::Trim(Type);
|
||||
Def.Definition = FEXCore::StringUtils::Trim(Def.Definition.substr(0, NameEnd));
|
||||
|
||||
auto DefinitionSize = DecodeValue<FEXCore::IR::TypeDefinition>(Type);
|
||||
if (!CheckPrintError(Def, DefinitionSize.first)) {
|
||||
return false;
|
||||
}
|
||||
Def.Size = DefinitionSize.second;
|
||||
}
|
||||
|
||||
if (Def.Definition == "%Invalid") {
|
||||
LogMan::Msg::EFmt("Error on Line: {}", i);
|
||||
LogMan::Msg::EFmt("{}", Lines[i]);
|
||||
LogMan::Msg::EFmt("Definition tried to define reserved %Invalid ssa node");
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// Let's get the IR op
|
||||
size_t OpNameEnd = fextl::string::npos;
|
||||
fextl::string RemainingLine = FEXCore::StringUtils::Trim(Line.substr(CurrentPos));
|
||||
CurrentPos = 0;
|
||||
if ((OpNameEnd = RemainingLine.find_first_of(" \t\n\r\0", CurrentPos)) != fextl::string::npos) {
|
||||
Def.IROp = RemainingLine.substr(CurrentPos, OpNameEnd);
|
||||
Def.IROp = FEXCore::StringUtils::Trim(Def.IROp);
|
||||
Def.HasArgs = true;
|
||||
CurrentPos = OpNameEnd;
|
||||
} else {
|
||||
if (RemainingLine.empty()) {
|
||||
LogMan::Msg::EFmt("Error on Line: {}", i);
|
||||
LogMan::Msg::EFmt("{}", Lines[i]);
|
||||
LogMan::Msg::EFmt("Line without an IROp?");
|
||||
return false;
|
||||
}
|
||||
|
||||
Def.IROp = RemainingLine;
|
||||
Def.HasArgs = false;
|
||||
}
|
||||
|
||||
if (Def.HasArgs) {
|
||||
RemainingLine = FEXCore::StringUtils::Trim(RemainingLine.substr(CurrentPos));
|
||||
if (RemainingLine.empty()) {
|
||||
// How did we get here?
|
||||
Def.HasArgs = false;
|
||||
} else {
|
||||
while (!RemainingLine.empty()) {
|
||||
const size_t ArgEnd = RemainingLine.find(',');
|
||||
fextl::string Arg = FEXCore::StringUtils::Trim(RemainingLine.substr(0, ArgEnd));
|
||||
|
||||
Def.Args.emplace_back(std::move(Arg));
|
||||
|
||||
RemainingLine.erase(0, ArgEnd + 1); // +1 to ensure we go past the ','
|
||||
if (ArgEnd == fextl::string::npos) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
CurrentDef = &Defs.emplace_back(std::move(Def));
|
||||
}
|
||||
|
||||
// Ensure all of the ops are real ops
|
||||
for (size_t i = 0; i < Defs.size(); ++i) {
|
||||
auto& Def = Defs[i];
|
||||
auto Op = NameToOpMap.find(Def.IROp);
|
||||
if (Op == NameToOpMap.end()) {
|
||||
LogMan::Msg::EFmt("Error on Line: {}", Def.LineNumber);
|
||||
LogMan::Msg::EFmt("{}", Lines[Def.LineNumber]);
|
||||
LogMan::Msg::EFmt("IROp '{}' doesn't exist", Def.IROp);
|
||||
return false;
|
||||
}
|
||||
Def.OpEnum = Op->second;
|
||||
}
|
||||
|
||||
// Emit the header op
|
||||
IRPair<IROp_IRHeader> IRHeader;
|
||||
{
|
||||
auto& Def = Defs[0];
|
||||
CurrentDef = &Def;
|
||||
if (Def.OpEnum != FEXCore::IR::IROps::OP_IRHEADER) {
|
||||
LogMan::Msg::EFmt("Error on Line: {}", Def.LineNumber);
|
||||
LogMan::Msg::EFmt("{}", Lines[Def.LineNumber]);
|
||||
LogMan::Msg::EFmt("First op needs to be IRHeader. Was '{}'", Def.IROp);
|
||||
return false;
|
||||
}
|
||||
|
||||
auto OriginalRIP = DecodeValue<uint64_t>(Def.Args[1]);
|
||||
|
||||
if (!CheckPrintError(Def, OriginalRIP.first)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
auto CodeBlockCount = DecodeValue<uint64_t>(Def.Args[2]);
|
||||
|
||||
if (!CheckPrintError(Def, CodeBlockCount.first)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
auto InstructionCount = DecodeValue<uint64_t>(Def.Args[3]);
|
||||
|
||||
if (!CheckPrintError(Def, InstructionCount.first)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
IRHeader = _IRHeader(InvalidNode, OriginalRIP.second, CodeBlockCount.second, InstructionCount.second);
|
||||
}
|
||||
|
||||
SetWriteCursor(nullptr); // isolate the header from everything following
|
||||
|
||||
// Initialize SSANameMapper with Invalid value
|
||||
SSANameMapper.insert_or_assign("%Invalid", Invalid());
|
||||
|
||||
// Spin through the blocks and generate basic block ops
|
||||
for (size_t i = 0; i < Defs.size(); ++i) {
|
||||
auto& Def = Defs[i];
|
||||
if (Def.OpEnum == FEXCore::IR::IROps::OP_CODEBLOCK) {
|
||||
auto CodeBlock = _CodeBlock(InvalidNode, InvalidNode);
|
||||
SSANameMapper.insert_or_assign(Def.Definition, CodeBlock.Node);
|
||||
Def.Node = CodeBlock.Node;
|
||||
|
||||
if (i == 1) {
|
||||
// First code block is the entry block
|
||||
// Link the header to the first block
|
||||
IRHeader.first->Blocks = CodeBlock.Node->Wrapped(DualListData.ListBegin());
|
||||
}
|
||||
CodeBlocks.emplace_back(CodeBlock.Node);
|
||||
}
|
||||
}
|
||||
SetWriteCursor(nullptr); // isolate the block headers too
|
||||
|
||||
// Spin through all the definitions and add the ops to the basic blocks
|
||||
OrderedNode* CurrentBlock {};
|
||||
FEXCore::IR::IROp_CodeBlock* CurrentBlockOp {};
|
||||
for (size_t i = 1; i < Defs.size(); ++i) {
|
||||
auto& Def = Defs[i];
|
||||
CurrentDef = &Def;
|
||||
|
||||
switch (Def.OpEnum) {
|
||||
// Special handled
|
||||
case FEXCore::IR::IROps::OP_IRHEADER:
|
||||
LogMan::Msg::EFmt("Error on Line: {}", Def.LineNumber);
|
||||
LogMan::Msg::EFmt("{}", Lines[Def.LineNumber]);
|
||||
LogMan::Msg::EFmt("IRHEADER used in the middle of the block!");
|
||||
return false; // only one OP_IRHEADER allowed per block
|
||||
|
||||
case FEXCore::IR::IROps::OP_CODEBLOCK: {
|
||||
SetWriteCursor(nullptr); // isolate from previous block
|
||||
if (CurrentBlock != nullptr) {
|
||||
LogMan::Msg::EFmt("Error on Line: {}", Def.LineNumber);
|
||||
LogMan::Msg::EFmt("{}", Lines[Def.LineNumber]);
|
||||
LogMan::Msg::EFmt("CodeBlock being used inside of already existing codeblock!");
|
||||
return false;
|
||||
}
|
||||
|
||||
CurrentBlock = Def.Node;
|
||||
CurrentBlockOp = CurrentBlock->Op(DualListData.DataBegin())->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
break;
|
||||
}
|
||||
|
||||
|
||||
case FEXCore::IR::IROps::OP_BEGINBLOCK: {
|
||||
if (CurrentBlock == nullptr) {
|
||||
LogMan::Msg::EFmt("Error on Line: {}", Def.LineNumber);
|
||||
LogMan::Msg::EFmt("{}", Lines[Def.LineNumber]);
|
||||
LogMan::Msg::EFmt("EndBlock being used outside of a block!");
|
||||
return false;
|
||||
}
|
||||
|
||||
auto Adjust = DecodeValue<OrderedNode*>(Def.Args[0]);
|
||||
if (!CheckPrintError(Def, Adjust.first)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
Def.Node = _BeginBlock(Adjust.second);
|
||||
CurrentBlockOp->Begin = Def.Node->Wrapped(DualListData.ListBegin());
|
||||
break;
|
||||
}
|
||||
|
||||
case FEXCore::IR::IROps::OP_ENDBLOCK: {
|
||||
if (CurrentBlock == nullptr) {
|
||||
LogMan::Msg::EFmt("Error on Line: {}", Def.LineNumber);
|
||||
LogMan::Msg::EFmt("{}", Lines[Def.LineNumber]);
|
||||
LogMan::Msg::EFmt("EndBlock being used outside of a block!");
|
||||
return false;
|
||||
}
|
||||
|
||||
auto Adjust = DecodeValue<OrderedNode*>(Def.Args[0]);
|
||||
if (!CheckPrintError(Def, Adjust.first)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
Def.Node = _EndBlock(Adjust.second);
|
||||
CurrentBlockOp->Last = Def.Node->Wrapped(DualListData.ListBegin());
|
||||
|
||||
CurrentBlock = nullptr;
|
||||
CurrentBlockOp = nullptr;
|
||||
|
||||
break;
|
||||
}
|
||||
|
||||
case FEXCore::IR::IROps::OP_DUMMY: {
|
||||
LogMan::Msg::EFmt("Error on Line: {}", Def.LineNumber);
|
||||
LogMan::Msg::EFmt("{}", Lines[Def.LineNumber]);
|
||||
LogMan::Msg::EFmt("Dummy op must not be used");
|
||||
break;
|
||||
}
|
||||
#define IROP_PARSER_SWITCH_HELPERS
|
||||
#include <FEXCore/IR/IRDefines.inc>
|
||||
default: {
|
||||
LogMan::Msg::EFmt("Error on Line: {}", Def.LineNumber);
|
||||
LogMan::Msg::EFmt("{}", Lines[Def.LineNumber]);
|
||||
LogMan::Msg::EFmt("Unhandled Op enum '{}' in parser", Def.IROp);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
if (Def.HasDefinition) {
|
||||
auto IROp = Def.Node->Op(DualListData.DataBegin());
|
||||
if (Def.Size.Elements()) {
|
||||
IROp->Size = Def.Size.Bytes() * Def.Size.Elements();
|
||||
IROp->ElementSize = Def.Size.Bytes();
|
||||
} else {
|
||||
IROp->Size = Def.Size.Bytes();
|
||||
IROp->ElementSize = 0;
|
||||
}
|
||||
SSANameMapper.insert_or_assign(Def.Definition, Def.Node);
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
void InitializeNameMap() {
|
||||
if (NameToOpMap.empty()) {
|
||||
for (FEXCore::IR::IROps Op = FEXCore::IR::IROps::OP_DUMMY; Op <= FEXCore::IR::IROps::OP_LAST;
|
||||
Op = static_cast<FEXCore::IR::IROps>(static_cast<uint32_t>(Op) + 1)) {
|
||||
NameToOpMap.insert_or_assign(FEXCore::IR::GetName(Op), Op);
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
} // namespace
|
||||
|
||||
fextl::unique_ptr<IREmitter> Parse(FEXCore::Utils::IntrusivePooledAllocator& ThreadAllocator, fextl::stringstream& MapsStream) {
|
||||
auto parser = fextl::make_unique<IRParser>(ThreadAllocator, MapsStream);
|
||||
|
||||
if (parser->Loaded) {
|
||||
return parser;
|
||||
} else {
|
||||
return nullptr;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -144,54 +144,22 @@ private:
|
||||
Utils::FixedSizePooledAllocation<uintptr_t, 5000, 500> PoolObject;
|
||||
};
|
||||
|
||||
class IRListView final : public FEXCore::Allocator::FEXAllocOperators {
|
||||
enum Flags {
|
||||
FLAG_IsCopy = 1,
|
||||
FLAG_Shared = 2,
|
||||
};
|
||||
|
||||
class IRListView final {
|
||||
public:
|
||||
IRListView() = delete;
|
||||
IRListView(IRListView&&) = delete;
|
||||
|
||||
IRListView(DualIntrusiveAllocator* Data, bool _IsCopy) {
|
||||
SetCopy(_IsCopy);
|
||||
DataSize = Data->DataSize();
|
||||
ListSize = Data->ListSize();
|
||||
IRListView(DualIntrusiveAllocator* Data)
|
||||
: IRListView(reinterpret_cast<void*>(Data->DataBegin()), reinterpret_cast<void*>(Data->ListBegin()), Data->DataSize(), Data->ListSize()) {}
|
||||
|
||||
if (_IsCopy) {
|
||||
IRDataInternal = FEXCore::Allocator::malloc(DataSize + ListSize);
|
||||
ListDataInternal = reinterpret_cast<void*>(reinterpret_cast<uintptr_t>(IRDataInternal) + DataSize);
|
||||
memcpy(IRDataInternal, reinterpret_cast<void*>(Data->DataBegin()), DataSize);
|
||||
memcpy(ListDataInternal, reinterpret_cast<void*>(Data->ListBegin()), ListSize);
|
||||
} else {
|
||||
// We are just pointing to the data
|
||||
IRDataInternal = reinterpret_cast<void*>(Data->DataBegin());
|
||||
ListDataInternal = reinterpret_cast<void*>(Data->ListBegin());
|
||||
}
|
||||
}
|
||||
IRListView(IRListView* Old)
|
||||
: IRListView(Old->IRDataInternal, Old->ListDataInternal, Old->DataSize, Old->ListSize) {}
|
||||
|
||||
IRListView(IRListView* Old, bool _IsCopy) {
|
||||
SetCopy(_IsCopy);
|
||||
DataSize = Old->DataSize;
|
||||
ListSize = Old->ListSize;
|
||||
if (_IsCopy) {
|
||||
IRDataInternal = FEXCore::Allocator::malloc(DataSize + ListSize);
|
||||
ListDataInternal = reinterpret_cast<void*>(reinterpret_cast<uintptr_t>(IRDataInternal) + DataSize);
|
||||
memcpy(IRDataInternal, Old->IRDataInternal, DataSize);
|
||||
memcpy(ListDataInternal, Old->ListDataInternal, ListSize);
|
||||
} else {
|
||||
IRDataInternal = Old->IRDataInternal;
|
||||
ListDataInternal = Old->ListDataInternal;
|
||||
}
|
||||
}
|
||||
|
||||
~IRListView() {
|
||||
if (IsCopy()) {
|
||||
FEXCore::Allocator::free(IRDataInternal);
|
||||
// ListData is just offset from IRData
|
||||
}
|
||||
}
|
||||
IRListView(void* IRData_, void* ListData_, size_t DataSize_, size_t ListSize_)
|
||||
: IRDataInternal(IRData_)
|
||||
, ListDataInternal(ListData_)
|
||||
, DataSize(DataSize_)
|
||||
, ListSize(ListSize_) {}
|
||||
|
||||
void Serialize(FEXCore::Context::AOTIRWriter& stream) const {
|
||||
void* nul = nullptr;
|
||||
@@ -203,9 +171,6 @@ public:
|
||||
stream.Write((const char*)&DataSize, sizeof(DataSize));
|
||||
// size_t ListSize;
|
||||
stream.Write((const char*)&ListSize, sizeof(ListSize));
|
||||
// uint64_t Flags;
|
||||
uint64_t WrittenFlags = FLAG_Shared; // on disk format always has the Shared flag
|
||||
stream.Write((const char*)&WrittenFlags, sizeof(WrittenFlags));
|
||||
|
||||
// inline data
|
||||
stream.Write((const char*)GetData(), DataSize);
|
||||
@@ -226,10 +191,6 @@ public:
|
||||
// size_t ListSize;
|
||||
memcpy(ptr, &ListSize, sizeof(ListSize));
|
||||
ptr += sizeof(ListSize);
|
||||
// uint64_t Flags;
|
||||
uint64_t WrittenFlags = FLAG_Shared; // on disk format always has the Shared flag
|
||||
memcpy(ptr, &WrittenFlags, sizeof(WrittenFlags));
|
||||
ptr += sizeof(WrittenFlags);
|
||||
|
||||
// inline data
|
||||
memcpy(ptr, (const void*)GetData(), DataSize);
|
||||
@@ -240,15 +201,10 @@ public:
|
||||
|
||||
[[nodiscard]]
|
||||
size_t GetInlineSize() const {
|
||||
static_assert(sizeof(*this) == 40);
|
||||
static_assert(sizeof(*this) == 32);
|
||||
return sizeof(*this) + DataSize + ListSize;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
IRListView* CreateCopy() {
|
||||
return new IRListView(this, true);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
size_t GetDataSize() const {
|
||||
return DataSize;
|
||||
@@ -263,36 +219,12 @@ public:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsCopy() const {
|
||||
return (Flags & FLAG_IsCopy) != 0;
|
||||
}
|
||||
void SetCopy(bool Set) {
|
||||
if (Set) {
|
||||
Flags |= FLAG_IsCopy;
|
||||
} else {
|
||||
Flags &= ~FLAG_IsCopy;
|
||||
}
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsShared() const {
|
||||
return (Flags & FLAG_Shared) != 0;
|
||||
}
|
||||
void SetShared(bool Set) {
|
||||
if (Set) {
|
||||
Flags |= FLAG_Shared;
|
||||
} else {
|
||||
Flags &= ~FLAG_Shared;
|
||||
}
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
NodeID GetID(const OrderedNode* Node) const {
|
||||
NodeID GetID(const Ref Node) const {
|
||||
return Node->Wrapped(GetListData()).ID();
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
OrderedNode* GetHeaderNode() const {
|
||||
Ref GetHeaderNode() const {
|
||||
OrderedNodeWrapper Wrapped;
|
||||
Wrapped.NodeOffset = sizeof(OrderedNode);
|
||||
return Wrapped.GetNode(GetListData());
|
||||
@@ -305,7 +237,7 @@ public:
|
||||
|
||||
template<typename T>
|
||||
[[nodiscard]]
|
||||
T* GetOp(OrderedNode* Node) const {
|
||||
T* GetOp(Ref Node) const {
|
||||
auto OpHeader = Node->Op(GetData());
|
||||
auto Op = OpHeader->template CW<T>();
|
||||
|
||||
@@ -326,13 +258,13 @@ public:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
OrderedNode* GetNode(OrderedNodeWrapper Wrapper) const {
|
||||
Ref GetNode(OrderedNodeWrapper Wrapper) const {
|
||||
return Wrapper.GetNode(GetListData());
|
||||
}
|
||||
|
||||
///< Gets an OrderedNode from the IRListView as an OrderedNodeWrapper.
|
||||
[[nodiscard]]
|
||||
OrderedNodeWrapper WrapNode(OrderedNode* Node) const {
|
||||
OrderedNodeWrapper WrapNode(Ref Node) const {
|
||||
return Node->Wrapped(GetListData());
|
||||
}
|
||||
|
||||
@@ -405,7 +337,7 @@ public:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
CodeRange GetCode(const OrderedNode* block) const {
|
||||
CodeRange GetCode(const Ref block) const {
|
||||
return CodeRange(this, block->Wrapped(GetListData()));
|
||||
}
|
||||
|
||||
@@ -450,7 +382,7 @@ public:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
iterator at(const OrderedNode* Node) const noexcept {
|
||||
iterator at(const Ref Node) const noexcept {
|
||||
const auto ListData = GetListData();
|
||||
auto Wrapped = Node->Wrapped(ListData);
|
||||
return iterator(ListData, GetData(), Wrapped);
|
||||
@@ -471,15 +403,17 @@ private:
|
||||
void* ListDataInternal;
|
||||
size_t DataSize;
|
||||
size_t ListSize;
|
||||
uint64_t Flags {0};
|
||||
uint8_t InlineData[0];
|
||||
};
|
||||
|
||||
struct IRListViewDeleter {
|
||||
void operator()(IRListView* r) {
|
||||
if (!r->IsShared()) {
|
||||
delete r;
|
||||
}
|
||||
}
|
||||
class IRStorageBase {
|
||||
public:
|
||||
virtual ~IRStorageBase() = default;
|
||||
|
||||
// Optional RA data. Returns nullptr if none present
|
||||
virtual const RegisterAllocationData* RAData() = 0;
|
||||
|
||||
virtual IRListView GetIRView() = 0;
|
||||
};
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -66,59 +66,39 @@ void PassManager::Finalize() {
|
||||
}
|
||||
}
|
||||
|
||||
void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl* ctx, bool InlineConstants) {
|
||||
void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl* ctx) {
|
||||
FEX_CONFIG_OPT(DisablePasses, O0);
|
||||
|
||||
if (!DisablePasses()) {
|
||||
InsertPass(CreateContextLoadStoreElimination(ctx->HostFeatures.SupportsAVX));
|
||||
|
||||
if (Is64BitMode()) {
|
||||
// This needs to run after RCLSE
|
||||
// This only matters for 64-bit code since these instructions don't exist in 32-bit
|
||||
InsertPass(CreateLongDivideEliminationPass());
|
||||
}
|
||||
|
||||
InsertPass(CreateDeadStoreElimination(ctx->HostFeatures.SupportsAVX));
|
||||
InsertPass(CreatePassDeadCodeElimination());
|
||||
InsertPass(CreateConstProp(InlineConstants, ctx->HostFeatures.SupportsTSOImm9, Is64BitMode()));
|
||||
|
||||
InsertPass(CreateDeadStoreElimination());
|
||||
InsertPass(CreateConstProp(ctx->HostFeatures.SupportsTSOImm9, &ctx->CPUID));
|
||||
InsertPass(CreateDeadFlagCalculationEliminination());
|
||||
|
||||
InsertPass(CreateInlineCallOptimization(&ctx->CPUID));
|
||||
InsertPass(CreatePassDeadCodeElimination());
|
||||
}
|
||||
|
||||
// If the IR is compacted post-RA then the node indexing gets messed up and the backend isn't able to find the register assigned to a node
|
||||
// Compact before IR, don't worry about RA generating spills/fills
|
||||
InsertPass(CreateIRCompaction(ctx->OpDispatcherAllocator), "Compaction");
|
||||
}
|
||||
|
||||
void PassManager::AddDefaultValidationPasses() {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
InsertValidationPass(Validation::CreateIRValidation(), "IRValidation");
|
||||
InsertValidationPass(Validation::CreateRAValidation());
|
||||
InsertValidationPass(Validation::CreateValueDominanceValidation());
|
||||
#endif
|
||||
}
|
||||
|
||||
void PassManager::InsertRegisterAllocationPass(bool SupportsAVX) {
|
||||
InsertPass(IR::CreateRegisterAllocationPass(GetPass("Compaction"), SupportsAVX), "RA");
|
||||
void PassManager::InsertRegisterAllocationPass() {
|
||||
InsertPass(IR::CreateRegisterAllocationPass(), "RA");
|
||||
}
|
||||
|
||||
bool PassManager::Run(IREmitter* IREmit) {
|
||||
void PassManager::Run(IREmitter* IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::Run");
|
||||
|
||||
bool Changed = false;
|
||||
for (const auto& Pass : Passes) {
|
||||
Changed |= Pass->Run(IREmit);
|
||||
Pass->Run(IREmit);
|
||||
}
|
||||
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
for (const auto& Pass : ValidationPasses) {
|
||||
Changed |= Pass->Run(IREmit);
|
||||
Pass->Run(IREmit);
|
||||
}
|
||||
#endif
|
||||
|
||||
return Changed;
|
||||
}
|
||||
} // namespace FEXCore::IR
|
||||
@@ -32,7 +32,7 @@ class IREmitter;
|
||||
class Pass {
|
||||
public:
|
||||
virtual ~Pass() = default;
|
||||
virtual bool Run(IREmitter* IREmit) = 0;
|
||||
virtual void Run(IREmitter* IREmit) = 0;
|
||||
|
||||
void RegisterPassManager(PassManager* _Manager) {
|
||||
Manager = _Manager;
|
||||
@@ -43,9 +43,9 @@ protected:
|
||||
};
|
||||
|
||||
class PassManager final {
|
||||
friend class InlineCallOptimization;
|
||||
friend class ConstProp;
|
||||
public:
|
||||
void AddDefaultPasses(FEXCore::Context::ContextImpl* ctx, bool InlineConstants);
|
||||
void AddDefaultPasses(FEXCore::Context::ContextImpl* ctx);
|
||||
void AddDefaultValidationPasses();
|
||||
Pass* InsertPass(fextl::unique_ptr<Pass> Pass, fextl::string Name = "") {
|
||||
auto PassPtr = InsertAt(Passes.end(), std::move(Pass))->get();
|
||||
@@ -56,9 +56,9 @@ public:
|
||||
return PassPtr;
|
||||
}
|
||||
|
||||
void InsertRegisterAllocationPass(bool SupportsAVX);
|
||||
void InsertRegisterAllocationPass();
|
||||
|
||||
bool Run(IREmitter* IREmit);
|
||||
void Run(IREmitter* IREmit);
|
||||
|
||||
bool HasPass(fextl::string Name) const {
|
||||
return NameToPassMaping.contains(Name);
|
||||
|
||||
@@ -16,20 +16,15 @@ class Pass;
|
||||
class RegisterAllocationPass;
|
||||
class RegisterAllocationData;
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool InlineConstants, bool SupportsTSOImm9, bool Is64BitMode);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool SupportsTSOImm9, const FEXCore::CPUIDEmu* CPUID);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateContextLoadStoreElimination(bool SupportsAVX);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateInlineCallOptimization(const FEXCore::CPUIDEmu* CPUID);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadFlagCalculationEliminination();
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadStoreElimination(bool SupportsAVX);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreatePassDeadCodeElimination();
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateIRCompaction(FEXCore::Utils::IntrusivePooledAllocator& Allocator);
|
||||
fextl::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass(FEXCore::IR::Pass* CompactionPass, bool SupportsAVX);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateLongDivideEliminationPass();
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadStoreElimination();
|
||||
fextl::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass();
|
||||
|
||||
namespace Validation {
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateIRValidation();
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateRAValidation();
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateValueDominanceValidation();
|
||||
} // namespace Validation
|
||||
|
||||
namespace Debug {
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -1,113 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: ir|opts
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <memory>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
class DeadCodeElimination final : public FEXCore::IR::Pass {
|
||||
bool Run(IREmitter* IREmit) override;
|
||||
|
||||
private:
|
||||
void markUsed(OrderedNodeWrapper* CodeOp, IROp_Header* IROp);
|
||||
};
|
||||
|
||||
bool DeadCodeElimination::Run(IREmitter* IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::DCE");
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
bool Changed = false;
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
|
||||
// Reverse iteration is not yet working with the iterators
|
||||
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
|
||||
// We grab these nodes this way so we can iterate easily
|
||||
auto CodeBegin = CurrentIR.at(BlockIROp->Begin);
|
||||
auto CodeLast = CurrentIR.at(BlockIROp->Last);
|
||||
|
||||
while (1) {
|
||||
auto [CodeNode, IROp] = CodeLast();
|
||||
|
||||
bool HasSideEffects = IR::HasSideEffects(IROp->Op);
|
||||
|
||||
switch (IROp->Op) {
|
||||
case OP_SYSCALL:
|
||||
case OP_INLINESYSCALL: {
|
||||
FEXCore::IR::SyscallFlags Flags {};
|
||||
if (IROp->Op == OP_SYSCALL) {
|
||||
auto Op = IROp->C<IR::IROp_Syscall>();
|
||||
Flags = Op->Flags;
|
||||
} else {
|
||||
auto Op = IROp->C<IR::IROp_InlineSyscall>();
|
||||
Flags = Op->Flags;
|
||||
}
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NOSIDEEFFECTS) == FEXCore::IR::SyscallFlags::NOSIDEEFFECTS) {
|
||||
HasSideEffects = false;
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_ATOMICFETCHADD:
|
||||
case OP_ATOMICFETCHSUB:
|
||||
case OP_ATOMICFETCHAND:
|
||||
case OP_ATOMICFETCHCLR:
|
||||
case OP_ATOMICFETCHOR:
|
||||
case OP_ATOMICFETCHXOR:
|
||||
case OP_ATOMICFETCHNEG: {
|
||||
// If the result of the atomic fetch is completely unused, convert it to a non-fetching atomic operation.
|
||||
if (CodeNode->GetUses() == 0) {
|
||||
switch (IROp->Op) {
|
||||
case OP_ATOMICFETCHADD: IROp->Op = OP_ATOMICADD; break;
|
||||
case OP_ATOMICFETCHSUB: IROp->Op = OP_ATOMICSUB; break;
|
||||
case OP_ATOMICFETCHAND: IROp->Op = OP_ATOMICAND; break;
|
||||
case OP_ATOMICFETCHCLR: IROp->Op = OP_ATOMICCLR; break;
|
||||
case OP_ATOMICFETCHOR: IROp->Op = OP_ATOMICOR; break;
|
||||
case OP_ATOMICFETCHXOR: IROp->Op = OP_ATOMICXOR; break;
|
||||
case OP_ATOMICFETCHNEG: IROp->Op = OP_ATOMICNEG; break;
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: break;
|
||||
}
|
||||
|
||||
// Skip over anything that has side effects
|
||||
// Use count tracking can't safely remove anything with side effects
|
||||
if (!HasSideEffects) {
|
||||
if (CodeNode->GetUses() == 0) {
|
||||
Changed = true;
|
||||
IREmit->Remove(CodeNode);
|
||||
}
|
||||
}
|
||||
|
||||
if (CodeLast == CodeBegin) {
|
||||
break;
|
||||
}
|
||||
--CodeLast;
|
||||
}
|
||||
}
|
||||
|
||||
return Changed;
|
||||
}
|
||||
|
||||
void DeadCodeElimination::markUsed(OrderedNodeWrapper* CodeOp, IROp_Header* IROp) {}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreatePassDeadCodeElimination() {
|
||||
return fextl::make_unique<DeadCodeElimination>();
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -75,9 +75,9 @@ struct ContextMemberInfo {
|
||||
uint32_t AccessOffset;
|
||||
uint8_t AccessSize;
|
||||
///< The last value that was loaded or stored.
|
||||
FEXCore::IR::OrderedNode* ValueNode;
|
||||
FEXCore::IR::Ref ValueNode;
|
||||
///< With a store access, the store node that is doing the operation.
|
||||
FEXCore::IR::OrderedNode* StoreNode;
|
||||
FEXCore::IR::Ref StoreNode;
|
||||
};
|
||||
|
||||
struct ContextInfo {
|
||||
@@ -357,17 +357,6 @@ static void ClassifyContextStruct(ContextInfo* ContextClassificationInfo, bool S
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
// DeferredSignalFaultAddress
|
||||
ContextClassification->emplace_back(ContextMemberInfo {
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress),
|
||||
sizeof(FEXCore::Core::CPUState::DeferredSignalFaultAddress),
|
||||
},
|
||||
LastAccessType::NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
|
||||
[[maybe_unused]] size_t ClassifiedStructSize {};
|
||||
ContextClassificationInfo->Lookup.reserve(sizeof(FEXCore::Core::CPUState));
|
||||
for (auto& it : *ContextClassification) {
|
||||
@@ -453,12 +442,11 @@ static void ResetClassificationAccesses(ContextInfo* ContextClassificationInfo,
|
||||
|
||||
SetAccess(Offset++, LastAccessType::INVALID);
|
||||
SetAccess(Offset++, LastAccessType::INVALID);
|
||||
SetAccess(Offset++, LastAccessType::INVALID);
|
||||
}
|
||||
|
||||
struct BlockInfo {
|
||||
fextl::vector<FEXCore::IR::OrderedNode*> Predecessors;
|
||||
fextl::vector<FEXCore::IR::OrderedNode*> Successors;
|
||||
fextl::vector<FEXCore::IR::Ref> Predecessors;
|
||||
fextl::vector<FEXCore::IR::Ref> Successors;
|
||||
ContextInfo IncomingClassifiedStruct;
|
||||
ContextInfo OutgoingClassifiedStruct;
|
||||
};
|
||||
@@ -468,12 +456,9 @@ public:
|
||||
explicit RCLSE(bool SupportsAVX_)
|
||||
: SupportsAVX {SupportsAVX_} {
|
||||
ClassifyContextStruct(&ClassifiedStruct, SupportsAVX);
|
||||
DCE = FEXCore::IR::CreatePassDeadCodeElimination();
|
||||
}
|
||||
bool Run(FEXCore::IR::IREmitter* IREmit) override;
|
||||
void Run(FEXCore::IR::IREmitter* IREmit) override;
|
||||
private:
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> DCE;
|
||||
|
||||
ContextInfo ClassifiedStruct;
|
||||
fextl::unordered_map<FEXCore::IR::NodeID, BlockInfo> OffsetToBlockMap;
|
||||
|
||||
@@ -481,20 +466,32 @@ private:
|
||||
|
||||
ContextMemberInfo* FindMemberInfo(ContextInfo* ClassifiedInfo, uint32_t Offset, uint8_t Size);
|
||||
ContextMemberInfo* RecordAccess(ContextMemberInfo* Info, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size,
|
||||
LastAccessType AccessType, FEXCore::IR::OrderedNode* Node, FEXCore::IR::OrderedNode* StoreNode = nullptr);
|
||||
LastAccessType AccessType, FEXCore::IR::Ref Node, FEXCore::IR::Ref StoreNode = nullptr);
|
||||
ContextMemberInfo* RecordAccess(ContextInfo* ClassifiedInfo, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size,
|
||||
LastAccessType AccessType, FEXCore::IR::OrderedNode* Node, FEXCore::IR::OrderedNode* StoreNode = nullptr);
|
||||
LastAccessType AccessType, FEXCore::IR::Ref Node, FEXCore::IR::Ref StoreNode = nullptr);
|
||||
|
||||
bool HandleLoadFlag(FEXCore::IR::IREmitter* IREmit, ContextInfo* LocalInfo, FEXCore::IR::OrderedNode* CodeNode, unsigned Flag);
|
||||
void HandleLoadFlag(FEXCore::IR::IREmitter* IREmit, ContextInfo* LocalInfo, FEXCore::IR::Ref CodeNode, unsigned Flag);
|
||||
|
||||
// Classify context loads and stores.
|
||||
bool ClassifyContextLoad(FEXCore::IR::IREmitter* IREmit, ContextInfo* LocalInfo, FEXCore::IR::RegisterClassType Class, uint32_t Offset,
|
||||
uint8_t Size, FEXCore::IR::OrderedNode* CodeNode, FEXCore::IR::NodeIterator BlockEnd);
|
||||
bool ClassifyContextStore(FEXCore::IR::IREmitter* IREmit, ContextInfo* LocalInfo, FEXCore::IR::RegisterClassType Class, uint32_t Offset,
|
||||
uint8_t Size, FEXCore::IR::OrderedNode* CodeNode, FEXCore::IR::OrderedNode* ValueNode);
|
||||
void ClassifyContextLoad(FEXCore::IR::IREmitter* IREmit, ContextInfo* LocalInfo, FEXCore::IR::RegisterClassType Class, uint32_t Offset,
|
||||
uint8_t Size, FEXCore::IR::Ref CodeNode, FEXCore::IR::NodeIterator BlockEnd);
|
||||
void ClassifyContextStore(FEXCore::IR::IREmitter* IREmit, ContextInfo* LocalInfo, FEXCore::IR::RegisterClassType Class, uint32_t Offset,
|
||||
uint8_t Size, FEXCore::IR::Ref CodeNode, FEXCore::IR::Ref ValueNode);
|
||||
|
||||
// Block local Passes
|
||||
bool RedundantStoreLoadElimination(FEXCore::IR::IREmitter* IREmit);
|
||||
void RedundantStoreLoadElimination(FEXCore::IR::IREmitter* IREmit);
|
||||
|
||||
unsigned OffsetForReg(FEXCore::IR::RegisterClassType Class, unsigned Reg, unsigned Size) {
|
||||
if (Class == FEXCore::IR::FPRClass) {
|
||||
return Size == 32 ? offsetof(FEXCore::Core::CPUState, xmm.avx.data[Reg][0]) : offsetof(FEXCore::Core::CPUState, xmm.sse.data[Reg][0]);
|
||||
} else if (Reg == FEXCore::Core::CPUState::PF_AS_GREG) {
|
||||
return offsetof(FEXCore::Core::CPUState, pf_raw);
|
||||
} else if (Reg == FEXCore::Core::CPUState::AF_AS_GREG) {
|
||||
return offsetof(FEXCore::Core::CPUState, af_raw);
|
||||
} else {
|
||||
return offsetof(FEXCore::Core::CPUState, gregs[Reg]);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
ContextMemberInfo* RCLSE::FindMemberInfo(ContextInfo* ContextClassificationInfo, uint32_t Offset, uint8_t Size) {
|
||||
@@ -502,7 +499,7 @@ ContextMemberInfo* RCLSE::FindMemberInfo(ContextInfo* ContextClassificationInfo,
|
||||
}
|
||||
|
||||
ContextMemberInfo* RCLSE::RecordAccess(ContextMemberInfo* Info, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size,
|
||||
LastAccessType AccessType, FEXCore::IR::OrderedNode* ValueNode, FEXCore::IR::OrderedNode* StoreNode) {
|
||||
LastAccessType AccessType, FEXCore::IR::Ref ValueNode, FEXCore::IR::Ref StoreNode) {
|
||||
LOGMAN_THROW_AA_FMT((Offset + Size) <= (Info->Class.Offset + Info->Class.Size), "Access to context item went over member size");
|
||||
LOGMAN_THROW_AA_FMT(Info->Accessed != LastAccessType::INVALID, "Tried to access invalid member");
|
||||
|
||||
@@ -526,13 +523,13 @@ ContextMemberInfo* RCLSE::RecordAccess(ContextMemberInfo* Info, FEXCore::IR::Reg
|
||||
}
|
||||
|
||||
ContextMemberInfo* RCLSE::RecordAccess(ContextInfo* ClassifiedInfo, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size,
|
||||
LastAccessType AccessType, FEXCore::IR::OrderedNode* ValueNode, FEXCore::IR::OrderedNode* StoreNode) {
|
||||
LastAccessType AccessType, FEXCore::IR::Ref ValueNode, FEXCore::IR::Ref StoreNode) {
|
||||
ContextMemberInfo* Info = FindMemberInfo(ClassifiedInfo, Offset, Size);
|
||||
return RecordAccess(Info, RegClass, Offset, Size, AccessType, ValueNode, StoreNode);
|
||||
}
|
||||
|
||||
bool RCLSE::ClassifyContextLoad(FEXCore::IR::IREmitter* IREmit, ContextInfo* LocalInfo, FEXCore::IR::RegisterClassType Class,
|
||||
uint32_t Offset, uint8_t Size, FEXCore::IR::OrderedNode* CodeNode, FEXCore::IR::NodeIterator BlockEnd) {
|
||||
void RCLSE::ClassifyContextLoad(FEXCore::IR::IREmitter* IREmit, ContextInfo* LocalInfo, FEXCore::IR::RegisterClassType Class,
|
||||
uint32_t Offset, uint8_t Size, FEXCore::IR::Ref CodeNode, FEXCore::IR::NodeIterator BlockEnd) {
|
||||
auto Info = FindMemberInfo(LocalInfo, Offset, Size);
|
||||
ContextMemberInfo PreviousMemberInfoCopy = *Info;
|
||||
RecordAccess(Info, Class, Offset, Size, LastAccessType::READ, CodeNode);
|
||||
@@ -544,14 +541,12 @@ bool RCLSE::ClassifyContextLoad(FEXCore::IR::IREmitter* IREmit, ContextInfo* Loc
|
||||
// - Previous access was a store, and we are redundantly loading immediately after the store. Eliminating the store.
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, PreviousMemberInfoCopy.ValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Class, Offset, Size, LastAccessType::READ, PreviousMemberInfoCopy.ValueNode);
|
||||
return true;
|
||||
}
|
||||
// TODO: Optimize the case of partial loads.
|
||||
return false;
|
||||
}
|
||||
|
||||
bool RCLSE::ClassifyContextStore(FEXCore::IR::IREmitter* IREmit, ContextInfo* LocalInfo, FEXCore::IR::RegisterClassType Class,
|
||||
uint32_t Offset, uint8_t Size, FEXCore::IR::OrderedNode* CodeNode, FEXCore::IR::OrderedNode* ValueNode) {
|
||||
void RCLSE::ClassifyContextStore(FEXCore::IR::IREmitter* IREmit, ContextInfo* LocalInfo, FEXCore::IR::RegisterClassType Class,
|
||||
uint32_t Offset, uint8_t Size, FEXCore::IR::Ref CodeNode, FEXCore::IR::Ref ValueNode) {
|
||||
auto Info = FindMemberInfo(LocalInfo, Offset, Size);
|
||||
ContextMemberInfo PreviousMemberInfoCopy = *Info;
|
||||
RecordAccess(Info, Class, Offset, Size, LastAccessType::WRITE, ValueNode, CodeNode);
|
||||
@@ -563,15 +558,13 @@ bool RCLSE::ClassifyContextStore(FEXCore::IR::IREmitter* IREmit, ContextInfo* Lo
|
||||
// Revisit when the new RA lands.
|
||||
#if 0
|
||||
IREmit->Remove(PreviousMemberInfoCopy.StoreNode);
|
||||
return true;
|
||||
#endif
|
||||
}
|
||||
|
||||
// TODO: Optimize the case of partial stores.
|
||||
return false;
|
||||
}
|
||||
|
||||
bool RCLSE::HandleLoadFlag(FEXCore::IR::IREmitter* IREmit, ContextInfo* LocalInfo, FEXCore::IR::OrderedNode* CodeNode, unsigned Flag) {
|
||||
void RCLSE::HandleLoadFlag(FEXCore::IR::IREmitter* IREmit, ContextInfo* LocalInfo, FEXCore::IR::Ref CodeNode, unsigned Flag) {
|
||||
const auto FlagOffset = offsetof(FEXCore::Core::CPUState, flags[Flag]);
|
||||
auto Info = FindMemberInfo(LocalInfo, FlagOffset, 1);
|
||||
LastAccessType LastAccess = Info->Accessed;
|
||||
@@ -582,14 +575,10 @@ bool RCLSE::HandleLoadFlag(FEXCore::IR::IREmitter* IREmit, ContextInfo* LocalInf
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, LastValueNode);
|
||||
RecordAccess(Info, FEXCore::IR::GPRClass, FlagOffset, 1, LastAccessType::READ, LastValueNode);
|
||||
return true;
|
||||
} else if (IsReadAccess(LastAccess)) {
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, LastValueNode);
|
||||
RecordAccess(Info, FEXCore::IR::GPRClass, FlagOffset, 1, LastAccessType::READ, LastValueNode);
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -624,11 +613,10 @@ bool RCLSE::HandleLoadFlag(FEXCore::IR::IREmitter* IREmit, ContextInfo* LocalInf
|
||||
* (%%176) StoreContext %175 i128, 0x10, 0xa0
|
||||
|
||||
*/
|
||||
bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter* IREmit) {
|
||||
void RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter* IREmit) {
|
||||
using namespace FEXCore;
|
||||
using namespace FEXCore::IR;
|
||||
|
||||
bool Changed = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
auto OriginalWriteCursor = IREmit->GetWriteCursor();
|
||||
|
||||
@@ -645,16 +633,19 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter* IREmit) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
if (IROp->Op == OP_STORECONTEXT) {
|
||||
auto Op = IROp->CW<IR::IROp_StoreContext>();
|
||||
Changed |= ClassifyContextStore(IREmit, &LocalInfo, Op->Class, Op->Offset, IROp->Size, CodeNode, CurrentIR.GetNode(Op->Value));
|
||||
ClassifyContextStore(IREmit, &LocalInfo, Op->Class, Op->Offset, IROp->Size, CodeNode, CurrentIR.GetNode(Op->Value));
|
||||
} else if (IROp->Op == OP_STOREREGISTER) {
|
||||
auto Op = IROp->CW<IR::IROp_StoreRegister>();
|
||||
Changed |= ClassifyContextStore(IREmit, &LocalInfo, Op->Class, Op->Offset, IROp->Size, CodeNode, CurrentIR.GetNode(Op->Value));
|
||||
auto Offset = OffsetForReg(Op->Class, Op->Reg, IROp->Size);
|
||||
|
||||
ClassifyContextStore(IREmit, &LocalInfo, Op->Class, Offset, IROp->Size, CodeNode, CurrentIR.GetNode(Op->Value));
|
||||
} else if (IROp->Op == OP_LOADREGISTER) {
|
||||
auto Op = IROp->CW<IR::IROp_LoadRegister>();
|
||||
Changed |= ClassifyContextLoad(IREmit, &LocalInfo, Op->Class, Op->Offset, IROp->Size, CodeNode, BlockEnd);
|
||||
auto Offset = OffsetForReg(Op->Class, Op->Reg, IROp->Size);
|
||||
ClassifyContextLoad(IREmit, &LocalInfo, Op->Class, Offset, IROp->Size, CodeNode, BlockEnd);
|
||||
} else if (IROp->Op == OP_LOADCONTEXT) {
|
||||
auto Op = IROp->CW<IR::IROp_LoadContext>();
|
||||
Changed |= ClassifyContextLoad(IREmit, &LocalInfo, Op->Class, Op->Offset, IROp->Size, CodeNode, BlockEnd);
|
||||
ClassifyContextLoad(IREmit, &LocalInfo, Op->Class, Op->Offset, IROp->Size, CodeNode, BlockEnd);
|
||||
} else if (IROp->Op == OP_STOREFLAG) {
|
||||
const auto Op = IROp->CW<IR::IROp_StoreFlag>();
|
||||
const auto FlagOffset = offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag;
|
||||
@@ -665,7 +656,6 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter* IREmit) {
|
||||
// Flags don't alias, so we can take the simple route here. Kill any flags that have been overwritten
|
||||
if (LastStoreNode != nullptr) {
|
||||
IREmit->Remove(LastStoreNode);
|
||||
Changed = true;
|
||||
}
|
||||
} else if (IROp->Op == OP_INVALIDATEFLAGS) {
|
||||
auto Op = IROp->CW<IR::IROp_InvalidateFlags>();
|
||||
@@ -686,15 +676,14 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter* IREmit) {
|
||||
RecordAccess(&LocalInfo, FEXCore::IR::GPRClass, FlagOffset, 1, LastAccessType::WRITE, IREmit->_Constant(0), CodeNode);
|
||||
|
||||
IREmit->Remove(LastStoreNode);
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
} else if (IROp->Op == OP_LOADFLAG) {
|
||||
const auto Op = IROp->CW<IR::IROp_LoadFlag>();
|
||||
|
||||
Changed |= HandleLoadFlag(IREmit, &LocalInfo, CodeNode, Op->Flag);
|
||||
HandleLoadFlag(IREmit, &LocalInfo, CodeNode, Op->Flag);
|
||||
} else if (IROp->Op == OP_LOADDF) {
|
||||
Changed |= HandleLoadFlag(IREmit, &LocalInfo, CodeNode, X86State::RFLAG_DF_RAW_LOC);
|
||||
HandleLoadFlag(IREmit, &LocalInfo, CodeNode, X86State::RFLAG_DF_RAW_LOC);
|
||||
} else if (IROp->Op == OP_SYSCALL || IROp->Op == OP_INLINESYSCALL) {
|
||||
FEXCore::IR::SyscallFlags Flags {};
|
||||
if (IROp->Op == OP_SYSCALL) {
|
||||
@@ -717,21 +706,11 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter* IREmit) {
|
||||
}
|
||||
|
||||
IREmit->SetWriteCursor(OriginalWriteCursor);
|
||||
|
||||
return Changed;
|
||||
}
|
||||
|
||||
bool RCLSE::Run(FEXCore::IR::IREmitter* IREmit) {
|
||||
void RCLSE::Run(FEXCore::IR::IREmitter* IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::RCLSE");
|
||||
bool Changed = false;
|
||||
|
||||
// Run up to 5 times
|
||||
for (int i = 0; i < 5 && RedundantStoreLoadElimination(IREmit); i++) {
|
||||
Changed = true;
|
||||
DCE->Run(IREmit);
|
||||
}
|
||||
|
||||
return Changed;
|
||||
RedundantStoreLoadElimination(IREmit);
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
@@ -14,7 +14,6 @@ $end_info$
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
|
||||
#include <memory>
|
||||
#include <stddef.h>
|
||||
@@ -24,221 +23,81 @@ namespace FEXCore::IR {
|
||||
|
||||
constexpr int PropagationRounds = 5;
|
||||
|
||||
// Return a bit representing a single GPR or FPR.
|
||||
static inline uint64_t RegBit(RegisterClassType Class, uint32_t Reg) {
|
||||
uint32_t AdjustedReg = (Class == FPRClass) ? (32 + Reg) : Reg;
|
||||
|
||||
return 1UL << AdjustedReg;
|
||||
}
|
||||
|
||||
class DeadStoreElimination final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
explicit DeadStoreElimination(bool SupportsAVX_)
|
||||
: SupportsAVX {SupportsAVX_} {}
|
||||
|
||||
bool Run(IREmitter* IREmit) override;
|
||||
|
||||
private:
|
||||
bool SupportsAVX;
|
||||
|
||||
bool IsFPR(uint32_t Offset) const {
|
||||
const auto [begin, end] = [this]() -> std::pair<ptrdiff_t, ptrdiff_t> {
|
||||
if (SupportsAVX) {
|
||||
return {offsetof(FEXCore::Core::CpuStateFrame, State.xmm.avx.data[0][0]),
|
||||
offsetof(FEXCore::Core::CpuStateFrame, State.xmm.avx.data[16][0])};
|
||||
} else {
|
||||
return {offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[0][0]),
|
||||
offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[16][0])};
|
||||
}
|
||||
}();
|
||||
|
||||
if (Offset < begin || Offset >= end) {
|
||||
return false;
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
bool IsTrackedWriteFPR(uint32_t Offset, uint8_t Size) const {
|
||||
if (Size != 16 && Size != 8 && Size != 4) {
|
||||
return false;
|
||||
}
|
||||
if (Offset & 15) {
|
||||
return false;
|
||||
}
|
||||
|
||||
return IsFPR(Offset);
|
||||
}
|
||||
|
||||
uint64_t FPRBit(uint32_t Offset, uint32_t Size) const {
|
||||
if (!IsFPR(Offset)) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
const auto begin = offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0]);
|
||||
|
||||
const auto regSize = SupportsAVX ? Core::CPUState::XMM_AVX_REG_SIZE : Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto regn = (Offset - begin) / regSize;
|
||||
const auto bitn = regn * 3;
|
||||
|
||||
if (!IsTrackedWriteFPR(Offset, Size)) {
|
||||
return 7UL << (bitn);
|
||||
}
|
||||
|
||||
if (Size == 16) {
|
||||
return 7UL << (bitn);
|
||||
} else if (Size == 8) {
|
||||
return 3UL << (bitn);
|
||||
} else if (Size == 4) {
|
||||
return 1UL << (bitn);
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unexpected FPR size {}", Size);
|
||||
}
|
||||
|
||||
return 7UL << (bitn); // Return maximum on failure case
|
||||
}
|
||||
void Run(IREmitter* IREmit) override;
|
||||
};
|
||||
|
||||
struct FlagInfo {
|
||||
uint64_t reads {0};
|
||||
uint64_t writes {0};
|
||||
uint64_t kill {0};
|
||||
};
|
||||
|
||||
|
||||
struct GPRInfo {
|
||||
uint32_t reads {0};
|
||||
uint32_t writes {0};
|
||||
uint32_t kill {0};
|
||||
};
|
||||
|
||||
bool IsFullGPR(uint32_t Offset, uint8_t Size) {
|
||||
if (Size != 8) {
|
||||
return false;
|
||||
}
|
||||
if (Offset & 7) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if (Offset < 8 || Offset >= (17 * 8)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
bool IsGPR(uint32_t Offset) {
|
||||
|
||||
if (Offset < 8 || Offset >= (17 * 8)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
uint32_t GPRBit(uint32_t Offset) {
|
||||
if (!IsGPR(Offset)) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
return 1 << ((Offset - 8) / 8);
|
||||
}
|
||||
|
||||
struct FPRInfo {
|
||||
struct ReadWriteKill {
|
||||
uint64_t reads {0};
|
||||
uint64_t writes {0};
|
||||
uint64_t kill {0};
|
||||
};
|
||||
|
||||
struct Info {
|
||||
FlagInfo flag;
|
||||
GPRInfo gpr;
|
||||
FPRInfo fpr;
|
||||
ReadWriteKill flag;
|
||||
ReadWriteKill reg;
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief This is a temporary pass to detect simple multiblock dead flag/gpr/fpr stores
|
||||
* @brief This is a temporary pass to detect simple multiblock dead flag/reg stores
|
||||
*
|
||||
* First pass computes which flags/gprs/fprs are read and written per block
|
||||
* First pass computes which flags/regs are read and written per block
|
||||
*
|
||||
* Second pass computes which flags/gprs/fprs are stored, but overwritten by the next block(s).
|
||||
* It also propagates this information a few times to catch dead flags/gprs/fprs across multiple blocks.
|
||||
* Second pass computes which flags/regs are stored, but overwritten by the next block(s).
|
||||
* It also propagates this information a few times to catch dead flags/regs across multiple blocks.
|
||||
*
|
||||
* Third pass removes the dead stores.
|
||||
*
|
||||
*/
|
||||
bool DeadStoreElimination::Run(IREmitter* IREmit) {
|
||||
void DeadStoreElimination::Run(IREmitter* IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::DSE");
|
||||
|
||||
fextl::unordered_map<OrderedNode*, Info> InfoMap;
|
||||
|
||||
bool Changed = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
fextl::vector<Info> InfoMap(CurrentIR.GetSSACount());
|
||||
|
||||
// Pass 1
|
||||
// Compute flags/gprs/fprs read/writes per block
|
||||
// Compute flags/regs read/writes per block
|
||||
// This is conservative and doesn't try to be smart about loads after writes
|
||||
{
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
auto& BlockInfo = InfoMap[CurrentIR.GetID(BlockNode).Value];
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
|
||||
auto ClassifyRegisterStore = [this](Info& BlockInfo, uint32_t Offset, uint8_t Size) {
|
||||
//// GPR ////
|
||||
if (IsFullGPR(Offset, Size)) {
|
||||
BlockInfo.gpr.writes |= GPRBit(Offset);
|
||||
} else {
|
||||
BlockInfo.gpr.reads |= GPRBit(Offset);
|
||||
}
|
||||
|
||||
//// FPR ////
|
||||
if (IsTrackedWriteFPR(Offset, Size)) {
|
||||
BlockInfo.fpr.writes |= FPRBit(Offset, Size);
|
||||
} else {
|
||||
BlockInfo.fpr.reads |= FPRBit(Offset, Size);
|
||||
}
|
||||
};
|
||||
|
||||
auto ClassifyRegisterLoad = [this](Info& BlockInfo, uint32_t Offset, uint8_t Size) {
|
||||
//// GPR ////
|
||||
BlockInfo.gpr.reads |= GPRBit(Offset);
|
||||
|
||||
//// FPR ////
|
||||
BlockInfo.fpr.reads |= FPRBit(Offset, Size);
|
||||
};
|
||||
|
||||
//// Flags ////
|
||||
if (IROp->Op == OP_STOREFLAG) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
BlockInfo.flag.writes |= 1UL << Op->Flag;
|
||||
} else if (IROp->Op == OP_INVALIDATEFLAGS) {
|
||||
auto Op = IROp->C<IR::IROp_InvalidateFlags>();
|
||||
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
BlockInfo.flag.writes |= Op->Flags;
|
||||
} else if (IROp->Op == OP_LOADFLAG) {
|
||||
auto Op = IROp->C<IR::IROp_LoadFlag>();
|
||||
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
BlockInfo.flag.reads |= 1UL << Op->Flag;
|
||||
} else if (IROp->Op == OP_LOADDF) {
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
BlockInfo.flag.reads |= 1UL << X86State::RFLAG_DF_RAW_LOC;
|
||||
} else if (IROp->Op == OP_STOREREGISTER) {
|
||||
auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
ClassifyRegisterStore(BlockInfo, Op->Offset, IROp->Size);
|
||||
BlockInfo.reg.writes |= RegBit(Op->Class, Op->Reg);
|
||||
} else if (IROp->Op == OP_LOADREGISTER) {
|
||||
auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
ClassifyRegisterLoad(BlockInfo, Op->Offset, IROp->Size);
|
||||
BlockInfo.reg.reads |= RegBit(Op->Class, Op->Reg);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Pass 2
|
||||
// Compute flags/gprs/fprs that are stored, but always ovewritten in the next blocks
|
||||
// Compute flags/registers that are stored, but always ovewritten in the next blocks
|
||||
// Propagate the information a few times to eliminate more
|
||||
for (int i = 0; i < PropagationRounds; i++) {
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
@@ -248,75 +107,33 @@ bool DeadStoreElimination::Run(IREmitter* IREmit) {
|
||||
|
||||
if (IROp->Op == OP_JUMP) {
|
||||
auto Op = IROp->C<IR::IROp_Jump>();
|
||||
OrderedNode* TargetNode = CurrentIR.GetNode(Op->Header.Args[0]);
|
||||
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
auto& TargetInfo = InfoMap[TargetNode];
|
||||
|
||||
//// Flags ////
|
||||
auto& BlockInfo = InfoMap[CurrentIR.GetID(BlockNode).Value];
|
||||
auto& TargetInfo = InfoMap[Op->Header.Args[0].ID().Value];
|
||||
|
||||
// stores to remove are written by the next block but not read
|
||||
BlockInfo.flag.kill = TargetInfo.flag.writes & ~(TargetInfo.flag.reads) & ~BlockInfo.flag.reads;
|
||||
BlockInfo.reg.kill = TargetInfo.reg.writes & ~(TargetInfo.reg.reads) & ~BlockInfo.reg.reads;
|
||||
|
||||
// Flags that are written by the next block can be considered as written by this block, if not read
|
||||
BlockInfo.flag.writes |= BlockInfo.flag.kill & ~BlockInfo.flag.reads;
|
||||
|
||||
|
||||
//// GPRs ////
|
||||
|
||||
// stores to remove are written by the next block but not read
|
||||
BlockInfo.gpr.kill = TargetInfo.gpr.writes & ~(TargetInfo.gpr.reads) & ~BlockInfo.gpr.reads;
|
||||
|
||||
// GPRs that are written by the next block can be considered as written by this block, if not read
|
||||
BlockInfo.gpr.writes |= BlockInfo.gpr.kill & ~BlockInfo.gpr.reads;
|
||||
|
||||
|
||||
//// FPRs ////
|
||||
|
||||
// stores to remove are written by the next block but not read
|
||||
BlockInfo.fpr.kill = TargetInfo.fpr.writes & ~(TargetInfo.fpr.reads) & ~BlockInfo.fpr.reads;
|
||||
|
||||
// FPRs that are written by the next block can be considered as written by this block, if not read
|
||||
BlockInfo.fpr.writes |= BlockInfo.fpr.kill & ~BlockInfo.fpr.reads;
|
||||
|
||||
BlockInfo.reg.writes |= BlockInfo.reg.kill & ~BlockInfo.reg.reads;
|
||||
} else if (IROp->Op == OP_CONDJUMP) {
|
||||
auto Op = IROp->C<IR::IROp_CondJump>();
|
||||
|
||||
OrderedNode* TrueTargetNode = CurrentIR.GetNode(Op->TrueBlock);
|
||||
OrderedNode* FalseTargetNode = CurrentIR.GetNode(Op->FalseBlock);
|
||||
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
auto& TrueTargetInfo = InfoMap[TrueTargetNode];
|
||||
auto& FalseTargetInfo = InfoMap[FalseTargetNode];
|
||||
|
||||
//// Flags ////
|
||||
auto& BlockInfo = InfoMap[CurrentIR.GetID(BlockNode).Value];
|
||||
auto& TrueTargetInfo = InfoMap[Op->TrueBlock.ID().Value];
|
||||
auto& FalseTargetInfo = InfoMap[Op->FalseBlock.ID().Value];
|
||||
|
||||
// stores to remove are written by the next blocks but not read
|
||||
BlockInfo.flag.kill = TrueTargetInfo.flag.writes & ~(TrueTargetInfo.flag.reads) & ~BlockInfo.flag.reads;
|
||||
BlockInfo.reg.kill = TrueTargetInfo.reg.writes & ~(TrueTargetInfo.reg.reads) & ~BlockInfo.reg.reads;
|
||||
|
||||
BlockInfo.flag.kill &= FalseTargetInfo.flag.writes & ~(FalseTargetInfo.flag.reads) & ~BlockInfo.flag.reads;
|
||||
BlockInfo.reg.kill &= FalseTargetInfo.reg.writes & ~(FalseTargetInfo.reg.reads) & ~BlockInfo.reg.reads;
|
||||
|
||||
// Flags that are written by the next blocks can be considered as written by this block, if not read
|
||||
BlockInfo.flag.writes |= BlockInfo.flag.kill & ~BlockInfo.flag.reads;
|
||||
|
||||
|
||||
//// GPRs ////
|
||||
|
||||
// stores to remove are written by the next blocks but not read
|
||||
BlockInfo.gpr.kill = TrueTargetInfo.gpr.writes & ~(TrueTargetInfo.gpr.reads) & ~BlockInfo.gpr.reads;
|
||||
BlockInfo.gpr.kill &= FalseTargetInfo.gpr.writes & ~(FalseTargetInfo.gpr.reads) & ~BlockInfo.gpr.reads;
|
||||
|
||||
// GPRs that are written by the next blocks can be considered as written by this block, if not read
|
||||
BlockInfo.gpr.writes |= BlockInfo.gpr.kill & ~BlockInfo.gpr.reads;
|
||||
|
||||
|
||||
//// FPRs ////
|
||||
|
||||
// stores to remove are written by the next blocks but not read
|
||||
BlockInfo.fpr.kill = TrueTargetInfo.fpr.writes & ~(TrueTargetInfo.fpr.reads) & ~BlockInfo.fpr.reads;
|
||||
BlockInfo.fpr.kill &= FalseTargetInfo.fpr.writes & ~(FalseTargetInfo.fpr.reads) & ~BlockInfo.fpr.reads;
|
||||
|
||||
// FPRs that are written by the next blocks can be considered as written by this block, if not read
|
||||
BlockInfo.fpr.writes |= BlockInfo.fpr.kill & ~BlockInfo.fpr.reads;
|
||||
BlockInfo.reg.writes |= BlockInfo.reg.kill & ~BlockInfo.reg.reads;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -325,55 +142,31 @@ bool DeadStoreElimination::Run(IREmitter* IREmit) {
|
||||
// Remove the dead stores
|
||||
{
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
auto& BlockInfo = InfoMap[CurrentIR.GetID(BlockNode).Value];
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
|
||||
auto RemoveDeadRegisterStore = [this](FEXCore::IR::IREmitter* IREmit, FEXCore::IR::OrderedNode* CodeNode, Info& BlockInfo,
|
||||
uint32_t Offset, uint8_t Size) -> bool {
|
||||
bool Changed {};
|
||||
//// GPRs ////
|
||||
// If this OP_STOREREGISTER is never read, remove it
|
||||
if (BlockInfo.gpr.kill & GPRBit(Offset)) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Changed = true;
|
||||
}
|
||||
|
||||
//// FPRs ////
|
||||
// If this OP_STOREREGISTER is never read, remove it
|
||||
if ((BlockInfo.fpr.kill & FPRBit(Offset, Size)) == FPRBit(Offset, Size) && (FPRBit(Offset, Size) != 0)) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Changed = true;
|
||||
}
|
||||
|
||||
return Changed;
|
||||
};
|
||||
|
||||
//// Flags ////
|
||||
if (IROp->Op == OP_STOREFLAG) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
// If this StoreFlag is never read, remove it
|
||||
if (BlockInfo.flag.kill & (1UL << Op->Flag)) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Changed = true;
|
||||
}
|
||||
} else if (IROp->Op == OP_STOREREGISTER) {
|
||||
auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
Changed |= RemoveDeadRegisterStore(IREmit, CodeNode, BlockInfo, Op->Offset, IROp->Size);
|
||||
// If this OP_STOREREGISTER is never read, remove it
|
||||
if (BlockInfo.reg.kill & RegBit(Op->Class, Op->Reg)) {
|
||||
IREmit->Remove(CodeNode);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return Changed;
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadStoreElimination(bool SupportsAVX) {
|
||||
return fextl::make_unique<DeadStoreElimination>(SupportsAVX);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadStoreElimination() {
|
||||
return fextl::make_unique<DeadStoreElimination>();
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -1,228 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: ir|opts
|
||||
desc: Sorts the ssa storage in memory, needed for RA and others
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <cstdint>
|
||||
#include <cstring>
|
||||
#include <memory>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
// struct to avoid zero-initialization
|
||||
struct RemapNode {
|
||||
IR::NodeID NodeID;
|
||||
};
|
||||
|
||||
static_assert(sizeof(RemapNode) == 4);
|
||||
|
||||
class IRCompaction final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
IRCompaction(FEXCore::Utils::IntrusivePooledAllocator& Allocator);
|
||||
bool Run(IREmitter* IREmit) override;
|
||||
|
||||
private:
|
||||
static constexpr size_t AlignSize = 0x2000;
|
||||
OpDispatchBuilder LocalBuilder;
|
||||
fextl::vector<RemapNode> OldToNewRemap;
|
||||
struct CodeBlockData {
|
||||
OrderedNode* OldNode;
|
||||
OrderedNode* NewNode;
|
||||
};
|
||||
|
||||
fextl::vector<CodeBlockData> GeneratedCodeBlocks {};
|
||||
};
|
||||
|
||||
IRCompaction::IRCompaction(FEXCore::Utils::IntrusivePooledAllocator& Allocator)
|
||||
: LocalBuilder {Allocator} {
|
||||
OldToNewRemap.resize(AlignSize);
|
||||
}
|
||||
|
||||
bool IRCompaction::Run(IREmitter* IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::IRCompaction");
|
||||
|
||||
LocalBuilder.ReownOrClaimBuffer();
|
||||
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
uint32_t NodeCount = CurrentIR.GetSSACount();
|
||||
|
||||
if (OldToNewRemap.size() < NodeCount) {
|
||||
OldToNewRemap.resize(std::max(OldToNewRemap.size() * 2U, AlignUp(NodeCount, AlignSize)));
|
||||
}
|
||||
#ifndef NDEBUG
|
||||
memset(&OldToNewRemap.at(0), 0xFF, NodeCount * sizeof(RemapNode));
|
||||
#endif
|
||||
|
||||
GeneratedCodeBlocks.clear();
|
||||
|
||||
// Reset our local working list
|
||||
LocalBuilder.ResetWorkingList();
|
||||
auto LocalIR = LocalBuilder.ViewIR();
|
||||
|
||||
uintptr_t LocalListBegin = LocalIR.GetListData();
|
||||
uintptr_t LocalDataBegin = LocalIR.GetData();
|
||||
|
||||
uintptr_t ListBegin = CurrentIR.GetListData();
|
||||
|
||||
auto HeaderNode = CurrentIR.GetHeaderNode();
|
||||
auto HeaderOp = CurrentIR.GetHeader();
|
||||
LOGMAN_THROW_AA_FMT(HeaderOp->Header.Op == OP_IRHEADER, "First op wasn't IRHeader");
|
||||
|
||||
// This compaction pass is something that we need to ensure correct ordering and distances between IROps
|
||||
// Later on we assume that an IROp's SSA value live range is its Node locations
|
||||
//
|
||||
// RA distance calculation is calculated purely on the Node locations
|
||||
// So we need to reorder those
|
||||
//
|
||||
// Additionally there may be some dead ops hanging out in the IR list that are orphaned.
|
||||
// These can also be dropped during this pass
|
||||
|
||||
// First thing is first, we need to do some housekeeping
|
||||
// Create the IRHeader op
|
||||
// Create the codeblocks
|
||||
// Then create all the ops inside the code blocks
|
||||
|
||||
// Zero is always zero(invalid)
|
||||
OldToNewRemap[0].NodeID.Invalidate();
|
||||
auto LocalHeaderOp = LocalBuilder._IRHeader(OrderedNodeWrapper::WrapOffset(0).GetNode(ListBegin), HeaderOp->OriginalRIP,
|
||||
HeaderOp->BlockCount, HeaderOp->NumHostInstructions);
|
||||
|
||||
OldToNewRemap[CurrentIR.GetID(HeaderNode).Value].NodeID = LocalIR.GetID(LocalHeaderOp.Node);
|
||||
|
||||
{
|
||||
// Generate our codeblocks and link them together
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
LOGMAN_THROW_AA_FMT(BlockHeader->Op == OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
|
||||
auto LocalBlockIRNode = LocalBuilder._CodeBlock(LocalHeaderOp, LocalHeaderOp); // Use LocalHeaderOp as a dummy arg for now
|
||||
OldToNewRemap[CurrentIR.GetID(BlockNode).Value].NodeID = LocalIR.GetID(LocalBlockIRNode.Node);
|
||||
GeneratedCodeBlocks.emplace_back(CodeBlockData {BlockNode, LocalBlockIRNode});
|
||||
}
|
||||
|
||||
// Link the IRHeader to the first code block
|
||||
LocalHeaderOp.first->Blocks = GeneratedCodeBlocks[0].NewNode->Wrapped(LocalListBegin);
|
||||
}
|
||||
|
||||
{
|
||||
// Copy all of our IR ops over to the new location
|
||||
for (auto& Block : GeneratedCodeBlocks) {
|
||||
|
||||
// Isolate block contents from any previous headers/blocks
|
||||
LocalBuilder.SetWriteCursor(nullptr);
|
||||
|
||||
CodeBlockData FirstNode {};
|
||||
CodeBlockData LastNode {};
|
||||
uint32_t i {};
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(Block.OldNode)) {
|
||||
const size_t OpSize = FEXCore::IR::GetSize(IROp->Op);
|
||||
const auto CodeID = CurrentIR.GetID(CodeNode);
|
||||
|
||||
// Allocate the ops locally for our local dispatch
|
||||
auto LocalPair = LocalBuilder.AllocateRawOp(OpSize);
|
||||
|
||||
// Copy usage infomation
|
||||
LocalPair.Node->NumUses = CodeNode->GetUses();
|
||||
|
||||
// Copy over the op
|
||||
memcpy(LocalPair.first, IROp, OpSize);
|
||||
|
||||
// Set our map remapper to map the new location
|
||||
// Even nodes that don't have a destination need to be in this map
|
||||
// Need to be able to remap branch targets any other bits
|
||||
OldToNewRemap[CodeID.Value].NodeID = LocalIR.GetID(LocalPair.Node);
|
||||
|
||||
if (i == 0) {
|
||||
FirstNode.OldNode = CodeNode;
|
||||
FirstNode.NewNode = LocalPair.Node;
|
||||
}
|
||||
|
||||
if (IROp->Op == OP_ENDBLOCK) {
|
||||
LastNode.OldNode = CodeNode;
|
||||
LastNode.NewNode = LocalPair.Node;
|
||||
}
|
||||
|
||||
++i;
|
||||
}
|
||||
|
||||
// Set the code block's begin and end correctly
|
||||
auto NewBlockIROp = Block.NewNode->Op(LocalDataBegin)->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
NewBlockIROp->Begin = FirstNode.NewNode->Wrapped(LocalListBegin);
|
||||
NewBlockIROp->Last = LastNode.NewNode->Wrapped(LocalListBegin);
|
||||
}
|
||||
}
|
||||
|
||||
{
|
||||
// Fixup the arguments of all the IROps
|
||||
for (auto& Block : GeneratedCodeBlocks) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
auto BlockIROp = LocalIR.GetOp<FEXCore::IR::IROp_CodeBlock>(Block.NewNode);
|
||||
LOGMAN_THROW_AA_FMT(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
#endif
|
||||
|
||||
for (auto [LocalNode, LocalIROp] : LocalIR.GetCode(Block.NewNode)) {
|
||||
|
||||
// Now that we have the op copied over, we need to modify SSA values to point to the new correct locations
|
||||
// This doesn't use IR::GetRAArgs(Op) because we need to remap all SSA nodes
|
||||
// Including ones that we don't RA
|
||||
const uint8_t NumArgs = IR::GetArgs(LocalIROp->Op);
|
||||
for (uint8_t i = 0; i < NumArgs; ++i) {
|
||||
const auto OldArg = LocalIROp->Args[i].ID();
|
||||
const auto NewArg = OldToNewRemap[OldArg.Value].NodeID;
|
||||
|
||||
#ifndef NDEBUG
|
||||
LOGMAN_THROW_A_FMT(NewArg.Value != UINT32_MAX, "Tried remapping unfound node %{}", OldArg);
|
||||
#endif
|
||||
|
||||
LocalIROp->Args[i].NodeOffset = NewArg.Value * sizeof(OrderedNode);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// uintptr_t OldListSize = CurrentIR.GetListSize();
|
||||
// uintptr_t OldDataSize = CurrentIR.GetDataSize();
|
||||
|
||||
// uintptr_t NewListSize = LocalIR.GetListSize();
|
||||
// uintptr_t NewDataSize = LocalIR.GetDataSize();
|
||||
|
||||
// if (NewListSize < OldListSize ||
|
||||
// NewDataSize < OldDataSize) {
|
||||
// if (NewListSize < OldListSize) {
|
||||
// LogMan::Msg::DFmt("Shaved {} bytes off the list size", OldListSize - NewListSize);
|
||||
// }
|
||||
// if (NewDataSize < OldDataSize) {
|
||||
// LogMan::Msg::DFmt("Shaved {} bytes off the data size", OldDataSize - NewDataSize);
|
||||
// }
|
||||
// }
|
||||
|
||||
// if (NewListSize > OldListSize ||
|
||||
// NewDataSize > OldDataSize) {
|
||||
// LOGMAN_MSG_A_FMT("Whoa. Compaction made the IR a different size when it shouldn't have. 0x{:x} > 0x{:x} or 0x{:x} > 0x{:x}",
|
||||
// NewListSize, OldListSize, NewDataSize, OldDataSize);
|
||||
// }
|
||||
|
||||
IREmit->CopyData(LocalBuilder);
|
||||
|
||||
LocalBuilder.DelayedDisownBuffer();
|
||||
return true;
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateIRCompaction(FEXCore::Utils::IntrusivePooledAllocator& Allocator) {
|
||||
return fextl::make_unique<IRCompaction>(Allocator);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -18,7 +18,7 @@ namespace FEXCore::IR::Debug {
|
||||
class IRDumper final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
IRDumper();
|
||||
bool Run(IREmitter* IREmit) override;
|
||||
void Run(IREmitter* IREmit) override;
|
||||
|
||||
private:
|
||||
FEX_CONFIG_OPT(DumpIR, DUMPIR);
|
||||
@@ -37,7 +37,7 @@ IRDumper::IRDumper() {
|
||||
}
|
||||
}
|
||||
|
||||
bool IRDumper::Run(IREmitter* IREmit) {
|
||||
void IRDumper::Run(IREmitter* IREmit) {
|
||||
auto RAPass = Manager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
IR::RegisterAllocationData* RA {};
|
||||
if (RAPass) {
|
||||
@@ -71,8 +71,6 @@ bool IRDumper::Run(IREmitter* IREmit) {
|
||||
LogMan::Msg::IFmt("IR-{} 0x{:x}:\n{}\n@@@@@\n", RA ? "post" : "pre", HeaderOp->OriginalRIP, out.str());
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateIRDumper() {
|
||||
|
||||
@@ -32,7 +32,7 @@ IRValidation::~IRValidation() {
|
||||
NodeIsLive.Free();
|
||||
}
|
||||
|
||||
bool IRValidation::Run(IREmitter* IREmit) {
|
||||
void IRValidation::Run(IREmitter* IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::IRValidation");
|
||||
|
||||
bool HadError = false;
|
||||
@@ -46,11 +46,12 @@ bool IRValidation::Run(IREmitter* IREmit) {
|
||||
OffsetToBlockMap.clear();
|
||||
EntryBlock = nullptr;
|
||||
|
||||
if (CurrentIR.GetSSACount() > MaxNodes) {
|
||||
NodeIsLive.Realloc(CurrentIR.GetSSACount());
|
||||
uint32_t Count = CurrentIR.GetSSACount();
|
||||
if (Count > MaxNodes) {
|
||||
NodeIsLive.Realloc(Count);
|
||||
}
|
||||
|
||||
fextl::vector<uint32_t> Uses(CurrentIR.GetSSACount(), 0);
|
||||
fextl::vector<uint32_t> Uses(Count, 0);
|
||||
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
auto HeaderOp = CurrentIR.GetHeader();
|
||||
@@ -62,8 +63,6 @@ bool IRValidation::Run(IREmitter* IREmit) {
|
||||
RAData = Manager->GetPass<IR::RegisterAllocationPass>("RA")->GetAllocationData();
|
||||
}
|
||||
|
||||
NodeIsLive.Set(1); // IRHEADER
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
LOGMAN_THROW_AA_FMT(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
@@ -75,6 +74,9 @@ bool IRValidation::Run(IREmitter* IREmit) {
|
||||
const auto BlockID = CurrentIR.GetID(BlockNode);
|
||||
BlockInfo* CurrentBlock = &OffsetToBlockMap.try_emplace(BlockID).first->second;
|
||||
|
||||
// We only allow defs local to a single block, so clear live set per block
|
||||
NodeIsLive.MemClear(Count);
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
const auto ID = CurrentIR.GetID(CodeNode);
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
@@ -125,21 +127,21 @@ bool IRValidation::Run(IREmitter* IREmit) {
|
||||
for (uint32_t i = 0; i < NumArgs; ++i) {
|
||||
OrderedNodeWrapper Arg = IROp->Args[i];
|
||||
const auto ArgID = Arg.ID();
|
||||
|
||||
// Was an argument defined after this node?
|
||||
if (ArgID >= ID) {
|
||||
HadError |= true;
|
||||
Errors << "%" << ID << ": Arg[" << i << "] has definition after use at %" << ArgID << std::endl;
|
||||
}
|
||||
|
||||
if (ArgID.IsValid() && !NodeIsLive.Get(ArgID.Value)) {
|
||||
HadError |= true;
|
||||
Errors << "%" << ID << ": Arg[" << i << "] references dead %" << ArgID << std::endl;
|
||||
}
|
||||
IROps Op = CurrentIR.GetOp<IROp_Header>(Arg)->Op;
|
||||
|
||||
if (ArgID.IsValid()) {
|
||||
Uses[ArgID.Value]++;
|
||||
}
|
||||
|
||||
// We do not validate the location of inline constants because it's
|
||||
// irrelevant, they're ignored by RA and always inlined to where they
|
||||
// need to be. This lets us pool inline constants globally.
|
||||
bool Ignore = (Op == OP_IRHEADER || Op == OP_INLINECONSTANT);
|
||||
|
||||
if (!Ignore && ArgID.IsValid() && !NodeIsLive.Get(ArgID.Value)) {
|
||||
HadError |= true;
|
||||
Errors << "%" << ID << ": Arg[" << i << "] references invalid %" << ArgID << std::endl;
|
||||
}
|
||||
}
|
||||
|
||||
NodeIsLive.Set(ID.Value);
|
||||
@@ -265,8 +267,6 @@ bool IRValidation::Run(IREmitter* IREmit) {
|
||||
Errors.clear();
|
||||
Warnings.clear();
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateIRValidation() {
|
||||
|
||||
@@ -21,7 +21,7 @@ class RAValidation;
|
||||
class IRValidation final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
~IRValidation();
|
||||
bool Run(IREmitter* IREmit) override;
|
||||
void Run(IREmitter* IREmit) override;
|
||||
|
||||
private:
|
||||
|
||||
|
||||
@@ -1,129 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: ir|opts
|
||||
desc: Removes unused arguments if known syscall number
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <memory>
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
class InlineCallOptimization final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
InlineCallOptimization(const FEXCore::CPUIDEmu* CPUID)
|
||||
: CPUID {CPUID} {}
|
||||
bool Run(IREmitter* IREmit) override;
|
||||
private:
|
||||
const FEXCore::CPUIDEmu* CPUID;
|
||||
};
|
||||
|
||||
bool InlineCallOptimization::Run(IREmitter* IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::SyscallOpt");
|
||||
|
||||
bool Changed = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
|
||||
|
||||
if (IROp->Op == FEXCore::IR::OP_SYSCALL) {
|
||||
auto Op = IROp->CW<IR::IROp_Syscall>();
|
||||
|
||||
// Is the first argument a constant?
|
||||
uint64_t Constant;
|
||||
if (IREmit->IsValueConstant(Op->SyscallID, &Constant)) {
|
||||
auto SyscallDef = Manager->SyscallHandler->GetSyscallABI(Constant);
|
||||
auto SyscallFlags = Manager->SyscallHandler->GetSyscallFlags(Constant);
|
||||
|
||||
// Update the syscall flags
|
||||
Op->Flags = SyscallFlags;
|
||||
|
||||
// XXX: Once we have the ability to do real function calls then we can call directly in to the syscall handler
|
||||
if (SyscallDef.NumArgs < FEXCore::HLE::SyscallArguments::MAX_ARGS) {
|
||||
// If the number of args are less than what the IR op supports then we can remove arg usage
|
||||
// We need +1 since we are still passing in syscall number here
|
||||
for (uint8_t Arg = (SyscallDef.NumArgs + 1); Arg < FEXCore::HLE::SyscallArguments::MAX_ARGS; ++Arg) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Arg, IREmit->Invalid());
|
||||
}
|
||||
#ifdef _M_ARM_64
|
||||
// Replace syscall with inline passthrough syscall if we can
|
||||
if (SyscallDef.HostSyscallNumber != -1) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
// Skip Args[0] since that is the syscallid
|
||||
auto InlineSyscall =
|
||||
IREmit->_InlineSyscall(CurrentIR.GetNode(IROp->Args[1]), CurrentIR.GetNode(IROp->Args[2]), CurrentIR.GetNode(IROp->Args[3]),
|
||||
CurrentIR.GetNode(IROp->Args[4]), CurrentIR.GetNode(IROp->Args[5]), CurrentIR.GetNode(IROp->Args[6]),
|
||||
SyscallDef.HostSyscallNumber, Op->Flags);
|
||||
|
||||
// Replace all syscall uses with this inline one
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, InlineSyscall);
|
||||
|
||||
// We must remove here since DCE can't remove a IROp with sideeffects
|
||||
IREmit->Remove(CodeNode);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
} else if (IROp->Op == FEXCore::IR::OP_CPUID) {
|
||||
auto Op = IROp->CW<IR::IROp_CPUID>();
|
||||
|
||||
uint64_t ConstantFunction {}, ConstantLeaf {};
|
||||
bool IsConstantFunction = IREmit->IsValueConstant(Op->Function, &ConstantFunction);
|
||||
bool IsConstantLeaf = IREmit->IsValueConstant(Op->Leaf, &ConstantLeaf);
|
||||
// If the CPUID function is constant then we can try and optimize.
|
||||
if (IsConstantFunction) { // && ConstantFunction != 1) {
|
||||
// Check if it supports constant data reporting for this function.
|
||||
const auto SupportsConstant = CPUID->DoesFunctionReportConstantData(ConstantFunction);
|
||||
if (SupportsConstant.SupportsConstantFunction == CPUIDEmu::SupportsConstant::CONSTANT) {
|
||||
// If the CPUID needs a constant leaf to be optimized then this can't work if we didn't const-prop the leaf register.
|
||||
if (!(SupportsConstant.NeedsLeaf == CPUIDEmu::NeedsLeafConstant::NEEDSLEAFCONSTANT && !IsConstantLeaf)) {
|
||||
// Calculate the constant data and replace all uses.
|
||||
// DCE will remove the CPUID IR operation.
|
||||
const auto ConstantCPUIDResult = CPUID->RunFunction(ConstantFunction, ConstantLeaf);
|
||||
uint64_t ResultsLower = (static_cast<uint64_t>(ConstantCPUIDResult.ebx) << 32) | ConstantCPUIDResult.eax;
|
||||
uint64_t ResultsUpper = (static_cast<uint64_t>(ConstantCPUIDResult.edx) << 32) | ConstantCPUIDResult.ecx;
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
auto ElementPair = IREmit->_CreateElementPair(IR::OpSize::i128Bit, IREmit->_Constant(ResultsLower), IREmit->_Constant(ResultsUpper));
|
||||
// Replace all CPUID uses with this inline one
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, ElementPair);
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
else if (IROp->Op == FEXCore::IR::OP_XGETBV) {
|
||||
auto Op = IROp->CW<IR::IROp_XGetBV>();
|
||||
|
||||
uint64_t ConstantFunction {};
|
||||
if (IREmit->IsValueConstant(Op->Function, &ConstantFunction) && CPUID->DoesXCRFunctionReportConstantData(ConstantFunction)) {
|
||||
const auto ConstantXCRResult = CPUID->RunXCRFunction(ConstantFunction);
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
auto ElementPair =
|
||||
IREmit->_CreateElementPair(IR::OpSize::i64Bit, IREmit->_Constant(ConstantXCRResult.eax), IREmit->_Constant(ConstantXCRResult.edx));
|
||||
// Replace all xgetbv uses with this inline one
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, ElementPair);
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
return Changed;
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateInlineCallOptimization(const FEXCore::CPUIDEmu* CPUID) {
|
||||
return fextl::make_unique<InlineCallOptimization>(CPUID);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -1,109 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: ir|opts
|
||||
desc: Long divide elimination pass
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <memory>
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
class LongDivideEliminationPass final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
bool Run(IREmitter* IREmit) override;
|
||||
private:
|
||||
bool IsZeroOp(IREmitter* IREmit, OrderedNodeWrapper Arg);
|
||||
bool IsSextOp(IREmitter* IREmit, OrderedNodeWrapper Lower, OrderedNodeWrapper Upper);
|
||||
};
|
||||
|
||||
bool LongDivideEliminationPass::IsZeroOp(IREmitter* IREmit, OrderedNodeWrapper Arg) {
|
||||
uint64_t Value;
|
||||
|
||||
if (IREmit->IsValueConstant(Arg, &Value)) {
|
||||
// Zero constant based zero op
|
||||
return Value == 0;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
bool LongDivideEliminationPass::IsSextOp(IREmitter* IREmit, OrderedNodeWrapper Lower, OrderedNodeWrapper Upper) {
|
||||
// We need to check if the upper source is a sext of the lower source
|
||||
auto UpperIROp = IREmit->GetOpHeader(Upper);
|
||||
if (UpperIROp->Op == OP_SBFE) {
|
||||
auto Op = UpperIROp->C<IR::IROp_Sbfe>();
|
||||
if (Op->Width == 1 && Op->lsb == 63) {
|
||||
// CQO: OrderedNode *Upper = _Sbfe(1, Size * 8 - 1, Src);
|
||||
// If the lower is the upper in this case then it can be optimized
|
||||
return Op->Header.Args[0] == Lower;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
bool LongDivideEliminationPass::Run(IREmitter* IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::LDE");
|
||||
|
||||
bool Changed = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
auto OriginalWriteCursor = IREmit->GetWriteCursor();
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
if (IROp->Size == 8) {
|
||||
if (IROp->Op == OP_LDIV || IROp->Op == OP_LREM) {
|
||||
auto Op = IROp->C<IR::IROp_LDiv>();
|
||||
// Check upper Op to see if it came from a CQO
|
||||
// CQO: OrderedNode *Upper = _Sbfe(1, Size * 8 - 1, Src);
|
||||
// If it does then it we only need a 64bit SDIV
|
||||
if (IsSextOp(IREmit, Op->Lower, Op->Upper)) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
OrderedNode* Lower = CurrentIR.GetNode(Op->Lower);
|
||||
OrderedNode* Divisor = CurrentIR.GetNode(Op->Divisor);
|
||||
OrderedNode* SDivOp {};
|
||||
if (IROp->Op == OP_LDIV) {
|
||||
SDivOp = IREmit->_Div(OpSize::i64Bit, Lower, Divisor);
|
||||
} else {
|
||||
SDivOp = IREmit->_Rem(OpSize::i64Bit, Lower, Divisor);
|
||||
}
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, SDivOp);
|
||||
Changed = true;
|
||||
}
|
||||
} else if (IROp->Op == OP_LUDIV || IROp->Op == OP_LUREM) {
|
||||
auto Op = IROp->C<IR::IROp_LUDiv>();
|
||||
// Check upper Op to see if it came from a zeroing op
|
||||
// If it does then it we only need a 64bit UDIV
|
||||
if (IsZeroOp(IREmit, Op->Upper)) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
OrderedNode* Lower = CurrentIR.GetNode(Op->Lower);
|
||||
OrderedNode* Divisor = CurrentIR.GetNode(Op->Divisor);
|
||||
OrderedNode* UDivOp {};
|
||||
if (IROp->Op == OP_LUDIV) {
|
||||
UDivOp = IREmit->_UDiv(OpSize::i64Bit, Lower, Divisor);
|
||||
} else {
|
||||
UDivOp = IREmit->_URem(OpSize::i64Bit, Lower, Divisor);
|
||||
}
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, UDivOp);
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
IREmit->SetWriteCursor(OriginalWriteCursor);
|
||||
|
||||
return Changed;
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateLongDivideEliminationPass() {
|
||||
return fextl::make_unique<LongDivideEliminationPass>();
|
||||
}
|
||||
} // namespace FEXCore::IR
|
||||
@@ -23,14 +23,12 @@ struct RegState {
|
||||
static constexpr IR::NodeID UninitializedValue {0};
|
||||
static constexpr IR::NodeID InvalidReg {0xffff'ffff};
|
||||
static constexpr IR::NodeID CorruptedPair {0xffff'fffe};
|
||||
static constexpr IR::NodeID ClobberedValue {0xffff'fffd};
|
||||
static constexpr IR::NodeID StaticAssigned {0xffff'ff00};
|
||||
|
||||
// This class makes some assumptions about how the host registers are arranged and mapped to virtual registers:
|
||||
// 1. There will be less than 32 GPRs and 32 FPRs
|
||||
// 2. If the GPRFixed class is used, there will be 16 GPRs and 16 FixedGPRs max
|
||||
// 3. Same with FPRFixed
|
||||
// 4. If the GPRPairClass is used, it is assumed each GPRPair N will map onto GPRs N*2 and N*2 + 1
|
||||
// 4. If the GPRPairClass is used, it is assumed each GPRPair N will map onto GPRs N and N + 1
|
||||
|
||||
// These assumptions were all true for the state of the arm64 and x86 jits at the time this was written
|
||||
|
||||
@@ -53,8 +51,8 @@ struct RegState {
|
||||
return true;
|
||||
case GPRPairClass:
|
||||
// Alias paired registers onto both
|
||||
GPRs[Reg.Reg * 2] = ssa;
|
||||
GPRs[Reg.Reg * 2 + 1] = ssa;
|
||||
GPRs[Reg.Reg] = ssa;
|
||||
GPRs[Reg.Reg + 1] = ssa;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
@@ -70,8 +68,8 @@ struct RegState {
|
||||
case FPRFixedClass: return FPRsFixed[Reg.Reg];
|
||||
case GPRPairClass:
|
||||
// Make sure both halves of the Pair contain the same SSA
|
||||
if (GPRs[Reg.Reg * 2] == GPRs[Reg.Reg * 2 + 1]) {
|
||||
return GPRs[Reg.Reg * 2];
|
||||
if (GPRs[Reg.Reg] == GPRs[Reg.Reg + 1]) {
|
||||
return GPRs[Reg.Reg];
|
||||
}
|
||||
return CorruptedPair;
|
||||
}
|
||||
@@ -83,91 +81,12 @@ struct RegState {
|
||||
Spills[SpillSlot] = ssa;
|
||||
}
|
||||
|
||||
// Consume (and return) the SSA id currently in a spill slot
|
||||
// Return the SSA id currently in a spill slot
|
||||
IR::NodeID Unspill(uint32_t SpillSlot) {
|
||||
if (Spills.contains(SpillSlot)) {
|
||||
const auto Value = Spills[SpillSlot];
|
||||
Spills.erase(SpillSlot);
|
||||
return Value;
|
||||
}
|
||||
return UninitializedValue;
|
||||
}
|
||||
|
||||
// Intersect another regstate with this one
|
||||
// Any registers/slots which contain the same SSA id will be persevered
|
||||
// Anything else will be marked as Clobbered
|
||||
//
|
||||
// Useful for merging two branches of control flow.
|
||||
// Any register that differs depending on control flow shouldn't be consumed by
|
||||
// code that follows
|
||||
void Intersect(RegState& other) {
|
||||
for (size_t i = 0; i < GPRs.size(); i++) {
|
||||
if (GPRs[i] != other.GPRs[i]) {
|
||||
GPRs[i] = ClobberedValue;
|
||||
}
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < GPRsFixed.size(); i++) {
|
||||
if (GPRsFixed[i] != other.GPRsFixed[i]) {
|
||||
GPRsFixed[i] = ClobberedValue;
|
||||
}
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < FPRs.size(); i++) {
|
||||
if (FPRs[i] != other.FPRs[i]) {
|
||||
FPRs[i] = ClobberedValue;
|
||||
}
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < FPRsFixed.size(); i++) {
|
||||
if (FPRsFixed[i] != other.FPRsFixed[i]) {
|
||||
FPRsFixed[i] = ClobberedValue;
|
||||
}
|
||||
}
|
||||
|
||||
for (auto it = Spills.begin(); it != Spills.end(); it++) {
|
||||
auto& [SlotID, Value] = *it;
|
||||
if (!other.Spills.contains(SlotID)) {
|
||||
Spills.erase(it);
|
||||
} else if (Value != other.Spills[SlotID]) {
|
||||
Value = ClobberedValue;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Filter out all registers/slots containing an SSA id larger than MaxSSA
|
||||
// Mark them as Clobbered.
|
||||
// Useful for backwards edges, where using an SSA from before the
|
||||
void Filter(IR::NodeID MaxSSA) {
|
||||
for (auto& gpr : GPRs) {
|
||||
if (gpr > MaxSSA) {
|
||||
gpr = ClobberedValue;
|
||||
}
|
||||
}
|
||||
|
||||
for (auto& gpr : GPRsFixed) {
|
||||
if (gpr > MaxSSA) {
|
||||
gpr = ClobberedValue;
|
||||
}
|
||||
}
|
||||
|
||||
for (auto& fpr : FPRs) {
|
||||
if (fpr > MaxSSA) {
|
||||
fpr = ClobberedValue;
|
||||
}
|
||||
}
|
||||
|
||||
for (auto& fpr : FPRsFixed) {
|
||||
if (fpr > MaxSSA) {
|
||||
fpr = ClobberedValue;
|
||||
}
|
||||
}
|
||||
|
||||
for (auto it = Spills.begin(); it != Spills.end(); it++) {
|
||||
auto& [SlotID, Value] = *it;
|
||||
if (Value > MaxSSA) {
|
||||
Spills.erase(it);
|
||||
}
|
||||
return Spills[SpillSlot];
|
||||
} else {
|
||||
return UninitializedValue;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -178,119 +97,32 @@ private:
|
||||
std::array<IR::NodeID, 32> FPRs = {};
|
||||
|
||||
fextl::unordered_map<uint32_t, IR::NodeID> Spills;
|
||||
|
||||
public:
|
||||
uint32_t Version {}; // Used to force regeneration of RegStates after following backward edges
|
||||
};
|
||||
|
||||
class RAValidation final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
~RAValidation() {}
|
||||
bool Run(IREmitter* IREmit) override;
|
||||
|
||||
private:
|
||||
// Holds the calculated RegState at the exit of each block
|
||||
fextl::unordered_map<IR::NodeID, RegState> BlockExitState;
|
||||
|
||||
// A queue of blocks we need to visit (or revisit)
|
||||
fextl::deque<OrderedNode*> BlocksToVisit;
|
||||
void Run(IREmitter* IREmit) override;
|
||||
};
|
||||
|
||||
|
||||
bool RAValidation::Run(IREmitter* IREmit) {
|
||||
void RAValidation::Run(IREmitter* IREmit) {
|
||||
if (!Manager->HasPass("RA")) {
|
||||
return false;
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::RAValidation");
|
||||
|
||||
IR::RegisterAllocationData* RAData = Manager->GetPass<IR::RegisterAllocationPass>("RA")->GetAllocationData();
|
||||
BlockExitState.clear();
|
||||
// BlocksToVisit will already be empty
|
||||
|
||||
// Get the control flow graph from the validation pass
|
||||
auto ValidationPass = Manager->GetPass<IRValidation>("IRValidation");
|
||||
LOGMAN_THROW_AA_FMT(ValidationPass != nullptr, "Couldn't find IRValidation pass");
|
||||
|
||||
auto& OffsetToBlockMap = ValidationPass->OffsetToBlockMap;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ValidationPass->EntryBlock != nullptr, "No entry point");
|
||||
BlocksToVisit.push_front(ValidationPass->EntryBlock); // Currently only a single entry point
|
||||
|
||||
bool HadError = false;
|
||||
fextl::ostringstream Errors;
|
||||
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
uint32_t CurrentVersion = 1; // Incremented every backwards edge
|
||||
|
||||
while (!BlocksToVisit.empty()) {
|
||||
auto BlockNode = BlocksToVisit.front();
|
||||
const auto BlockID = CurrentIR.GetID(BlockNode);
|
||||
auto& BlockInfo = OffsetToBlockMap[BlockID];
|
||||
|
||||
const auto IsFowardsEdge = [&](IR::NodeID PredecessorID) {
|
||||
// Blocks are sorted in FEXes IR, so backwards edges always go to a lower (or equal) Block ID
|
||||
return PredecessorID < BlockID;
|
||||
};
|
||||
|
||||
// First, make sure we have the exit state for all Predecessors that
|
||||
// get here via a forwards branch.
|
||||
bool MissingPredecessor = false;
|
||||
|
||||
for (auto Predecessor : BlockInfo.Predecessors) {
|
||||
const auto PredecessorID = CurrentIR.GetID(Predecessor);
|
||||
const bool HaveState = BlockExitState.contains(PredecessorID) && BlockExitState[PredecessorID].Version == CurrentVersion;
|
||||
|
||||
if (IsFowardsEdge(PredecessorID) && !HaveState) {
|
||||
// We are probably about to visit this node anyway, remove it
|
||||
std::remove(BlocksToVisit.begin(), BlocksToVisit.end(), Predecessor);
|
||||
|
||||
// Add the missing predecessor to start of queue
|
||||
BlocksToVisit.push_front(Predecessor);
|
||||
MissingPredecessor = true;
|
||||
}
|
||||
}
|
||||
|
||||
if (MissingPredecessor) {
|
||||
// We'll have to come back to this block later
|
||||
continue;
|
||||
}
|
||||
|
||||
// We have committed to processing this block
|
||||
// Remove from queue
|
||||
BlocksToVisit.pop_front();
|
||||
|
||||
bool FirstVisit = !BlockExitState.contains(BlockID);
|
||||
|
||||
// Second, we need to determine the register status as of Block entry
|
||||
auto BlockOp = CurrentIR.GetOp<IROp_CodeBlock>(BlockNode);
|
||||
const auto FirstSSA = BlockOp->Begin.ID();
|
||||
|
||||
auto& BlockRegState = BlockExitState.try_emplace(BlockID).first->second;
|
||||
bool EmptyRegState = true;
|
||||
auto Intersect = [&](RegState& Other) {
|
||||
if (EmptyRegState) {
|
||||
BlockRegState = Other;
|
||||
EmptyRegState = false;
|
||||
} else {
|
||||
BlockRegState.Intersect(Other);
|
||||
}
|
||||
};
|
||||
|
||||
for (auto Predecessor : BlockInfo.Predecessors) {
|
||||
auto PredecessorID = CurrentIR.GetID(Predecessor);
|
||||
if (BlockExitState.contains(PredecessorID)) {
|
||||
if (IsFowardsEdge(PredecessorID)) {
|
||||
Intersect(BlockExitState[PredecessorID]);
|
||||
} else {
|
||||
RegState Filtered = BlockExitState[PredecessorID];
|
||||
Filtered.Filter(FirstSSA);
|
||||
Intersect(Filtered);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Third, we need to iterate over all IR ops in the block
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
// We only allocate registers locally, so state is reset each block
|
||||
struct RegState BlockRegState = {};
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
const auto ID = CurrentIR.GetID(CodeNode);
|
||||
@@ -319,11 +151,6 @@ bool RAValidation::Run(IREmitter* IREmit) {
|
||||
HadError |= true;
|
||||
|
||||
Errors << fextl::fmt::format("%{}: Arg[{}] expects reg{} to contain %{}, but it is uninitialized\n", ID, i, PhyReg.Reg, ArgID);
|
||||
} else if (CurrentSSAAtReg == RegState::ClobberedValue) {
|
||||
HadError |= true;
|
||||
|
||||
Errors << fextl::fmt::format("%{}: Arg[{}] expects reg{} to contain %{}, but contents vary depending on control flow\n", ID, i,
|
||||
PhyReg.Reg, ArgID);
|
||||
} else if (CurrentSSAAtReg != ArgID) {
|
||||
HadError |= true;
|
||||
Errors << fextl::fmt::format("%{}: Arg[{}] expects reg{} to contain %{}, but it actually contains %{}\n", ID, i, PhyReg.Reg,
|
||||
@@ -346,20 +173,13 @@ bool RAValidation::Run(IREmitter* IREmit) {
|
||||
const auto Value = BlockRegState.Unspill(FillRegister->Slot);
|
||||
|
||||
// TODO: This only proves that the Spill has a consistent SSA value
|
||||
// In the future we need to prove it contains the correct SSA value
|
||||
// In the future we need to prove it contains the correct SSA value. For
|
||||
// this we need to analyze copies/swaps properly. As a hot fix, don't
|
||||
// compare Value with ExpectedValue.
|
||||
|
||||
if (Value == RegState::UninitializedValue) {
|
||||
HadError |= true;
|
||||
Errors << fextl::fmt::format("%{}: FillRegister expected %{} in Slot {}, but was undefined in at least one control flow path\n",
|
||||
ID, ExpectedValue, FillRegister->Slot);
|
||||
} else if (Value == RegState::ClobberedValue) {
|
||||
HadError |= true;
|
||||
Errors << fextl::fmt::format("%{}: FillRegister expected %{} in Slot {}, but contents vary depending on control flow\n", ID,
|
||||
ExpectedValue, FillRegister->Slot);
|
||||
} else if (Value != ExpectedValue) {
|
||||
HadError |= true;
|
||||
Errors << fextl::fmt::format("%{}: FillRegister expected %{} in Slot {}, but it actually contains %{}\n", ID, ExpectedValue,
|
||||
FillRegister->Slot, Value);
|
||||
Errors << fextl::fmt::format("%{}: FillRegister expected %{} in Slot {}, but was undefined\n", ID, ExpectedValue, FillRegister->Slot);
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -375,74 +195,10 @@ bool RAValidation::Run(IREmitter* IREmit) {
|
||||
}
|
||||
|
||||
// Update BlockState map
|
||||
BlockRegState.Set(RAData->GetNodeRegister(ID), ID);
|
||||
}
|
||||
|
||||
// Forth, Add successors to the queue of blocks to validate
|
||||
for (auto Successor : BlockInfo.Successors) {
|
||||
auto SuccessorID = CurrentIR.GetID(Successor);
|
||||
|
||||
// Blocks are sorted in FEXes IR, so backwards edges always go to a lower (or equal) Block ID
|
||||
bool FowardsEdge = SuccessorID > BlockID;
|
||||
|
||||
if (FowardsEdge) {
|
||||
// Always follow forwards edges, assuming it's not already on the queue
|
||||
if (std::find(BlocksToVisit.begin(), BlocksToVisit.end(), Successor) == std::end(BlocksToVisit)) {
|
||||
// Push to the back of queue so there is a higher chance all predecessors for this block are done first
|
||||
BlocksToVisit.push_back(Successor);
|
||||
}
|
||||
} else if (FirstVisit) {
|
||||
// Now that we have the block data for the backwards edge, we can visit it again and make
|
||||
// sure it (and all it's successors) are still valid.
|
||||
|
||||
// But only do this the first time we encounter each backwards edge.
|
||||
|
||||
// Push to the front of queue, so we get this re-checking done before examining future nodes.
|
||||
BlocksToVisit.push_front(Successor);
|
||||
|
||||
// Make sure states are reprocessed
|
||||
CurrentVersion++;
|
||||
if (IROp->Op != OP_SPILLREGISTER) {
|
||||
BlockRegState.Set(RAData->GetNodeRegister(ID), ID);
|
||||
}
|
||||
}
|
||||
|
||||
BlockRegState.Version = CurrentVersion;
|
||||
|
||||
if (CurrentVersion > 10000) {
|
||||
Errors << "Infinite Loop\n";
|
||||
HadError |= true;
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
const auto BlockID = CurrentIR.GetID(BlockNode);
|
||||
const auto& BlockInfo = OffsetToBlockMap[BlockID];
|
||||
|
||||
Errors << fextl::fmt::format("Block {}\n\tPredecessors: ", BlockID);
|
||||
|
||||
for (auto Predecessor : BlockInfo.Predecessors) {
|
||||
const auto PredecessorID = CurrentIR.GetID(Predecessor);
|
||||
const bool FowardsEdge = PredecessorID < BlockID;
|
||||
if (!FowardsEdge) {
|
||||
Errors << "(Backwards): ";
|
||||
}
|
||||
Errors << fextl::fmt::format("Block {} ", PredecessorID);
|
||||
}
|
||||
|
||||
Errors << "\n\tSuccessors: ";
|
||||
|
||||
for (auto Successor : BlockInfo.Successors) {
|
||||
const auto SuccessorID = CurrentIR.GetID(Successor);
|
||||
const bool FowardsEdge = SuccessorID > BlockID;
|
||||
|
||||
if (!FowardsEdge) {
|
||||
Errors << "(Backwards): ";
|
||||
}
|
||||
Errors << fextl::fmt::format("Block {} ", SuccessorID);
|
||||
}
|
||||
|
||||
Errors << "\n\n";
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (HadError) {
|
||||
@@ -454,8 +210,6 @@ bool RAValidation::Run(IREmitter* IREmit) {
|
||||
|
||||
Errors.clear();
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateRAValidation() {
|
||||
|
||||
@@ -53,16 +53,17 @@ struct FlagInfo {
|
||||
|
||||
class DeadFlagCalculationEliminination final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
bool Run(IREmitter* IREmit) override;
|
||||
void Run(IREmitter* IREmit) override;
|
||||
|
||||
private:
|
||||
FlagInfo Classify(IROp_Header* Node);
|
||||
unsigned FlagForOffset(unsigned Offset);
|
||||
unsigned FlagForReg(unsigned Reg);
|
||||
unsigned FlagsForCondClassType(CondClassType Cond);
|
||||
bool EliminateDeadCode(IREmitter* IREmit, Ref CodeNode, IROp_Header* IROp);
|
||||
};
|
||||
|
||||
unsigned DeadFlagCalculationEliminination::FlagForOffset(unsigned Offset) {
|
||||
return Offset == offsetof(FEXCore::Core::CPUState, pf_raw) ? FLAG_P : Offset == offsetof(FEXCore::Core::CPUState, af_raw) ? FLAG_A : 0;
|
||||
unsigned DeadFlagCalculationEliminination::FlagForReg(unsigned Reg) {
|
||||
return Reg == Core::CPUState::PF_AS_GREG ? FLAG_P : Reg == Core::CPUState::AF_AS_GREG ? FLAG_A : 0;
|
||||
};
|
||||
|
||||
unsigned DeadFlagCalculationEliminination::FlagsForCondClassType(CondClassType Cond) {
|
||||
@@ -283,21 +284,20 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
|
||||
|
||||
case OP_LOADREGISTER: {
|
||||
auto Op = IROp->CW<IR::IROp_LoadRegister>();
|
||||
if (Op->Class != GPRClass || Op->StaticClass != GPRFixedClass) {
|
||||
if (Op->Class != GPRClass) {
|
||||
break;
|
||||
}
|
||||
|
||||
return {.Read = FlagForOffset(Op->Offset)};
|
||||
return {.Read = FlagForReg(Op->Reg)};
|
||||
}
|
||||
|
||||
case OP_STOREREGISTER: {
|
||||
auto Op = IROp->CW<IR::IROp_StoreRegister>();
|
||||
if (Op->Class != GPRClass || Op->StaticClass != GPRFixedClass) {
|
||||
if (Op->Class != GPRClass) {
|
||||
break;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(!Op->IsPrewrite, "PF/AF writes are fixed-form");
|
||||
unsigned Flag = FlagForOffset(Op->Offset);
|
||||
unsigned Flag = FlagForReg(Op->Reg);
|
||||
|
||||
return {
|
||||
.Write = Flag,
|
||||
@@ -311,13 +311,53 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
|
||||
return {.Trivial = true};
|
||||
}
|
||||
|
||||
// General purpose dead code elimination. Returns whether flag handling should
|
||||
// be skipped (because it was removed or could not possibly affect flags).
|
||||
bool DeadFlagCalculationEliminination::EliminateDeadCode(IREmitter* IREmit, Ref CodeNode, IROp_Header* IROp) {
|
||||
// Can't remove anything used or with side effects.
|
||||
if (CodeNode->GetUses() > 0 || IR::HasSideEffects(IROp->Op)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
switch (IROp->Op) {
|
||||
case OP_SYSCALL: {
|
||||
auto Op = IROp->C<IR::IROp_Syscall>();
|
||||
if ((Op->Flags & IR::SyscallFlags::NOSIDEEFFECTS) != IR::SyscallFlags::NOSIDEEFFECTS) {
|
||||
return false;
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_INLINESYSCALL: {
|
||||
auto Op = IROp->C<IR::IROp_Syscall>();
|
||||
if ((Op->Flags & IR::SyscallFlags::NOSIDEEFFECTS) != IR::SyscallFlags::NOSIDEEFFECTS) {
|
||||
return false;
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
|
||||
// If the result of the atomic fetch is completely unused, convert it to a non-fetching atomic operation.
|
||||
case OP_ATOMICFETCHADD: IROp->Op = OP_ATOMICADD; return true;
|
||||
case OP_ATOMICFETCHSUB: IROp->Op = OP_ATOMICSUB; return true;
|
||||
case OP_ATOMICFETCHAND: IROp->Op = OP_ATOMICAND; return true;
|
||||
case OP_ATOMICFETCHCLR: IROp->Op = OP_ATOMICCLR; return true;
|
||||
case OP_ATOMICFETCHOR: IROp->Op = OP_ATOMICOR; return true;
|
||||
case OP_ATOMICFETCHXOR: IROp->Op = OP_ATOMICXOR; return true;
|
||||
case OP_ATOMICFETCHNEG: IROp->Op = OP_ATOMICNEG; return true;
|
||||
default: break;
|
||||
}
|
||||
|
||||
IREmit->Remove(CodeNode);
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief This pass removes flag calculations that will otherwise be unused INSIDE of that block
|
||||
* @brief This pass removes dead code locally.
|
||||
*/
|
||||
bool DeadFlagCalculationEliminination::Run(IREmitter* IREmit) {
|
||||
void DeadFlagCalculationEliminination::Run(IREmitter* IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::DFE");
|
||||
|
||||
bool Changed = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
@@ -339,13 +379,7 @@ bool DeadFlagCalculationEliminination::Run(IREmitter* IREmit) {
|
||||
// Optimizing flags can cause earlier flag reads to become dead but dead
|
||||
// flag reads should not impede optimiation of earlier dead flag writes.
|
||||
// We must DCE as we go to ensure we converge in a single iteration.
|
||||
//
|
||||
// TODO: This whole pass could be merged with DCE?
|
||||
bool HasSideEffects = IR::HasSideEffects(IROp->Op);
|
||||
if (!HasSideEffects && CodeNode->GetUses() == 0) {
|
||||
Changed = true;
|
||||
IREmit->Remove(CodeNode);
|
||||
} else {
|
||||
if (!EliminateDeadCode(IREmit, CodeNode, IROp)) {
|
||||
// Optimiation algorithm: For each flag written...
|
||||
//
|
||||
// If the flag has a later read (per FlagsRead), remove the flag from
|
||||
@@ -365,13 +399,11 @@ bool DeadFlagCalculationEliminination::Run(IREmitter* IREmit) {
|
||||
bool Eliminated = false;
|
||||
|
||||
if ((FlagsRead & Info.Write) == 0) {
|
||||
if (Info.CanEliminate && CodeNode->GetUses() == 0) {
|
||||
if ((Info.CanEliminate || Info.CanReplace) && CodeNode->GetUses() == 0) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Eliminated = true;
|
||||
Changed = true;
|
||||
} else if (Info.CanReplace) {
|
||||
IROp->Op = Info.Replacement;
|
||||
Changed = true;
|
||||
}
|
||||
} else {
|
||||
FlagsRead &= ~Info.Write;
|
||||
@@ -393,8 +425,6 @@ bool DeadFlagCalculationEliminination::Run(IREmitter* IREmit) {
|
||||
--CodeLast;
|
||||
}
|
||||
}
|
||||
|
||||
return Changed;
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadFlagCalculationEliminination() {
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -18,31 +18,10 @@ struct RegisterClassType;
|
||||
|
||||
class RegisterAllocationPass : public FEXCore::IR::Pass {
|
||||
public:
|
||||
bool HasFullRA() const {
|
||||
return HadFullRA;
|
||||
}
|
||||
|
||||
virtual void AllocateRegisterSet(uint32_t ClassCount) = 0;
|
||||
virtual void AddRegisters(FEXCore::IR::RegisterClassType Class, uint32_t RegisterCount) = 0;
|
||||
|
||||
/**
|
||||
* @brief Adds a conflict between the two registers (and their respective classes) so if one is allocated then the other can not be
|
||||
* allocated in the same live range.
|
||||
*
|
||||
* Conflict is added both directions, so only necessary to add a conflict one way
|
||||
*
|
||||
* ex:
|
||||
* AddRegisters(GPRClass, 2); -> {x0, x1} added to register class GPR
|
||||
* AddRegisters(PairClass, 1); -> {{x0, x1}} pair added to class GPR
|
||||
* AddRegisterConflict(GPRClass, 0, PairClass, 0); -> Make sure the pair interferes with x0
|
||||
* AddRegisterConflict(GPRClass, 1, PairClass, 1); -> Make sure the pair interferes with x1
|
||||
*/
|
||||
virtual void AddRegisterConflict(FEXCore::IR::RegisterClassType ClassConflict, uint32_t RegConflict, FEXCore::IR::RegisterClassType Class,
|
||||
uint32_t Reg) = 0;
|
||||
|
||||
/**
|
||||
* @name Inference graph handling
|
||||
* @{ */
|
||||
// Number of GPRs usable for pairs at start of GPR set. Must be even.
|
||||
uint32_t PairRegs;
|
||||
|
||||
/**
|
||||
* @brief Returns the register and class map array
|
||||
@@ -53,15 +32,6 @@ public:
|
||||
* @brief Returns and transfers ownership of the register and class map array
|
||||
*/
|
||||
virtual std::unique_ptr<RegisterAllocationData, RegisterAllocationDataDeleter> PullAllocationData() = 0;
|
||||
/** @} */
|
||||
|
||||
protected:
|
||||
bool HasSpills {};
|
||||
// Debug option to disable split slot reuse
|
||||
// Can be useful for testing if there is a bug with spill slots
|
||||
constexpr static bool ReuseSpillSlots {true};
|
||||
uint32_t SpillSlotCount {};
|
||||
bool HadFullRA {};
|
||||
};
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -1,106 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: ir|opts
|
||||
desc: Sanity Checking
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/fextl/set.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <functional>
|
||||
#include <memory>
|
||||
#include <stdint.h>
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::IR::Validation {
|
||||
class ValueDominanceValidation final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
bool Run(IREmitter* IREmit) override;
|
||||
};
|
||||
|
||||
bool ValueDominanceValidation::Run(IREmitter* IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::ValueDominanceValidation");
|
||||
|
||||
bool HadError = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
fextl::ostringstream Errors;
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
const auto CodeID = CurrentIR.GetID(CodeNode);
|
||||
|
||||
const uint8_t NumArgs = IR::GetRAArgs(IROp->Op);
|
||||
for (uint32_t i = 0; i < NumArgs; ++i) {
|
||||
if (IROp->Args[i].IsInvalid()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
// We do not validate the location of inline constants because it's
|
||||
// irrelevant, they're ignored by RA and always inlined to where they
|
||||
// need to be. This lets us pool inline constants globally.
|
||||
IROps Op = CurrentIR.GetOp<IROp_Header>(IROp->Args[i])->Op;
|
||||
if (Op == OP_IRHEADER || Op == OP_INLINECONSTANT) {
|
||||
continue;
|
||||
}
|
||||
|
||||
OrderedNodeWrapper Arg = IROp->Args[i];
|
||||
|
||||
// If the SSA argument is not defined INSIDE the block, we have
|
||||
// cross-block liveness, which we forbid in the IR to simplify RA.
|
||||
if (!(Arg.ID() >= BlockIROp->Begin.ID() && Arg.ID() < BlockIROp->Last.ID())) {
|
||||
HadError |= true;
|
||||
Errors << "Inst %" << CodeID << ": Arg[" << i << "] %" << Arg.ID() << " definition not local!" << std::endl;
|
||||
continue;
|
||||
}
|
||||
|
||||
// The SSA argument is defined INSIDE this block.
|
||||
// It must only be declared prior to this instruction
|
||||
// Eg: Valid
|
||||
// CodeBlock_1:
|
||||
// %_1 = Load
|
||||
// %_2 = Load
|
||||
// %_3 = <Op> %_1, %_2
|
||||
//
|
||||
// Eg: Invalid
|
||||
// CodeBlock_1:
|
||||
// %_1 = Load
|
||||
// %_2 = <Op> %_1, %_3
|
||||
// %_3 = Load
|
||||
if (Arg.ID() > CodeID) {
|
||||
HadError |= true;
|
||||
Errors << "Inst %" << CodeID << ": Arg[" << i << "] %" << Arg.ID() << " definition does not dominate this use!" << std::endl;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (HadError) {
|
||||
fextl::stringstream Out;
|
||||
FEXCore::IR::Dump(&Out, &CurrentIR, nullptr);
|
||||
Out << "Errors:" << std::endl << Errors.str() << std::endl;
|
||||
LogMan::Msg::EFmt("{}", Out.str());
|
||||
LOGMAN_MSG_A_FMT("Encountered IR validation Error");
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateValueDominanceValidation() {
|
||||
return fextl::make_unique<ValueDominanceValidation>();
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR::Validation
|
||||
@@ -43,7 +43,6 @@ class FEX_PACKED RegisterAllocationData {
|
||||
public:
|
||||
uint32_t SpillSlotCount {};
|
||||
uint32_t MapCount {};
|
||||
bool IsShared {false};
|
||||
PhysicalRegister Map[0];
|
||||
|
||||
PhysicalRegister GetNodeRegister(NodeID Node) const {
|
||||
@@ -67,18 +66,13 @@ public:
|
||||
stream.Write((const char*)&SpillSlotCount, sizeof(SpillSlotCount));
|
||||
stream.Write((const char*)&MapCount, sizeof(MapCount));
|
||||
// RAData (inline)
|
||||
// In file, IsShared is always set
|
||||
bool _IsShared = true;
|
||||
stream.Write((const char*)&_IsShared, sizeof(IsShared));
|
||||
stream.Write((const char*)&Map[0], sizeof(Map[0]) * MapCount);
|
||||
}
|
||||
};
|
||||
|
||||
struct RegisterAllocationDataDeleter {
|
||||
void operator()(RegisterAllocationData* r) const {
|
||||
if (!r->IsShared) {
|
||||
FEXCore::Allocator::free(r);
|
||||
}
|
||||
FEXCore::Allocator::free(r);
|
||||
}
|
||||
};
|
||||
|
||||
@@ -87,7 +81,6 @@ inline auto RegisterAllocationData::Create(uint32_t NodeCount) -> UniquePtr {
|
||||
memset(&Ret->Map[0], PhysicalRegister::Invalid().Raw, NodeCount);
|
||||
Ret->SpillSlotCount = 0;
|
||||
Ret->MapCount = NodeCount;
|
||||
Ret->IsShared = false;
|
||||
return UniquePtr {Ret};
|
||||
}
|
||||
|
||||
@@ -96,7 +89,6 @@ inline auto RegisterAllocationData::CreateCopy() const -> UniquePtr {
|
||||
memcpy((void*)©->Map[0], (void*)&Map[0], MapCount * sizeof(Map[0]));
|
||||
copy->SpillSlotCount = SpillSlotCount;
|
||||
copy->MapCount = MapCount;
|
||||
copy->IsShared = IsShared;
|
||||
return UniquePtr {copy};
|
||||
}
|
||||
|
||||
|
||||
@@ -22,13 +22,13 @@ namespace FEXCore::Utils::SpinWaitLock {
|
||||
*/
|
||||
#ifdef _M_ARM_64
|
||||
|
||||
#define LOADEXCLUSIVE(LoadExclusiveOp, RegSize) \
|
||||
#define LOADEXCLUSIVE(LoadExclusiveOp, RegSize) \
|
||||
/* Prime the exclusive monitor with the passed in address. */ \
|
||||
#LoadExclusiveOp " %" #RegSize "[Result], [%[Futex]];"
|
||||
|
||||
#define SPINLOOP_BODY(LoadAtomicOp, RegSize) \
|
||||
#define SPINLOOP_BODY(LoadAtomicOp, RegSize) \
|
||||
/* WFE will wait for either the memory to change or spurious wake-up. */ \
|
||||
"wfe;" /* Load with acquire to get the result of memory. */ \
|
||||
"wfe;" /* Load with acquire to get the result of memory. */ \
|
||||
#LoadAtomicOp " %" #RegSize "[Result], [%[Futex]]; "
|
||||
|
||||
#define SPINLOOP_WFE_LDX_8BIT LOADEXCLUSIVE(ldaxrb, w)
|
||||
|
||||
@@ -0,0 +1,152 @@
|
||||
# What is x86-TSO and what is different compared to ARM's weak memory model?
|
||||
x86's memory model is a very strictly coherent memory model that effectively mandates that all memory accesses are "atomic". While atomicity is
|
||||
actually a bit more strict, we actually need to emulate it in ARM using atomic instructions. We are also required to emulate this strictness with
|
||||
unaligned accesses, which is due to x86 CPUs allowing unaligned atomics for "free" within a cacheline. Intel also takes this a step more and allowing
|
||||
full atomics with a feature called "split-locks", AMD gains this same feature in Zen 5.
|
||||
|
||||
# Emulating loads
|
||||
Due to x86 SIB addressing, this can happen on most instructions. FEX emulates these in a variety of ways depending on features.
|
||||
Most instructions are emulated with an atomic instruction but we also implement a feature called "half-barrier" atomics for unaligned atomics.
|
||||
|
||||
## Base ARMv8.0
|
||||
- Addressing limitations
|
||||
- Register only
|
||||
This is emulated using an atomic load instruction plus a nop.
|
||||
- On unaligned access the code gets backpatched to a non-atomic load plus a memory barrier
|
||||
|
||||
## FEAT_LRCPC
|
||||
- Addressing limitations
|
||||
- Register only
|
||||
This matches the base ARMv8.0 implementation but adds new instructions that match x86-TSO behaviour, making the emulation slightly quicker.
|
||||
- On unaligned access it still gets backpatched to non-atomic load plus a memory barrier.
|
||||
|
||||
## FEAT_LRCPC2
|
||||
- Addressing limitations
|
||||
- Register plus 9-bit signed immediate (-256, 255)
|
||||
Adds some new instructions that allow immediate encoding inside of the previous LRCPC instructions
|
||||
|
||||
## FEAT_LRCPC3
|
||||
Adds a handful of GPR instructions that aren't super interesting
|
||||
|
||||
FEX doesn't currently implement these since no hardware supports it.
|
||||
|
||||
- ldapr - Post-index load for stack
|
||||
- ldiapr - Post-index load pair for stack
|
||||
- stilp - pre-index store pair for stack
|
||||
- stlr - pre-index store for stack
|
||||
|
||||
# Emulating stores
|
||||
Again due to x86 SIB addressing, this can also happen on most instructions. There are less options for FEX with this extension, so in most cases this
|
||||
just turns in to an atomic store with half-barrier backpatching for unaligned accesses
|
||||
|
||||
## FEAT_LRCPC, FEAT_LRCPC2
|
||||
Adds nothing for emulating stores
|
||||
|
||||
## FEAT_LRCPC3
|
||||
|
||||
# Emulating atomic instructions
|
||||
x86 has atomic memory operations that can do a variety of operations. For unaligned atomic operations FEX will emulate the operation inside the signal
|
||||
handler if it happens to be unaligned.
|
||||
|
||||
## CASPair - cmpxchg
|
||||
|
||||
## Base ARMv8.0
|
||||
- Addressing limitations
|
||||
- Register only
|
||||
This is emulated with a ldaxp+stlxp pair of instructions.
|
||||
|
||||
## FEAT_LSE
|
||||
- Addressing limitations
|
||||
- Register only
|
||||
Adds a new caspal instruction that does the operation almost exactly like x86.
|
||||
|
||||
## CAS - cmpxchg8b/cmpxchg16b
|
||||
|
||||
## Base ARMv8.0
|
||||
- Addressing limitations
|
||||
- Register only
|
||||
Similar to CASPair but now only uses a ldaxr+stlxr pair
|
||||
|
||||
## FEAT_LSE
|
||||
- Addressing limitations
|
||||
- Register only
|
||||
Similar to CASPair adds a new casal instruction that operates basically like x86
|
||||
|
||||
# AtomicFetch<Op>
|
||||
## Op from the following list
|
||||
- Add
|
||||
- Sub
|
||||
- And
|
||||
- CLR
|
||||
- Or
|
||||
- Xor
|
||||
- Neg
|
||||
- Swap
|
||||
|
||||
## Base ARMv8.0
|
||||
- Addressing limitations
|
||||
- Register only
|
||||
All operations get emulated with an ldaxr+stlxr+<op> instruction
|
||||
|
||||
## FEAT_LSE
|
||||
- Addressing limitations
|
||||
- Register only
|
||||
Almost all operations now have a native atomic memory operation instruction. The only outlier is atomicNeg which doesn't have an LSE equivalent and
|
||||
uses the ARMv8.0 implementation.
|
||||
|
||||
# Vector loads
|
||||
Since almost all memory accesses on x86 are TSO, this includes vector operations.
|
||||
|
||||
## Base ARMv8.0
|
||||
- Addressing limitations
|
||||
- Register plus 9-bit signed immediate (-256, 255)
|
||||
- Register plus 12-bit unsigned scaled immediate (Scaled by access size)
|
||||
Emulated using half-barriers, which means a load+dmb
|
||||
|
||||
## FEAT_LRCPC3
|
||||
- LDAP1 added for element loads. Register only address encoding
|
||||
- LDAPUR added for vector register loads, supports 9-bit simm offset
|
||||
|
||||
# Vector stores
|
||||
Just like loads, these are emulated using half-barriers
|
||||
|
||||
## Base ARMv8.0
|
||||
- Addressing limitations
|
||||
- Register plus 9-bit signed immediate (-256, 255)
|
||||
- Register plus 12-bit unsigned scaled immediate (Scaled by access size)
|
||||
Emulated using half-barriers, which means a dmb+str
|
||||
|
||||
|
||||
## FEAT_LRCPC3
|
||||
- STL1 added for element stores. Register only address encoding
|
||||
- STLUR added for vector register stores, supports 9-bit simm offset
|
||||
|
||||
# Addressing limitations depending on operating mode
|
||||
## GPR loadstores
|
||||
### TSO Emulation disabled
|
||||
- Register only (ldr/str)
|
||||
- Register + Register + scale (ldr/str)
|
||||
- Register + 9-bit simm (ldur/stru)
|
||||
- Register + 12-bit unsigned scaled imm (ldr/str)
|
||||
|
||||
### TSO Emulation enabled
|
||||
- Register only (ldar/stlr)
|
||||
- Register only (ldapr/stlr) - FEAT_LRCPC
|
||||
- Register + 9-bit simm (ldapr/stlur) - FEAT_LRCPC2
|
||||
|
||||
## Vector loadstores
|
||||
### TSO Emulation disabled
|
||||
- Register only (ldr/str)
|
||||
- Register + Register + scale (ldr/str)
|
||||
- Register + 9-bit simm (ldur/stru)
|
||||
- Register + 12-bit unsigned scaled imm (ldr/str)
|
||||
|
||||
### TSO Emulation enabled
|
||||
- Same as TSO emulation disabled due to half-barrier implementation
|
||||
|
||||
### TSO Emulation enabled (FEAT_LRCPC3)
|
||||
- Register only (ldap1/stl1) - Element loadstore
|
||||
- Register + 9-bit simm (ldapur/stlur)
|
||||
|
||||
## Atomic memory operations
|
||||
Always TSO emulation enabled, always register only.
|
||||
@@ -136,10 +136,10 @@ namespace DefaultValues {
|
||||
#define OPT_STRARRAY(group, enum, json, default) OPT_STR(group, enum, json, default)
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
} // namespace Type
|
||||
#define FEX_CONFIG_OPT(name, enum) \
|
||||
#define FEX_CONFIG_OPT(name, enum) \
|
||||
FEXCore::Config::Value<FEXCore::Config::DefaultValues::Type::enum> name { \
|
||||
FEXCore::Config::CONFIG_##enum, \
|
||||
FEXCore::Config::DefaultValues::enum \
|
||||
FEXCore::Config::CONFIG_##enum, \
|
||||
FEXCore::Config::DefaultValues::enum \
|
||||
}
|
||||
|
||||
#undef P
|
||||
|
||||
@@ -117,8 +117,11 @@ struct CPUState {
|
||||
// Reference counter for FEX's per-thread deferred signals.
|
||||
// Counts the nesting depth of program sections that cause signals to be deferred.
|
||||
NonAtomicRefCounter<uint64_t> DeferredSignalRefCount;
|
||||
// Since this memory region is thread local, we use NonAtomicRefCounter for fast atomic access.
|
||||
NonAtomicRefCounter<uint64_t>* DeferredSignalFaultAddress;
|
||||
|
||||
// PF/AF are statically mapped as-if they were r16/r17 (which do not exist in
|
||||
// x86 otherwise). This allows a straightforward mapping for SRA.
|
||||
static constexpr uint8_t PF_AS_GREG = 16;
|
||||
static constexpr uint8_t AF_AS_GREG = 17;
|
||||
|
||||
static constexpr size_t FLAG_SIZE = sizeof(flags[0]);
|
||||
static constexpr size_t GDT_SIZE = sizeof(gdt[0]);
|
||||
@@ -257,6 +260,8 @@ struct JITPointers {
|
||||
* @{ */
|
||||
uint64_t DispatcherLoopTop {};
|
||||
uint64_t DispatcherLoopTopFillSRA {};
|
||||
uint64_t DispatcherLoopTopEnterEC {};
|
||||
uint64_t DispatcherLoopTopEnterECFillSRA {};
|
||||
uint64_t ExitFunctionLinker {};
|
||||
uint64_t ThreadStopHandlerSpillSRA {};
|
||||
uint64_t ThreadPauseHandlerSpillSRA {};
|
||||
|
||||
@@ -22,7 +22,7 @@ class RegisterAllocationData;
|
||||
enum class SyscallFlags : uint8_t {
|
||||
DEFAULT = 0,
|
||||
// Syscalldoesn't care about CPUState being serialized up to the syscall instruction.
|
||||
// Means DeadCodeElimination can optimize through a syscall operation.
|
||||
// Means dead code elimination can optimize through a syscall operation.
|
||||
OPTIMIZETHROUGH = 1 << 0,
|
||||
// Syscall only reads the passed in arguments. Doesn't read CPUState.
|
||||
NOSYNCSTATEONENTRY = 1 << 1,
|
||||
|
||||
@@ -1,6 +1,8 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/EnumOperators.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#ifndef ENABLE_JEMALLOC
|
||||
#include <stdlib.h>
|
||||
@@ -38,6 +40,14 @@ FEX_DEFAULT_VISIBILITY JEMALLOC_NOTHROW extern void* je_aligned_alloc(size_t a,
|
||||
}
|
||||
|
||||
namespace FEXCore::Allocator {
|
||||
enum class ProtectOptions : uint32_t {
|
||||
None = 0,
|
||||
Read = (1U << 0),
|
||||
Write = (1U << 1),
|
||||
Exec = (1U << 2),
|
||||
};
|
||||
FEX_DEF_NUM_OPS(ProtectOptions)
|
||||
|
||||
#ifdef _WIN32
|
||||
inline void* VirtualAlloc(void* Base, size_t Size, bool Execute = false) {
|
||||
#ifdef _M_ARM_64EC
|
||||
@@ -66,6 +76,26 @@ inline void VirtualDontNeed(void* Ptr, size_t Size) {
|
||||
::VirtualAlloc(Ptr, Size, MEM_RESET, PAGE_NOACCESS);
|
||||
}
|
||||
|
||||
inline bool VirtualProtect(void* Ptr, size_t Size, ProtectOptions options) {
|
||||
DWORD prot {PAGE_NOACCESS};
|
||||
|
||||
if (options == ProtectOptions::None) {
|
||||
prot = PAGE_NOACCESS;
|
||||
} else if (options == ProtectOptions::Read) {
|
||||
prot = PAGE_READONLY;
|
||||
} else if (options == (ProtectOptions::Read | ProtectOptions::Write)) {
|
||||
prot = PAGE_READWRITE;
|
||||
} else if (options == (ProtectOptions::Read | ProtectOptions::Exec)) {
|
||||
prot = PAGE_EXECUTE_READ;
|
||||
} else if (options == (ProtectOptions::Read | ProtectOptions::Write | ProtectOptions::Exec)) {
|
||||
prot = PAGE_EXECUTE_READWRITE;
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unknown VirtualProtect options combination");
|
||||
}
|
||||
|
||||
return ::VirtualProtect(Ptr, Size, prot, nullptr) == 0;
|
||||
}
|
||||
|
||||
#else
|
||||
using MMAP_Hook = void* (*)(void*, size_t, int, int, int, off_t);
|
||||
using MUNMAP_Hook = int (*)(void*, size_t);
|
||||
@@ -87,6 +117,21 @@ inline void VirtualFree(void* Ptr, size_t Size) {
|
||||
inline void VirtualDontNeed(void* Ptr, size_t Size) {
|
||||
::madvise(reinterpret_cast<void*>(Ptr), Size, MADV_DONTNEED);
|
||||
}
|
||||
inline bool VirtualProtect(void* Ptr, size_t Size, ProtectOptions options) {
|
||||
int prot {PROT_NONE};
|
||||
if ((options & ProtectOptions::Read) == ProtectOptions::Read) {
|
||||
prot |= PROT_READ;
|
||||
}
|
||||
if ((options & ProtectOptions::Write) == ProtectOptions::Write) {
|
||||
prot |= PROT_WRITE;
|
||||
}
|
||||
if ((options & ProtectOptions::Exec) == ProtectOptions::Exec) {
|
||||
prot |= PROT_EXEC;
|
||||
}
|
||||
|
||||
return ::mprotect(Ptr, Size, prot) == 0;
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
// Memory allocation routines aliased to jemalloc functions.
|
||||
|
||||
@@ -1,27 +1,28 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <type_traits>
|
||||
|
||||
#define FEX_DEF_ENUM_CLASS_BIN_OP(Enum, Op) \
|
||||
[[maybe_unused]] static constexpr Enum operator Op(Enum lhs, Enum rhs) { \
|
||||
using Type = std::underlying_type_t<Enum>; \
|
||||
Type _lhs = static_cast<Type>(lhs); \
|
||||
Type _rhs = static_cast<Type>(rhs); \
|
||||
return static_cast<Enum>(_lhs Op _rhs); \
|
||||
} \
|
||||
#define FEX_DEF_ENUM_CLASS_BIN_OP(Enum, Op) \
|
||||
[[maybe_unused]] static constexpr Enum operator Op(Enum lhs, Enum rhs) { \
|
||||
using Type = std::underlying_type_t<Enum>; \
|
||||
Type _lhs = static_cast<Type>(lhs); \
|
||||
Type _rhs = static_cast<Type>(rhs); \
|
||||
return static_cast<Enum>(_lhs Op _rhs); \
|
||||
} \
|
||||
[[maybe_unused]] static constexpr uint64_t operator Op(uint64_t lhs, Enum rhs) { \
|
||||
using Type = std::underlying_type_t<Enum>; \
|
||||
Type _rhs = static_cast<Type>(rhs); \
|
||||
return lhs Op _rhs; \
|
||||
using Type = std::underlying_type_t<Enum>; \
|
||||
Type _rhs = static_cast<Type>(rhs); \
|
||||
return lhs Op _rhs; \
|
||||
}
|
||||
|
||||
#define FEX_DEF_ENUM_CLASS_UNARY_OP(Enum, Op) \
|
||||
#define FEX_DEF_ENUM_CLASS_UNARY_OP(Enum, Op) \
|
||||
[[maybe_unused]] static constexpr Enum operator Op(Enum rhs) { \
|
||||
using Type = std::underlying_type_t<Enum>; \
|
||||
Type _rhs = static_cast<Type>(rhs); \
|
||||
return static_cast<Enum>(Op _rhs); \
|
||||
using Type = std::underlying_type_t<Enum>; \
|
||||
Type _rhs = static_cast<Type>(rhs); \
|
||||
return static_cast<Enum>(Op _rhs); \
|
||||
}
|
||||
|
||||
#define FEX_DEF_NUM_OPS(Enum) \
|
||||
#define FEX_DEF_NUM_OPS(Enum) \
|
||||
FEX_DEF_ENUM_CLASS_BIN_OP(Enum, |) \
|
||||
FEX_DEF_ENUM_CLASS_BIN_OP(Enum, &) \
|
||||
FEX_DEF_ENUM_CLASS_BIN_OP(Enum, ^) \
|
||||
|
||||
@@ -10,52 +10,52 @@ namespace FEXCore {
|
||||
// Macro that defines all of the built in operators for conveniently using
|
||||
// enum classes as flag types without needing to define all of the basic
|
||||
// boilerplate.
|
||||
#define FEX_DECLARE_ENUM_FLAG_OPERATORS(type) \
|
||||
[[nodiscard]] \
|
||||
constexpr type \
|
||||
operator|(type a, type b) noexcept { \
|
||||
using T = std::underlying_type_t<type>; \
|
||||
#define FEX_DECLARE_ENUM_FLAG_OPERATORS(type) \
|
||||
[[nodiscard]] \
|
||||
constexpr type \
|
||||
operator|(type a, type b) noexcept { \
|
||||
using T = std::underlying_type_t<type>; \
|
||||
return static_cast<type>(static_cast<T>(a) | static_cast<T>(b)); \
|
||||
} \
|
||||
[[nodiscard]] \
|
||||
constexpr type \
|
||||
operator&(type a, type b) noexcept { \
|
||||
using T = std::underlying_type_t<type>; \
|
||||
} \
|
||||
[[nodiscard]] \
|
||||
constexpr type \
|
||||
operator&(type a, type b) noexcept { \
|
||||
using T = std::underlying_type_t<type>; \
|
||||
return static_cast<type>(static_cast<T>(a) & static_cast<T>(b)); \
|
||||
} \
|
||||
[[nodiscard]] \
|
||||
constexpr type \
|
||||
operator^(type a, type b) noexcept { \
|
||||
using T = std::underlying_type_t<type>; \
|
||||
} \
|
||||
[[nodiscard]] \
|
||||
constexpr type \
|
||||
operator^(type a, type b) noexcept { \
|
||||
using T = std::underlying_type_t<type>; \
|
||||
return static_cast<type>(static_cast<T>(a) ^ static_cast<T>(b)); \
|
||||
} \
|
||||
constexpr type& operator|=(type& a, type b) noexcept { \
|
||||
a = a | b; \
|
||||
return a; \
|
||||
} \
|
||||
constexpr type& operator&=(type& a, type b) noexcept { \
|
||||
a = a & b; \
|
||||
return a; \
|
||||
} \
|
||||
constexpr type& operator^=(type& a, type b) noexcept { \
|
||||
a = a ^ b; \
|
||||
return a; \
|
||||
} \
|
||||
[[nodiscard]] \
|
||||
constexpr type \
|
||||
operator~(type key) noexcept { \
|
||||
using T = std::underlying_type_t<type>; \
|
||||
return static_cast<type>(~static_cast<T>(key)); \
|
||||
} \
|
||||
[[nodiscard]] \
|
||||
constexpr bool True(type key) noexcept { \
|
||||
using T = std::underlying_type_t<type>; \
|
||||
return static_cast<T>(key) != 0; \
|
||||
} \
|
||||
[[nodiscard]] \
|
||||
constexpr bool False(type key) noexcept { \
|
||||
using T = std::underlying_type_t<type>; \
|
||||
return static_cast<T>(key) == 0; \
|
||||
} \
|
||||
constexpr type& operator|=(type& a, type b) noexcept { \
|
||||
a = a | b; \
|
||||
return a; \
|
||||
} \
|
||||
constexpr type& operator&=(type& a, type b) noexcept { \
|
||||
a = a & b; \
|
||||
return a; \
|
||||
} \
|
||||
constexpr type& operator^=(type& a, type b) noexcept { \
|
||||
a = a ^ b; \
|
||||
return a; \
|
||||
} \
|
||||
[[nodiscard]] \
|
||||
constexpr type \
|
||||
operator~(type key) noexcept { \
|
||||
using T = std::underlying_type_t<type>; \
|
||||
return static_cast<type>(~static_cast<T>(key)); \
|
||||
} \
|
||||
[[nodiscard]] \
|
||||
constexpr bool True(type key) noexcept { \
|
||||
using T = std::underlying_type_t<type>; \
|
||||
return static_cast<T>(key) != 0; \
|
||||
} \
|
||||
[[nodiscard]] \
|
||||
constexpr bool False(type key) noexcept { \
|
||||
using T = std::underlying_type_t<type>; \
|
||||
return static_cast<T>(key) == 0; \
|
||||
}
|
||||
|
||||
// Equivalent to C++23's std::to_underlying.
|
||||
|
||||
@@ -63,25 +63,25 @@ namespace Throw {
|
||||
MFmt(fmt, fmt::make_format_args(args...));
|
||||
}
|
||||
|
||||
#define LOGMAN_THROW_A_FMT(pred, ...) \
|
||||
do { \
|
||||
#define LOGMAN_THROW_A_FMT(pred, ...) \
|
||||
do { \
|
||||
LogMan::Throw::AFmt(pred, __VA_ARGS__); \
|
||||
} while (0)
|
||||
#define LOGMAN_THROW_AA_FMT(pred, ...) \
|
||||
do { \
|
||||
#define LOGMAN_THROW_AA_FMT(pred, ...) \
|
||||
do { \
|
||||
LogMan::Throw::AFmt(pred, __VA_ARGS__); \
|
||||
} while (0)
|
||||
#else
|
||||
static inline void AFmt(bool, const char*, ...) {}
|
||||
#define LOGMAN_THROW_A_FMT(pred, ...) \
|
||||
do { \
|
||||
do { \
|
||||
} while (0)
|
||||
static inline void AAFmt(bool pred, const char*, ...) {
|
||||
__builtin_assume(pred);
|
||||
}
|
||||
#define LOGMAN_THROW_AA_FMT(pred, ...) \
|
||||
do { \
|
||||
__builtin_assume(pred); \
|
||||
do { \
|
||||
__builtin_assume(pred); \
|
||||
} while (0)
|
||||
#endif
|
||||
|
||||
@@ -144,31 +144,31 @@ namespace Msg {
|
||||
MFmtImpl(ASSERT, fmt, fmt::make_format_args(args...));
|
||||
FEX_TRAP_EXECUTION;
|
||||
}
|
||||
#define LOGMAN_MSG_A_FMT(...) \
|
||||
do { \
|
||||
#define LOGMAN_MSG_A_FMT(...) \
|
||||
do { \
|
||||
LogMan::Msg::AFmt(__VA_ARGS__); \
|
||||
} while (0)
|
||||
#else
|
||||
template<typename... Args>
|
||||
static inline void AFmt(const char*, const Args&...) {}
|
||||
#define LOGMAN_MSG_A_FMT(...) \
|
||||
do { \
|
||||
do { \
|
||||
} while (0)
|
||||
#endif
|
||||
|
||||
#define WARN_ONCE_FMT(...) \
|
||||
do { \
|
||||
static bool Warned {}; \
|
||||
if (!Warned) { \
|
||||
#define WARN_ONCE_FMT(...) \
|
||||
do { \
|
||||
static bool Warned {}; \
|
||||
if (!Warned) { \
|
||||
LogMan::Msg::DFmt(__VA_ARGS__); \
|
||||
Warned = true; \
|
||||
} \
|
||||
Warned = true; \
|
||||
} \
|
||||
} while (0);
|
||||
|
||||
#define ERROR_AND_DIE_FMT(...) \
|
||||
do { \
|
||||
#define ERROR_AND_DIE_FMT(...) \
|
||||
do { \
|
||||
LogMan::Msg::EFmt(__VA_ARGS__); \
|
||||
FEX_TRAP_EXECUTION; \
|
||||
FEX_TRAP_EXECUTION; \
|
||||
} while (0)
|
||||
|
||||
} // namespace Msg
|
||||
|
||||
@@ -54,10 +54,10 @@ static void TraceObject(std::string_view const Format) {}
|
||||
static void TraceObject(std::string_view const, uint64_t) {}
|
||||
|
||||
#define FEXCORE_PROFILE_INSTANT(...) \
|
||||
do { \
|
||||
do { \
|
||||
} while (0)
|
||||
#define FEXCORE_PROFILE_SCOPED(...) \
|
||||
do { \
|
||||
do { \
|
||||
} while (0)
|
||||
#endif
|
||||
} // namespace FEXCore::Profiler
|
||||
@@ -137,11 +137,13 @@ public:
|
||||
// X86-64 must do an additional check around the store.
|
||||
if ((Result - 1) == 0) {
|
||||
// Must happen after the refcount store
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
|
||||
auto InterruptFaultPage = reinterpret_cast<Core::NonAtomicRefCounter<uint64_t>*>(&Thread->InterruptFaultPage);
|
||||
InterruptFaultPage->Store(0);
|
||||
}
|
||||
#else
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
|
||||
auto InterruptFaultPage = reinterpret_cast<Core::NonAtomicRefCounter<uint64_t>*>(&Thread->InterruptFaultPage);
|
||||
InterruptFaultPage->Store(0);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
@@ -92,16 +92,16 @@ static inline void Shutdown(const fextl::string& ApplicationName) {}
|
||||
#define FEXCORE_TELEMETRY_STATIC_INIT(Name, Type)
|
||||
#define FEXCORE_TELEMETRY_INIT(Name, Type)
|
||||
#define FEXCORE_TELEMETRY(Name, Value) \
|
||||
do { \
|
||||
do { \
|
||||
} while (0)
|
||||
#define FEXCORE_TELEMETRY_SET(Name, Value) \
|
||||
do { \
|
||||
do { \
|
||||
} while (0)
|
||||
#define FEXCORE_TELEMETRY_OR(Name, Value) \
|
||||
do { \
|
||||
do { \
|
||||
} while (0)
|
||||
#define FEXCORE_TELEMETRY_INC(Name) \
|
||||
do { \
|
||||
do { \
|
||||
} while (0)
|
||||
#define FEXCORE_TELEMETRY_Addr(Name) reinterpret_cast<std::atomic<uint64_t>*>(nullptr)
|
||||
#endif
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
#include <fcntl.h>
|
||||
|
||||
using namespace FEXCore::ARMEmitter;
|
||||
using namespace ARMEmitter;
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: PC relative") {
|
||||
{
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
#include <fcntl.h>
|
||||
|
||||
using namespace FEXCore::ARMEmitter;
|
||||
using namespace ARMEmitter;
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ASIMD: Cryptographic AES") {
|
||||
if (false) {
|
||||
@@ -2091,14 +2091,19 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ASIMD: Advanced SIMD three same") {
|
||||
TEST_SINGLE(bif(DReg::d30, DReg::d29, DReg::d28), "bif v30.8b, v29.8b, v28.8b");
|
||||
}
|
||||
|
||||
#if TEST_FP16
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ASIMD: Advanced SIMD modified immediate : fp16") {
|
||||
TEST_SINGLE(fmov(SubRegSize::i16Bit, QReg::q30, 1.0), "fmov v30.8h, #0x70 (1.0000)");
|
||||
TEST_SINGLE(fmov(SubRegSize::i16Bit, DReg::d30, 1.0), "fmov v30.4h, #0x70 (1.0000)");
|
||||
}
|
||||
#endif
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ASIMD: Advanced SIMD modified immediate") {
|
||||
// XXX: ORR - 32-bit/16-bit
|
||||
// XXX: MOVI - Shifting ones
|
||||
TEST_SINGLE(fmov(SubRegSize::i16Bit, QReg::q30, 1.0), "fmov v30.8h, #0x70 (1.0000)");
|
||||
TEST_SINGLE(fmov(SubRegSize::i32Bit, QReg::q30, 1.0), "fmov v30.4s, #0x70 (1.0000)");
|
||||
TEST_SINGLE(fmov(SubRegSize::i64Bit, QReg::q30, 1.0), "fmov v30.2d, #0x70 (1.0000)");
|
||||
|
||||
TEST_SINGLE(fmov(SubRegSize::i16Bit, DReg::d30, 1.0), "fmov v30.4h, #0x70 (1.0000)");
|
||||
TEST_SINGLE(fmov(SubRegSize::i32Bit, DReg::d30, 1.0), "fmov v30.2s, #0x70 (1.0000)");
|
||||
// TEST_SINGLE(fmov(SubRegSize::i64Bit, DReg::d30, 1.0), "fmov v30.1d, #0x70 (1.0000)");
|
||||
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
#include <fcntl.h>
|
||||
|
||||
using namespace FEXCore::ARMEmitter;
|
||||
using namespace ARMEmitter;
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: Branch: Conditional branch immediate") {
|
||||
{
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
#include <fcntl.h>
|
||||
|
||||
using namespace FEXCore::ARMEmitter;
|
||||
using namespace ARMEmitter;
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Compare and swap pair") {
|
||||
TEST_SINGLE(casp(Size::i32Bit, Reg::r28, Reg::r29, Reg::r26, Reg::r27, Reg::r30), "casp w28, w29, w26, w27, [x30]");
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
#include <fcntl.h>
|
||||
|
||||
using namespace FEXCore::ARMEmitter;
|
||||
using namespace ARMEmitter;
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: Base Encodings") {
|
||||
TEST_SINGLE(dup(SubRegSize::i8Bit, ZReg::z30, ZReg::z29, 0), "mov z30.b, b29");
|
||||
@@ -287,11 +287,11 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE floating-point convert pre
|
||||
TEST_SINGLE(fcvtlt(SubRegSize::i64Bit, ZReg::z30, PReg::p6.Merging(), ZReg::z29), "fcvtlt z30.d, p6/m, z29.s");
|
||||
|
||||
|
||||
// void fcvtxnt(FEXCore::ARMEmitter::ZRegister zd, FEXCore::ARMEmitter::PRegister pg, FEXCore::ARMEmitter::ZRegister zn) {
|
||||
// void fcvtxnt(ARMEmitter::ZRegister zd, ARMEmitter::PRegister pg, ARMEmitter::ZRegister zn) {
|
||||
/////< Size is destination size
|
||||
// void fcvtnt(FEXCore::ARMEmitter::SubRegSize size, FEXCore::ARMEmitter::ZRegister zd, FEXCore::ARMEmitter::PRegister pg, FEXCore::ARMEmitter::ZRegister zn) {
|
||||
// void fcvtnt(ARMEmitter::SubRegSize size, ARMEmitter::ZRegister zd, ARMEmitter::PRegister pg, ARMEmitter::ZRegister zn) {
|
||||
/////< Size is destination size
|
||||
// void fcvtlt(FEXCore::ARMEmitter::SubRegSize size, FEXCore::ARMEmitter::ZRegister zd, FEXCore::ARMEmitter::PRegister pg, FEXCore::ARMEmitter::ZRegister zn) {
|
||||
// void fcvtlt(ARMEmitter::SubRegSize size, ARMEmitter::ZRegister zd, ARMEmitter::PRegister pg, ARMEmitter::ZRegister zn) {
|
||||
|
||||
// XXX: BFCVTNT
|
||||
}
|
||||
@@ -1634,12 +1634,12 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE conditionally extract elem
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE Permute Vector - Extract") {
|
||||
TEST_SINGLE(ext<FEXCore::ARMEmitter::OpType::Destructive>(ZReg::z30, ZReg::z30, ZReg::z29, 0), "ext z30.b, z30.b, z29.b, #0");
|
||||
TEST_SINGLE(ext<FEXCore::ARMEmitter::OpType::Destructive>(ZReg::z30, ZReg::z30, ZReg::z29, 255), "ext z30.b, z30.b, z29.b, #255");
|
||||
TEST_SINGLE(ext<ARMEmitter::OpType::Destructive>(ZReg::z30, ZReg::z30, ZReg::z29, 0), "ext z30.b, z30.b, z29.b, #0");
|
||||
TEST_SINGLE(ext<ARMEmitter::OpType::Destructive>(ZReg::z30, ZReg::z30, ZReg::z29, 255), "ext z30.b, z30.b, z29.b, #255");
|
||||
|
||||
TEST_SINGLE(ext<FEXCore::ARMEmitter::OpType::Constructive>(ZReg::z30, ZReg::z28, ZReg::z29, 0), "ext z30.b, {z28.b, z29.b}, #0");
|
||||
TEST_SINGLE(ext<FEXCore::ARMEmitter::OpType::Constructive>(ZReg::z30, ZReg::z28, ZReg::z29, 255), "ext z30.b, {z28.b, z29.b}, #255");
|
||||
TEST_SINGLE(ext<FEXCore::ARMEmitter::OpType::Constructive>(ZReg::z30, ZReg::z31, ZReg::z0, 255), "ext z30.b, {z31.b, z0.b}, #255");
|
||||
TEST_SINGLE(ext<ARMEmitter::OpType::Constructive>(ZReg::z30, ZReg::z28, ZReg::z29, 0), "ext z30.b, {z28.b, z29.b}, #0");
|
||||
TEST_SINGLE(ext<ARMEmitter::OpType::Constructive>(ZReg::z30, ZReg::z28, ZReg::z29, 255), "ext z30.b, {z28.b, z29.b}, #255");
|
||||
TEST_SINGLE(ext<ARMEmitter::OpType::Constructive>(ZReg::z30, ZReg::z31, ZReg::z0, 255), "ext z30.b, {z31.b, z0.b}, #255");
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE permute vector segments") {
|
||||
@@ -2335,70 +2335,78 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE broadcast integer immediat
|
||||
TEST_SINGLE(mov_imm(SubRegSize::i64Bit, ZReg::z30, 127), "mov z30.d, #127");
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE broadcast floating-point immediate (predicated)") {
|
||||
#if TEST_FP16
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE broadcast floating-point immediate (predicated) : fp16") {
|
||||
TEST_SINGLE(fcpy(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Merging(), -0.125), "fmov z30.h, p6/m, #0xc0 (-0.1250)");
|
||||
TEST_SINGLE(fcpy(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Merging(), 0.5), "fmov z30.h, p6/m, #0x60 (0.5000)");
|
||||
TEST_SINGLE(fcpy(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Merging(), 1.0), "fmov z30.h, p6/m, #0x70 (1.0000)");
|
||||
TEST_SINGLE(fcpy(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Merging(), 31.0), "fmov z30.h, p6/m, #0x3f (31.0000)");
|
||||
TEST_SINGLE(fmov(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Merging(), -0.125), "fmov z30.h, p6/m, #0xc0 (-0.1250)");
|
||||
TEST_SINGLE(fmov(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Merging(), 0.5), "fmov z30.h, p6/m, #0x60 (0.5000)");
|
||||
TEST_SINGLE(fmov(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Merging(), 1.0), "fmov z30.h, p6/m, #0x70 (1.0000)");
|
||||
TEST_SINGLE(fmov(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Merging(), 31.0), "fmov z30.h, p6/m, #0x3f (31.0000)");
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE broadcast floating-point immediate (unpredicated)") {
|
||||
TEST_SINGLE(fdup(SubRegSize::i16Bit, ZReg::z30, -0.125), "fmov z30.h, #0xc0 (-0.1250)");
|
||||
TEST_SINGLE(fdup(SubRegSize::i16Bit, ZReg::z30, 0.5), "fmov z30.h, #0x60 (0.5000)");
|
||||
TEST_SINGLE(fdup(SubRegSize::i16Bit, ZReg::z30, 1.0), "fmov z30.h, #0x70 (1.0000)");
|
||||
TEST_SINGLE(fdup(SubRegSize::i16Bit, ZReg::z30, 31.0), "fmov z30.h, #0x3f (31.0000)");
|
||||
TEST_SINGLE(fmov(SubRegSize::i16Bit, ZReg::z30, -0.125), "fmov z30.h, #0xc0 (-0.1250)");
|
||||
TEST_SINGLE(fmov(SubRegSize::i16Bit, ZReg::z30, 0.5), "fmov z30.h, #0x60 (0.5000)");
|
||||
TEST_SINGLE(fmov(SubRegSize::i16Bit, ZReg::z30, 1.0), "fmov z30.h, #0x70 (1.0000)");
|
||||
TEST_SINGLE(fmov(SubRegSize::i16Bit, ZReg::z30, 31.0), "fmov z30.h, #0x3f (31.0000)");
|
||||
}
|
||||
#endif
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE broadcast floating-point immediate (predicated)") {
|
||||
TEST_SINGLE(fcpy(SubRegSize::i32Bit, ZReg::z30, PReg::p6.Merging(), -0.125), "fmov z30.s, p6/m, #0xc0 (-0.1250)");
|
||||
TEST_SINGLE(fcpy(SubRegSize::i64Bit, ZReg::z30, PReg::p6.Merging(), -0.125), "fmov z30.d, p6/m, #0xc0 (-0.1250)");
|
||||
|
||||
TEST_SINGLE(fcpy(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Merging(), 0.5), "fmov z30.h, p6/m, #0x60 (0.5000)");
|
||||
TEST_SINGLE(fcpy(SubRegSize::i32Bit, ZReg::z30, PReg::p6.Merging(), 0.5), "fmov z30.s, p6/m, #0x60 (0.5000)");
|
||||
TEST_SINGLE(fcpy(SubRegSize::i64Bit, ZReg::z30, PReg::p6.Merging(), 0.5), "fmov z30.d, p6/m, #0x60 (0.5000)");
|
||||
|
||||
TEST_SINGLE(fcpy(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Merging(), 1.0), "fmov z30.h, p6/m, #0x70 (1.0000)");
|
||||
TEST_SINGLE(fcpy(SubRegSize::i32Bit, ZReg::z30, PReg::p6.Merging(), 1.0), "fmov z30.s, p6/m, #0x70 (1.0000)");
|
||||
TEST_SINGLE(fcpy(SubRegSize::i64Bit, ZReg::z30, PReg::p6.Merging(), 1.0), "fmov z30.d, p6/m, #0x70 (1.0000)");
|
||||
|
||||
TEST_SINGLE(fcpy(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Merging(), 31.0), "fmov z30.h, p6/m, #0x3f (31.0000)");
|
||||
TEST_SINGLE(fcpy(SubRegSize::i32Bit, ZReg::z30, PReg::p6.Merging(), 31.0), "fmov z30.s, p6/m, #0x3f (31.0000)");
|
||||
TEST_SINGLE(fcpy(SubRegSize::i64Bit, ZReg::z30, PReg::p6.Merging(), 31.0), "fmov z30.d, p6/m, #0x3f (31.0000)");
|
||||
|
||||
TEST_SINGLE(fmov(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Merging(), -0.125), "fmov z30.h, p6/m, #0xc0 (-0.1250)");
|
||||
TEST_SINGLE(fmov(SubRegSize::i32Bit, ZReg::z30, PReg::p6.Merging(), -0.125), "fmov z30.s, p6/m, #0xc0 (-0.1250)");
|
||||
TEST_SINGLE(fmov(SubRegSize::i64Bit, ZReg::z30, PReg::p6.Merging(), -0.125), "fmov z30.d, p6/m, #0xc0 (-0.1250)");
|
||||
|
||||
TEST_SINGLE(fmov(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Merging(), 0.5), "fmov z30.h, p6/m, #0x60 (0.5000)");
|
||||
TEST_SINGLE(fmov(SubRegSize::i32Bit, ZReg::z30, PReg::p6.Merging(), 0.5), "fmov z30.s, p6/m, #0x60 (0.5000)");
|
||||
TEST_SINGLE(fmov(SubRegSize::i64Bit, ZReg::z30, PReg::p6.Merging(), 0.5), "fmov z30.d, p6/m, #0x60 (0.5000)");
|
||||
|
||||
TEST_SINGLE(fmov(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Merging(), 1.0), "fmov z30.h, p6/m, #0x70 (1.0000)");
|
||||
TEST_SINGLE(fmov(SubRegSize::i32Bit, ZReg::z30, PReg::p6.Merging(), 1.0), "fmov z30.s, p6/m, #0x70 (1.0000)");
|
||||
TEST_SINGLE(fmov(SubRegSize::i64Bit, ZReg::z30, PReg::p6.Merging(), 1.0), "fmov z30.d, p6/m, #0x70 (1.0000)");
|
||||
|
||||
TEST_SINGLE(fmov(SubRegSize::i16Bit, ZReg::z30, PReg::p6.Merging(), 31.0), "fmov z30.h, p6/m, #0x3f (31.0000)");
|
||||
TEST_SINGLE(fmov(SubRegSize::i32Bit, ZReg::z30, PReg::p6.Merging(), 31.0), "fmov z30.s, p6/m, #0x3f (31.0000)");
|
||||
TEST_SINGLE(fmov(SubRegSize::i64Bit, ZReg::z30, PReg::p6.Merging(), 31.0), "fmov z30.d, p6/m, #0x3f (31.0000)");
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE broadcast floating-point immediate (unpredicated)") {
|
||||
TEST_SINGLE(fdup(SubRegSize::i16Bit, ZReg::z30, -0.125), "fmov z30.h, #0xc0 (-0.1250)");
|
||||
TEST_SINGLE(fdup(SubRegSize::i32Bit, ZReg::z30, -0.125), "fmov z30.s, #0xc0 (-0.1250)");
|
||||
TEST_SINGLE(fdup(SubRegSize::i64Bit, ZReg::z30, -0.125), "fmov z30.d, #0xc0 (-0.1250)");
|
||||
|
||||
TEST_SINGLE(fdup(SubRegSize::i16Bit, ZReg::z30, 0.5), "fmov z30.h, #0x60 (0.5000)");
|
||||
TEST_SINGLE(fdup(SubRegSize::i32Bit, ZReg::z30, 0.5), "fmov z30.s, #0x60 (0.5000)");
|
||||
TEST_SINGLE(fdup(SubRegSize::i64Bit, ZReg::z30, 0.5), "fmov z30.d, #0x60 (0.5000)");
|
||||
|
||||
TEST_SINGLE(fdup(SubRegSize::i16Bit, ZReg::z30, 1.0), "fmov z30.h, #0x70 (1.0000)");
|
||||
TEST_SINGLE(fdup(SubRegSize::i32Bit, ZReg::z30, 1.0), "fmov z30.s, #0x70 (1.0000)");
|
||||
TEST_SINGLE(fdup(SubRegSize::i64Bit, ZReg::z30, 1.0), "fmov z30.d, #0x70 (1.0000)");
|
||||
|
||||
TEST_SINGLE(fdup(SubRegSize::i16Bit, ZReg::z30, 31.0), "fmov z30.h, #0x3f (31.0000)");
|
||||
TEST_SINGLE(fdup(SubRegSize::i32Bit, ZReg::z30, 31.0), "fmov z30.s, #0x3f (31.0000)");
|
||||
TEST_SINGLE(fdup(SubRegSize::i64Bit, ZReg::z30, 31.0), "fmov z30.d, #0x3f (31.0000)");
|
||||
|
||||
TEST_SINGLE(fmov(SubRegSize::i16Bit, ZReg::z30, -0.125), "fmov z30.h, #0xc0 (-0.1250)");
|
||||
TEST_SINGLE(fmov(SubRegSize::i32Bit, ZReg::z30, -0.125), "fmov z30.s, #0xc0 (-0.1250)");
|
||||
TEST_SINGLE(fmov(SubRegSize::i64Bit, ZReg::z30, -0.125), "fmov z30.d, #0xc0 (-0.1250)");
|
||||
|
||||
TEST_SINGLE(fmov(SubRegSize::i16Bit, ZReg::z30, 0.5), "fmov z30.h, #0x60 (0.5000)");
|
||||
TEST_SINGLE(fmov(SubRegSize::i32Bit, ZReg::z30, 0.5), "fmov z30.s, #0x60 (0.5000)");
|
||||
TEST_SINGLE(fmov(SubRegSize::i64Bit, ZReg::z30, 0.5), "fmov z30.d, #0x60 (0.5000)");
|
||||
|
||||
TEST_SINGLE(fmov(SubRegSize::i16Bit, ZReg::z30, 1.0), "fmov z30.h, #0x70 (1.0000)");
|
||||
TEST_SINGLE(fmov(SubRegSize::i32Bit, ZReg::z30, 1.0), "fmov z30.s, #0x70 (1.0000)");
|
||||
TEST_SINGLE(fmov(SubRegSize::i64Bit, ZReg::z30, 1.0), "fmov z30.d, #0x70 (1.0000)");
|
||||
|
||||
TEST_SINGLE(fmov(SubRegSize::i16Bit, ZReg::z30, 31.0), "fmov z30.h, #0x3f (31.0000)");
|
||||
TEST_SINGLE(fmov(SubRegSize::i32Bit, ZReg::z30, 31.0), "fmov z30.s, #0x3f (31.0000)");
|
||||
TEST_SINGLE(fmov(SubRegSize::i64Bit, ZReg::z30, 31.0), "fmov z30.d, #0x3f (31.0000)");
|
||||
}
|
||||
|
||||
Loaded 100 of 202 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user