mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-07 18:00:17 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
a8c1a36c12 | ||
|
|
acfb24f871 | ||
|
|
ecff5aea71 | ||
|
|
74cb225ccb | ||
|
|
d5db2ccf18 | ||
|
|
cebcf50c65 | ||
|
|
3a3c9101c7 | ||
|
|
ec1c7797f4 | ||
|
|
313528c34b | ||
|
|
907fd6b04b | ||
|
|
aa0fc9071e | ||
|
|
0bd924eb7e | ||
|
|
852109c142 | ||
|
|
9b0bb29d78 | ||
|
|
c3e71de1d7 | ||
|
|
0256d6820c | ||
|
|
1b18bfaff5 | ||
|
|
6065e7a62b | ||
|
|
8aecdc536c | ||
|
|
86a2e9e655 | ||
|
|
8f50106187 | ||
|
|
02d3a319f9 | ||
|
|
d18d0435ae | ||
|
|
cdaf1c5262 | ||
|
|
4b0e3bff54 | ||
|
|
c4f7b27459 | ||
|
|
3eb8be9953 | ||
|
|
1f08f8df0d | ||
|
|
0038a0b19c | ||
|
|
166a7c7e53 | ||
|
|
8ac296bd6f | ||
|
|
e092a38e0f | ||
|
|
b145e894e4 | ||
|
|
63a8b66b28 | ||
|
|
a16cc87852 | ||
|
|
1050b60057 | ||
|
|
9188e85164 | ||
|
|
f9b369c550 | ||
|
|
949b205f42 | ||
|
|
31ee8d8178 | ||
|
|
3155590e87 | ||
|
|
feab0bce4b | ||
|
|
0599d80b13 | ||
|
|
d58e12c5e4 | ||
|
|
68939c5a5c | ||
|
|
25c4fb8508 | ||
|
|
88e6c48db7 | ||
|
|
218b0d491a | ||
|
|
bfba74dab9 | ||
|
|
5708846a07 | ||
|
|
fd28783f85 | ||
|
|
0ca34d11ad | ||
|
|
bf3275ba4a | ||
|
|
d8e4e00b2b | ||
|
|
8b38a6dd08 | ||
|
|
63f72621fc | ||
|
|
580a8c9c61 | ||
|
|
249351cef4 | ||
|
|
46690ae352 | ||
|
|
68b5a90518 | ||
|
|
cb54823622 | ||
|
|
3a2ca41724 | ||
|
|
dd4e6d29ca | ||
|
|
497ee32f59 | ||
|
|
d34c287f69 | ||
|
|
b744ca16e1 | ||
|
|
573c262bb2 | ||
|
|
ff2c2f1e1f | ||
|
|
8058391955 | ||
|
|
54dbb9f248 | ||
|
|
7d1351f402 | ||
|
|
21233430aa | ||
|
|
bd1d6820b7 | ||
|
|
18e84ae2e8 | ||
|
|
f9c29056e6 | ||
|
|
40b1c32008 | ||
|
|
b5ed804578 | ||
|
|
fbe3a86c4e | ||
|
|
498c86b47d | ||
|
|
b8b6f81c44 | ||
|
|
aab9e1b751 | ||
|
|
ec976f3f75 | ||
|
|
f169fa5da2 | ||
|
|
35ee12e7e9 | ||
|
|
4497ab8844 | ||
|
|
9f399f3313 | ||
|
|
3939213336 | ||
|
|
cc27a0f666 | ||
|
|
1ff2216063 | ||
|
|
b8165813b4 | ||
|
|
2215b153db | ||
|
|
cb91d585c3 | ||
|
|
c971d4044f | ||
|
|
fdb4a078f2 | ||
|
|
42ea711850 | ||
|
|
96721978c0 | ||
|
|
d64698e4c9 | ||
|
|
4565f2b689 | ||
|
|
5fa7f1d50d | ||
|
|
2cfc42bd6a | ||
|
|
003de2b659 | ||
|
|
4ab6c2252d | ||
|
|
8f9d818368 | ||
|
|
4ebc307744 | ||
|
|
0b519b29d9 | ||
|
|
8881e8d96e | ||
|
|
9a99608f68 | ||
|
|
26c26308db | ||
|
|
123c5d809e | ||
|
|
7c42c7798c | ||
|
|
0f98daf1d9 | ||
|
|
c3de7c63b4 | ||
|
|
ab9123a427 | ||
|
|
e504a8c979 | ||
|
|
f6dd87a3a1 | ||
|
|
89c530054d | ||
|
|
f87edbe1cb | ||
|
|
4aa477de3e | ||
|
|
694e674fe6 | ||
|
|
7ebc0f32b8 | ||
|
|
68cacc2fc3 | ||
|
|
43eb597044 | ||
|
|
7efd827e78 | ||
|
|
2f6f8b93e9 | ||
|
|
d48413cbec | ||
|
|
6bafca688b | ||
|
|
53356f1aa7 | ||
|
|
51281f6a3a | ||
|
|
7a5e08c5ab | ||
|
|
642903a7bf | ||
|
|
f51fd6c78d | ||
|
|
03cf15a9e1 | ||
|
|
fe19c04f6a | ||
|
|
b86cbda03d | ||
|
|
b1af6e23cf | ||
|
|
e26b9b12fa | ||
|
|
9bf47b3f23 | ||
|
|
2887416b2e | ||
|
|
9a92f6f743 | ||
|
|
4929480719 | ||
|
|
c8437d2303 | ||
|
|
ef25ae4d3b | ||
|
|
9399790c11 | ||
|
|
ab7f6484cf | ||
|
|
2d8c5da379 | ||
|
|
b18f148575 | ||
|
|
6426428718 | ||
|
|
af2ee426f5 | ||
|
|
6c4e9ff42d | ||
|
|
62a37a7d70 | ||
|
|
ef83addc74 | ||
|
|
fb308f5947 | ||
|
|
cedb93c11c | ||
|
|
61d2a09827 | ||
|
|
9c6423f37a | ||
|
|
e18a661b50 | ||
|
|
16e5777816 | ||
|
|
64020e8828 | ||
|
|
30798556fc | ||
|
|
cd3518d99d | ||
|
|
bc87c3d494 | ||
|
|
726228c418 | ||
|
|
929b111648 | ||
|
|
1700a73382 | ||
|
|
0f9d791911 | ||
|
|
61acc76be7 | ||
|
|
0f8ba5bf32 | ||
|
|
2cb8a96f0e | ||
|
|
1102122639 | ||
|
|
66e026a9e8 | ||
|
|
b901b42417 | ||
|
|
9f2f10a65f | ||
|
|
b0b41d00ee | ||
|
|
2d56f5eba0 | ||
|
|
8ffc6fbc6b | ||
|
|
0ffd94dadb | ||
|
|
585320093a | ||
|
|
bdd351a42c | ||
|
|
feaee702e9 | ||
|
|
d89fc84fcf | ||
|
|
688cd1a4bb | ||
|
|
c056875a00 | ||
|
|
054118139f | ||
|
|
447148d95a | ||
|
|
f6b1e42a0b | ||
|
|
90e74e5572 | ||
|
|
e4fc5fe5be | ||
|
|
809b2c6115 | ||
|
|
659538ef4b | ||
|
|
4b677f4b42 | ||
|
|
87699ea5a0 | ||
|
|
a2ae113ee9 | ||
|
|
901e2c75d4 | ||
|
|
871d140b7c | ||
|
|
2f6ae1ad02 | ||
|
|
806e98925c | ||
|
|
c7dbd2fac2 | ||
|
|
1b7729efed | ||
|
|
6d1d5aeffb | ||
|
|
9ae04b5771 | ||
|
|
fc61e3b1b5 | ||
|
|
fa1d9910e4 | ||
|
|
d425873eed | ||
|
|
d89c54dd95 | ||
|
|
5e023c55bd | ||
|
|
2487172df3 | ||
|
|
bdd078df17 | ||
|
|
db75335ad0 | ||
|
|
531ab5b5b1 | ||
|
|
0601a863f9 | ||
|
|
10d34c9564 | ||
|
|
a37d6a3841 | ||
|
|
d5234a43da | ||
|
|
b7733540c1 | ||
|
|
2ae02ded74 | ||
|
|
2e57ad644d | ||
|
|
031afbfb18 | ||
|
|
377ce2e2f6 | ||
|
|
a2fd3c077d | ||
|
|
0f5aff73ab | ||
|
|
f0b208e692 | ||
|
|
7fcb5fc590 | ||
|
|
b76f819759 | ||
|
|
4b36d4f1ea | ||
|
|
f4d0c6c807 | ||
|
|
79ab76b42e | ||
|
|
8c3ac61f8d | ||
|
|
a8bc20f76b | ||
|
|
53a55baf23 | ||
|
|
63d6800cf3 | ||
|
|
44492b4828 | ||
|
|
8ed9f8aef5 | ||
|
|
711021e24e | ||
|
|
55dea335fe | ||
|
|
7097532ddf | ||
|
|
66841ce35c | ||
|
|
3d814cb7c1 | ||
|
|
07afdca58b | ||
|
|
93ed346fd2 | ||
|
|
c9eee9bf7f |
No files matched your search
@@ -28,7 +28,7 @@ jobs:
|
||||
|
||||
- name: Get changed files
|
||||
id: changed-files
|
||||
uses: tj-actions/changed-files@v39
|
||||
uses: step-security/changed-files@3dbe17c78367e7d60f00d78ae6781a35be47b4a1 # v45.0.1
|
||||
with:
|
||||
separator: ","
|
||||
skip_initial_fetch: true
|
||||
|
||||
@@ -5,10 +5,6 @@
|
||||
[submodule "External/cpp-optparse"]
|
||||
path = Source/Common/cpp-optparse
|
||||
url = https://github.com/Sonicadvance1/cpp-optparse
|
||||
[submodule "External/xbyak"]
|
||||
shallow = true
|
||||
path = External/xbyak
|
||||
url = https://github.com/herumi/xbyak.git
|
||||
[submodule "External/fex-posixtest-bins"]
|
||||
shallow = true
|
||||
path = External/fex-posixtest-bins
|
||||
|
||||
@@ -384,10 +384,6 @@ if (TUNE_CPU STREQUAL "native")
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-mcpu=native")
|
||||
endif()
|
||||
else()
|
||||
# Due to an oversight in llvm, it declares any reasonably new Kryo CPU to only be ARMv8.0
|
||||
# Manually detect newer CPU revisions until clang and llvm fixes their bug
|
||||
# This script will either provide a supported CPU or 'native'
|
||||
# Additionally -march doesn't work under AArch64+Clang, so you have to use -mcpu or -mtune
|
||||
execute_process(COMMAND python3 "${PROJECT_SOURCE_DIR}/Scripts/aarch64_fit_native.py" "/proc/cpuinfo" "${CMAKE_CXX_COMPILER_VERSION}"
|
||||
OUTPUT_VARIABLE AARCH64_CPU)
|
||||
|
||||
|
||||
@@ -64,7 +64,7 @@ public:
|
||||
}
|
||||
void sha256h2(ARMEmitter::VRegister rd, ARMEmitter::VRegister rn, ARMEmitter::VRegister rm) {
|
||||
constexpr uint32_t Op = 0b0101'1110'0000'0000'0000'00 << 10;
|
||||
Crypto3RegSHA(Op, 0b100, rd, rn, rm);
|
||||
Crypto3RegSHA(Op, 0b101, rd, rn, rm);
|
||||
}
|
||||
void sha256su1(ARMEmitter::VRegister rd, ARMEmitter::VRegister rn, ARMEmitter::VRegister rm) {
|
||||
constexpr uint32_t Op = 0b0101'1110'0000'0000'0000'00 << 10;
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <cstring>
|
||||
#include <type_traits>
|
||||
|
||||
namespace ARMEmitter {
|
||||
class Buffer {
|
||||
@@ -21,29 +22,25 @@ public:
|
||||
Size = BaseSize;
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
requires (std::is_trivially_copyable_v<T>)
|
||||
void dcn(const T& Data) {
|
||||
std::memcpy(CurrentOffset, &Data, sizeof(Data));
|
||||
CurrentOffset += sizeof(Data);
|
||||
}
|
||||
void dc8(uint8_t Data) {
|
||||
decltype(Data)* Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
|
||||
*Memory = Data;
|
||||
CurrentOffset += sizeof(Data);
|
||||
dcn(Data);
|
||||
}
|
||||
|
||||
void dc16(uint16_t Data) {
|
||||
decltype(Data)* Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
|
||||
*Memory = Data;
|
||||
CurrentOffset += sizeof(Data);
|
||||
dcn(Data);
|
||||
}
|
||||
|
||||
void dc32(uint32_t Data) {
|
||||
decltype(Data)* Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
|
||||
*Memory = Data;
|
||||
CurrentOffset += sizeof(Data);
|
||||
dcn(Data);
|
||||
}
|
||||
void dc64(uint64_t Data) {
|
||||
dcn(Data);
|
||||
}
|
||||
|
||||
void dc64(uint64_t Data) {
|
||||
decltype(Data)* Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
|
||||
*Memory = Data;
|
||||
CurrentOffset += sizeof(Data);
|
||||
}
|
||||
void EmitString(const char* String) {
|
||||
const auto StringLength = strlen(String);
|
||||
memcpy(CurrentOffset, String, StringLength);
|
||||
|
||||
@@ -30,9 +30,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(Register) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<Register>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<Register>, "Needs to be standard");
|
||||
static_assert(sizeof(Register) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<Register>);
|
||||
static_assert(std::is_standard_layout_v<Register>);
|
||||
|
||||
/* 32-bit GPR register class.
|
||||
* This class will imply a 32-bit register size being used.
|
||||
@@ -58,9 +58,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(Register) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<Register>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<Register>, "Needs to be standard");
|
||||
static_assert(sizeof(WRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<WRegister>);
|
||||
static_assert(std::is_standard_layout_v<WRegister>);
|
||||
|
||||
/* 64-bit GPR register class.
|
||||
* This class will imply a 64-bit register size being used.
|
||||
@@ -86,9 +86,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(Register) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<Register>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<Register>, "Needs to be standard");
|
||||
static_assert(sizeof(XRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<XRegister>);
|
||||
static_assert(std::is_standard_layout_v<XRegister>);
|
||||
|
||||
inline constexpr WRegister Register::W() const {
|
||||
return WRegister {Index};
|
||||
@@ -283,9 +283,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(VRegister) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<VRegister>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<VRegister>, "Needs to be standard");
|
||||
static_assert(sizeof(VRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<VRegister>);
|
||||
static_assert(std::is_standard_layout_v<VRegister>);
|
||||
|
||||
/* 8-bit ASIMD register class
|
||||
* This class implies 8-bit scalar register.
|
||||
@@ -315,9 +315,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(BRegister) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<BRegister>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<BRegister>, "Needs to be standard");
|
||||
static_assert(sizeof(BRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<BRegister>);
|
||||
static_assert(std::is_standard_layout_v<BRegister>);
|
||||
|
||||
/* 16-bit ASIMD register class
|
||||
* This class implies 16-bit scalar register.
|
||||
@@ -347,9 +347,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(HRegister) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<HRegister>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<HRegister>, "Needs to be standard");
|
||||
static_assert(sizeof(HRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<HRegister>);
|
||||
static_assert(std::is_standard_layout_v<HRegister>);
|
||||
|
||||
/* 32-bit ASIMD register class
|
||||
* This class implies 32-bit scalar register.
|
||||
@@ -379,9 +379,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(SRegister) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<SRegister>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<SRegister>, "Needs to be standard");
|
||||
static_assert(sizeof(SRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<SRegister>);
|
||||
static_assert(std::is_standard_layout_v<SRegister>);
|
||||
|
||||
/* 64-bit ASIMD register class
|
||||
* This class doesn't imply Vector or Scalar.
|
||||
@@ -412,9 +412,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(DRegister) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<DRegister>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<DRegister>, "Needs to be standard");
|
||||
static_assert(sizeof(DRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<DRegister>);
|
||||
static_assert(std::is_standard_layout_v<DRegister>);
|
||||
|
||||
/* 128-bit ASIMD register class
|
||||
* This class doesn't imply Vector or Scalar.
|
||||
@@ -445,9 +445,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(QRegister) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<QRegister>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<QRegister>, "Needs to be standard");
|
||||
static_assert(sizeof(QRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<QRegister>);
|
||||
static_assert(std::is_standard_layout_v<QRegister>);
|
||||
|
||||
/* Unsized SVE register class.
|
||||
* This class explicitly implies the instruction will operate using SVE.
|
||||
@@ -474,9 +474,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(ZRegister) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<ZRegister>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<ZRegister>, "Needs to be standard");
|
||||
static_assert(sizeof(ZRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<ZRegister>);
|
||||
static_assert(std::is_standard_layout_v<ZRegister>);
|
||||
|
||||
// VRegister
|
||||
inline constexpr BRegister VRegister::B() const {
|
||||
@@ -919,9 +919,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(PRegister) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<PRegister>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<PRegister>, "Needs to be standard");
|
||||
static_assert(sizeof(PRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<PRegister>);
|
||||
static_assert(std::is_standard_layout_v<PRegister>);
|
||||
|
||||
// Unsized predicate register for SVE with zeroing semantics.
|
||||
class PRegisterZero {
|
||||
@@ -947,9 +947,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(PRegisterZero) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<PRegisterZero>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<PRegisterZero>, "Needs to be standard");
|
||||
static_assert(sizeof(PRegisterZero) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<PRegisterZero>);
|
||||
static_assert(std::is_standard_layout_v<PRegisterZero>);
|
||||
|
||||
// Unsized predicate register for SVE with merging semantics.
|
||||
class PRegisterMerge {
|
||||
@@ -975,9 +975,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(PRegisterZero) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<PRegisterZero>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<PRegisterZero>, "Needs to be standard");
|
||||
static_assert(sizeof(PRegisterMerge) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<PRegisterMerge>);
|
||||
static_assert(std::is_standard_layout_v<PRegisterMerge>);
|
||||
|
||||
// PRegister
|
||||
inline constexpr PRegisterZero PRegister::Zeroing() const {
|
||||
|
||||
@@ -2134,11 +2134,31 @@ public:
|
||||
|
||||
// SVE2 Crypto Extensions
|
||||
// SVE2 crypto unary operations
|
||||
// XXX:
|
||||
void aesimc(ZRegister zdn, ZRegister zn) {
|
||||
SVE2CryptoUnaryOperation(1, zdn, zn);
|
||||
}
|
||||
void aesmc(ZRegister zdn, ZRegister zn) {
|
||||
SVE2CryptoUnaryOperation(0, zdn, zn);
|
||||
}
|
||||
|
||||
// SVE2 crypto destructive binary operations
|
||||
// XXX:
|
||||
void aese(ZRegister zdn, ZRegister zn, ZRegister zm) {
|
||||
SVE2CryptoDestructiveBinaryOperation(0, 0, zdn, zn, zm);
|
||||
}
|
||||
void aesd(ZRegister zdn, ZRegister zn, ZRegister zm) {
|
||||
SVE2CryptoDestructiveBinaryOperation(0, 1, zdn, zn, zm);
|
||||
}
|
||||
void sm4e(ZRegister zdn, ZRegister zn, ZRegister zm) {
|
||||
SVE2CryptoDestructiveBinaryOperation(1, 0, zdn, zn, zm);
|
||||
}
|
||||
|
||||
// SVE2 crypto constructive binary operations
|
||||
// XXX:
|
||||
void sm4ekey(ZRegister zd, ZRegister zn, ZRegister zm) {
|
||||
SVE2CryptoConstructiveBinaryOperation(0, zd, zn, zm);
|
||||
}
|
||||
void rax1(ZRegister zd, ZRegister zn, ZRegister zm) {
|
||||
SVE2CryptoConstructiveBinaryOperation(1, zd, zn, zm);
|
||||
}
|
||||
|
||||
// SVE Floating Point Widening Multiply-Add - Indexed
|
||||
// SVE BFloat16 floating-point dot product (indexed)
|
||||
@@ -3892,6 +3912,35 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
void SVE2CryptoUnaryOperation(uint32_t op, ZRegister zdn, ZRegister zn) {
|
||||
LOGMAN_THROW_A_FMT(zdn == zn, "zdn and zn must be the same register");
|
||||
|
||||
uint32_t Instr = 0b0100'0101'0010'0000'1110'0000'0000'0000;
|
||||
Instr |= op << 10;
|
||||
Instr |= zdn.Idx();
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
void SVE2CryptoDestructiveBinaryOperation(uint32_t op, uint32_t o2, ZRegister zdn, ZRegister zn, ZRegister zm) {
|
||||
LOGMAN_THROW_A_FMT(zdn == zn, "zdn and zn must be the same register");
|
||||
|
||||
uint32_t Instr = 0b0100'0101'0010'0010'1110'0000'0000'0000;
|
||||
Instr |= op << 16;
|
||||
Instr |= o2 << 10;
|
||||
Instr |= zm.Idx() << 5;
|
||||
Instr |= zdn.Idx();
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
void SVE2CryptoConstructiveBinaryOperation(uint32_t op, ZRegister zd, ZRegister zn, ZRegister zm) {
|
||||
uint32_t Instr = 0b0100'0101'0010'0000'1111'0000'0000'0000;
|
||||
Instr |= zm.Idx() << 16;
|
||||
Instr |= op << 10;
|
||||
Instr |= zn.Idx() << 5;
|
||||
Instr |= zd.Idx();
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
void SVE2BitwisePermute(SubRegSize size, uint32_t opc, ZRegister zd, ZRegister zn, ZRegister zm) {
|
||||
LOGMAN_THROW_A_FMT(size != SubRegSize::i128Bit, "Can't use 128-bit element size");
|
||||
|
||||
@@ -5051,7 +5100,7 @@ private:
|
||||
const uint32_t element_size = SubRegSizeInBits(size);
|
||||
|
||||
if (is_left_shift) {
|
||||
LOGMAN_THROW_A_FMT(shift >= 0 && shift < element_size, "Invalid left shift value ({}). Must be within [0, {}]", shift, element_size - 1);
|
||||
LOGMAN_THROW_A_FMT(shift < element_size, "Invalid left shift value ({}). Must be within [0, {}]", shift, element_size - 1);
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(shift > 0 && shift <= element_size, "Invalid right shift value ({}). Must be within [1, {}]", shift, element_size);
|
||||
}
|
||||
|
||||
Vendored
+1
-1
Submodule External/Vulkan-Headers updated: 29f979ee5a...cacef3039d.
Vendored
+1
-1
Submodule External/fmt updated: 873670ba3f...123913715a.
Vendored
-1
Submodule External/xbyak deleted from c68cc53d18.
@@ -423,7 +423,7 @@ def print_parse_argloader_options(options):
|
||||
# these need a bit more help
|
||||
output_argloader.write("\tauto Array = Options.all(\"{0}\");\n".format(op_key))
|
||||
output_argloader.write("\tfor (auto iter = Array.begin(); iter != Array.end(); ++iter) {\n")
|
||||
output_argloader.write("\t\tSet(FEXCore::Config::ConfigOption::CONFIG_{0}, *iter);\n".format(op_key.upper()))
|
||||
output_argloader.write("\t\tAppendStrArrayValue(FEXCore::Config::ConfigOption::CONFIG_{0}, *iter);\n".format(op_key.upper()))
|
||||
output_argloader.write("\t}\n")
|
||||
else:
|
||||
if (NeedsString):
|
||||
@@ -461,14 +461,19 @@ def print_parse_jsonloader_options(options):
|
||||
output_argloader.write("#ifdef JSONLOADER\n")
|
||||
output_argloader.write("#undef JSONLOADER\n")
|
||||
output_argloader.write("if (false) {}\n")
|
||||
op_key = None
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
value_type = op_vals["Type"]
|
||||
if (value_type == "strenum"):
|
||||
output_argloader.write("else if (KeyName == \"{0}\") {{\n".format(op_key))
|
||||
output_argloader.write("Set(KeyOption, FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, Value_View));\n".format(op_key, op_key, op_key))
|
||||
output_argloader.write("\tSet(KeyOption, FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, Value_View));\n".format(op_key, op_key, op_key))
|
||||
output_argloader.write("}\n")
|
||||
|
||||
elif (value_type == "strarray"):
|
||||
output_argloader.write("else if (KeyName == \"{0}\") {{\n".format(op_key))
|
||||
output_argloader.write("\tAppendStrArrayValue(KeyOption, ConfigString);\n")
|
||||
output_argloader.write("}\n")
|
||||
assert op_key is not None, "No options found in JSONLOADER"
|
||||
output_argloader.write("else {{\n".format(op_key))
|
||||
output_argloader.write("Set(KeyOption, ConfigString);\n")
|
||||
output_argloader.write("}\n")
|
||||
|
||||
@@ -374,7 +374,7 @@ def print_ir_structs(defines):
|
||||
output_file.write("};\n")
|
||||
|
||||
# Add a static assert that the IR ops must be pod
|
||||
output_file.write("static_assert(std::is_trivial_v<IROp_{}>);\n".format(op.Name))
|
||||
output_file.write("static_assert(std::is_trivially_copyable_v<IROp_{}>);\n".format(op.Name))
|
||||
output_file.write("static_assert(std::is_standard_layout_v<IROp_{}>);\n\n".format(op.Name))
|
||||
|
||||
output_file.write("#undef IROP_STRUCTS\n")
|
||||
|
||||
@@ -24,9 +24,6 @@ set (SRCS
|
||||
Common/SoftFloat-3e/extF80_mul.c
|
||||
Common/SoftFloat-3e/extF80_rem.c
|
||||
Common/SoftFloat-3e/extF80_sqrt.c
|
||||
Common/SoftFloat-3e/s_add128.c
|
||||
Common/SoftFloat-3e/s_sub128.c
|
||||
Common/SoftFloat-3e/s_le128.c
|
||||
Common/SoftFloat-3e/extF80_to_i32.c
|
||||
Common/SoftFloat-3e/extF80_to_i64.c
|
||||
Common/SoftFloat-3e/extF80_to_ui64.c
|
||||
@@ -39,7 +36,6 @@ set (SRCS
|
||||
Common/SoftFloat-3e/s_roundToUI64.c
|
||||
Common/SoftFloat-3e/s_f128UIToCommonNaN.c
|
||||
Common/SoftFloat-3e/s_commonNaNToF128UI.c
|
||||
Common/SoftFloat-3e/s_shortShiftRight128.c
|
||||
Common/SoftFloat-3e/s_normSubnormalF128Sig.c
|
||||
Common/SoftFloat-3e/s_roundToI32.c
|
||||
Common/SoftFloat-3e/s_roundToI64.c
|
||||
@@ -48,22 +44,14 @@ set (SRCS
|
||||
Common/SoftFloat-3e/s_extF80UIToCommonNaN.c
|
||||
Common/SoftFloat-3e/s_commonNaNToF32UI.c
|
||||
Common/SoftFloat-3e/s_commonNaNToF64UI.c
|
||||
Common/SoftFloat-3e/s_shortShiftRightJam64.c
|
||||
Common/SoftFloat-3e/s_shortShiftRightJam64Extra.c
|
||||
Common/SoftFloat-3e/s_shiftRightJam64Extra.c
|
||||
Common/SoftFloat-3e/s_shortShiftRightJam64Extra.c
|
||||
Common/SoftFloat-3e/s_roundPackToF64.c
|
||||
Common/SoftFloat-3e/s_propagateNaNExtF80UI.c
|
||||
Common/SoftFloat-3e/s_roundPackToExtF80.c
|
||||
Common/SoftFloat-3e/s_normSubnormalExtF80Sig.c
|
||||
Common/SoftFloat-3e/s_shiftRightJam64.c
|
||||
Common/SoftFloat-3e/s_subMagsExtF80.c
|
||||
Common/SoftFloat-3e/s_shiftRightJam32.c
|
||||
Common/SoftFloat-3e/s_shiftRightJam128.c
|
||||
Common/SoftFloat-3e/s_shiftRightJam128Extra.c
|
||||
Common/SoftFloat-3e/s_normRoundPackToExtF80.c
|
||||
Common/SoftFloat-3e/s_shortShiftLeft128.c
|
||||
Common/SoftFloat-3e/s_approxRecip32_1.c
|
||||
Common/SoftFloat-3e/s_approxRecip_1Ks.c
|
||||
Common/SoftFloat-3e/s_approxRecipSqrt32_1.c
|
||||
Common/SoftFloat-3e/s_approxRecipSqrt_1Ks.c
|
||||
@@ -75,12 +63,6 @@ set (SRCS
|
||||
Common/SoftFloat-3e/extF80_roundToInt.c
|
||||
Common/SoftFloat-3e/extF80_eq.c
|
||||
Common/SoftFloat-3e/extF80_lt.c
|
||||
Common/SoftFloat-3e/s_lt128.c
|
||||
Common/SoftFloat-3e/s_mul64ByShifted32To128.c
|
||||
Common/SoftFloat-3e/s_mul64To128.c
|
||||
Common/SoftFloat-3e/s_countLeadingZeros8.c
|
||||
Common/SoftFloat-3e/s_countLeadingZeros32.c
|
||||
Common/SoftFloat-3e/s_countLeadingZeros64.c
|
||||
Common/SoftFloat-3e/f32_to_extF80.c
|
||||
Common/SoftFloat-3e/s_normSubnormalF32Sig.c
|
||||
Common/SoftFloat-3e/s_f32UIToCommonNaN.c
|
||||
@@ -88,6 +70,7 @@ set (SRCS
|
||||
Interface/Core/LookupCache.cpp
|
||||
Interface/Core/Core.cpp
|
||||
Interface/Core/CPUBackend.cpp
|
||||
Interface/Core/Addressing.cpp
|
||||
Interface/Core/CPUID.cpp
|
||||
Interface/Core/Frontend.cpp
|
||||
Interface/Core/ObjectCache/JobHandling.cpp
|
||||
@@ -176,7 +159,7 @@ else()
|
||||
endif()
|
||||
|
||||
# Some defines for the softfloat library
|
||||
list(APPEND DEFINES "-DSOFTFLOAT_BUILTIN_CLZ")
|
||||
list(APPEND DEFINES "-DSOFTFLOAT_BUILTIN_CLZ=1;-DINLINE=static inline;-DINLINE_LEVEL=4;-DSOFTFLOAT_FAST_INT64=1;-DSOFTFLOAT_FAST_DIV32TO16=1;-DSOFTFLOAT_FAST_DIV64TO32=1")
|
||||
|
||||
set (LIBS fmt::fmt xxHash::xxhash FEXHeaderUtils CodeEmitter)
|
||||
|
||||
|
||||
@@ -19,12 +19,24 @@ static bool Conv(std::string_view Value, uint8_t* Result) {
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, int8_t* Result) {
|
||||
*Result = std::strtol(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, uint16_t* Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, int16_t* Result) {
|
||||
*Result = std::strtol(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, uint32_t* Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
@@ -42,6 +54,13 @@ static bool Conv(std::string_view Value, uint64_t* Result) {
|
||||
*Result = std::strtoull(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, int64_t* Result) {
|
||||
*Result = std::strtoll(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
template<typename T, typename = std::enable_if<std::is_enum<T>::value, T>>
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, T* Result) {
|
||||
|
||||
@@ -19,12 +19,10 @@
|
||||
|
||||
#include <array>
|
||||
#include <cstdlib>
|
||||
#include <functional>
|
||||
#include <optional>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <string_view>
|
||||
#include <system_error>
|
||||
#include <type_traits>
|
||||
#include <utility>
|
||||
|
||||
@@ -113,16 +111,9 @@ fextl::string GetApplicationConfig(const std::string_view Program, bool Global)
|
||||
return fextl::fmt::format("{}{}.json", ConfigFile, Program);
|
||||
}
|
||||
|
||||
void SetConfig(FEXCore::Context::Context* CTX, ConfigOption Option, uint64_t Config) {}
|
||||
|
||||
void SetConfig(FEXCore::Context::Context* CTX, ConfigOption Option, const fextl::string& Config) {}
|
||||
|
||||
uint64_t GetConfig(FEXCore::Context::Context* CTX, ConfigOption Option) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
static fextl::map<FEXCore::Config::LayerType, fextl::unique_ptr<FEXCore::Config::Layer>> ConfigLayers;
|
||||
static FEXCore::Config::Layer* Meta {};
|
||||
class MetaLayer;
|
||||
static FEXCore::Config::MetaLayer* Meta {};
|
||||
|
||||
constexpr std::array<FEXCore::Config::LayerType, 10> LoadOrder = {
|
||||
FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN, FEXCore::Config::LayerType::LAYER_MAIN,
|
||||
@@ -143,9 +134,39 @@ public:
|
||||
~MetaLayer() {}
|
||||
void Load();
|
||||
|
||||
template<typename T>
|
||||
requires (!std::is_same_v<fextl::string, T> && !std::is_same_v<DefaultValues::Type::StringArrayType, T>)
|
||||
std::optional<T> GetConv(ConfigOption Option) {
|
||||
const auto it = OptionMap.find(Option);
|
||||
if (it == OptionMap.end()) {
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
const auto& Value = it->second;
|
||||
LOGMAN_THROW_A_FMT(!std::holds_alternative<DefaultValues::Type::StringArrayType>(Value), "Tried to get config of invalid type!");
|
||||
|
||||
if (std::holds_alternative<T>(Value)) [[likely]] {
|
||||
return std::get<T>(Value);
|
||||
}
|
||||
|
||||
T ConvertedValue;
|
||||
if (std::holds_alternative<fextl::string>(Value)) {
|
||||
const auto& StrVal = std::get<fextl::string>(Value);
|
||||
if (FEXCore::StrConv::Conv(StrVal, &ConvertedValue)) {
|
||||
// Convert the value.
|
||||
OptionMap[Option].emplace<T>(ConvertedValue);
|
||||
return ConvertedValue;
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Couldn't Convert {} to specified type!", StrVal);
|
||||
}
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
private:
|
||||
void MergeConfigMap(const LayerOptions& Options);
|
||||
void MergeEnvironmentVariables(const ConfigOption& Option, const LayerValue& Value);
|
||||
void MergeEnvironmentVariables(const ConfigOption& Option, const DefaultValues::Type::StringArrayType& Value);
|
||||
};
|
||||
|
||||
void MetaLayer::Load() {
|
||||
@@ -161,7 +182,7 @@ void MetaLayer::Load() {
|
||||
}
|
||||
|
||||
|
||||
void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const LayerValue& Value) {
|
||||
void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const DefaultValues::Type::StringArrayType& Value) {
|
||||
// Environment variables need a bit of additional work
|
||||
// We want to merge the arrays rather than overwrite entirely
|
||||
auto MetaEnvironment = OptionMap.find(Option);
|
||||
@@ -173,7 +194,7 @@ void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const Laye
|
||||
|
||||
// If an environment variable exists in both current meta and in the incoming layer then the meta layer value is overwritten
|
||||
fextl::unordered_map<fextl::string, fextl::string> LookupMap;
|
||||
const auto AddToMap = [&LookupMap](const FEXCore::Config::LayerValue& Value) {
|
||||
const auto AddToMap = [&LookupMap](const DefaultValues::Type::StringArrayType& Value) {
|
||||
for (const auto& EnvVar : Value) {
|
||||
const auto ItEq = EnvVar.find_first_of('=');
|
||||
if (ItEq == fextl::string::npos) {
|
||||
@@ -189,7 +210,7 @@ void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const Laye
|
||||
}
|
||||
};
|
||||
|
||||
AddToMap(MetaEnvironment->second);
|
||||
AddToMap(std::get<DefaultValues::Type::StringArrayType>(MetaEnvironment->second));
|
||||
AddToMap(Value);
|
||||
|
||||
// Now with the two layers merged in the map
|
||||
@@ -197,7 +218,7 @@ void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const Laye
|
||||
Erase(Option);
|
||||
for (auto& Val : LookupMap) {
|
||||
// Set will emplace multiple options in to its list
|
||||
Set(Option, Val.first + "=" + Val.second);
|
||||
AppendStrArrayValue(Option, Val.first + "=" + Val.second);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -205,7 +226,8 @@ void MetaLayer::MergeConfigMap(const LayerOptions& Options) {
|
||||
// Insert this layer's options, overlaying previous options that exist here
|
||||
for (auto& it : Options) {
|
||||
if (it.first == FEXCore::Config::ConfigOption::CONFIG_ENV || it.first == FEXCore::Config::ConfigOption::CONFIG_HOSTENV) {
|
||||
MergeEnvironmentVariables(it.first, it.second);
|
||||
LOGMAN_THROW_A_FMT(std::holds_alternative<DefaultValues::Type::StringArrayType>(it.second), "Tried to get config of invalid type!");
|
||||
MergeEnvironmentVariables(it.first, std::get<DefaultValues::Type::StringArrayType>(it.second));
|
||||
} else {
|
||||
OptionMap.insert_or_assign(it.first, it.second);
|
||||
}
|
||||
@@ -214,7 +236,7 @@ void MetaLayer::MergeConfigMap(const LayerOptions& Options) {
|
||||
|
||||
void Initialize() {
|
||||
AddLayer(fextl::make_unique<MetaLayer>(FEXCore::Config::LayerType::LAYER_TOP));
|
||||
Meta = ConfigLayers.begin()->second.get();
|
||||
Meta = dynamic_cast<MetaLayer*>(ConfigLayers.begin()->second.get());
|
||||
}
|
||||
|
||||
void Shutdown() {
|
||||
@@ -322,7 +344,7 @@ void ReloadMetaLayer() {
|
||||
auto ExpandPathIfExists = [&ContainerPrefix](FEXCore::Config::ConfigOption Config, const fextl::string& PathName) {
|
||||
const auto NewPath = ExpandPath(ContainerPrefix, PathName);
|
||||
if (!NewPath.empty()) {
|
||||
FEXCore::Config::EraseSet(Config, NewPath);
|
||||
FEXCore::Config::Set(Config, NewPath);
|
||||
}
|
||||
};
|
||||
|
||||
@@ -331,7 +353,7 @@ void ReloadMetaLayer() {
|
||||
const auto ExpandedString = ExpandPath(ContainerPrefix, *PathName);
|
||||
if (!ExpandedString.empty()) {
|
||||
// Adjust the path if it ended up being relative
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, ExpandedString);
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_ROOTFS, ExpandedString);
|
||||
} else if (!PathName->empty()) {
|
||||
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
|
||||
const auto PathNameCopy = *PathName;
|
||||
@@ -339,7 +361,7 @@ void ReloadMetaLayer() {
|
||||
for (auto DirectoryFetchers : {GetDataDirectory, GetConfigDirectory}) {
|
||||
fextl::string NamedRootFS = DirectoryFetchers(Global) + "RootFS/" + PathNameCopy;
|
||||
if (FHU::Filesystem::Exists(NamedRootFS)) {
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, NamedRootFS);
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_ROOTFS, NamedRootFS);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -358,7 +380,7 @@ void ReloadMetaLayer() {
|
||||
const auto ExpandedString = ExpandPath(ContainerPrefix, *PathName);
|
||||
if (!ExpandedString.empty()) {
|
||||
// Adjust the path if it ended up being relative
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, ExpandedString);
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_THUNKCONFIG, ExpandedString);
|
||||
} else if (!PathName->empty()) {
|
||||
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
|
||||
const auto PathNameCopy = *PathName;
|
||||
@@ -366,7 +388,7 @@ void ReloadMetaLayer() {
|
||||
for (auto DirectoryFetchers : {GetDataDirectory, GetConfigDirectory}) {
|
||||
fextl::string NamedConfig = DirectoryFetchers(Global) + "ThunkConfigs/" + PathNameCopy;
|
||||
if (FHU::Filesystem::Exists(NamedConfig)) {
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, NamedConfig);
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_THUNKCONFIG, NamedConfig);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -383,12 +405,12 @@ void ReloadMetaLayer() {
|
||||
// If DumpIR is set but no PassManagerDumpIR configuration is set, then default to `afteropt`
|
||||
const auto PathName = *Meta->Get(FEXCore::Config::CONFIG_DUMPIR);
|
||||
if (*PathName != "no") {
|
||||
EraseSet(FEXCore::Config::ConfigOption::CONFIG_PASSMANAGERDUMPIR,
|
||||
fextl::fmt::format("{}", static_cast<uint64_t>(FEXCore::Config::PassManagerDumpIR::AFTEROPT)));
|
||||
Set(FEXCore::Config::ConfigOption::CONFIG_PASSMANAGERDUMPIR,
|
||||
fextl::fmt::format("{}", static_cast<uint64_t>(FEXCore::Config::PassManagerDumpIR::AFTEROPT)));
|
||||
}
|
||||
}
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_SINGLESTEP)) {
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_SINGLESTEP) && Meta->GetConv<bool>(FEXCore::Config::CONFIG_SINGLESTEP).value_or(false)) {
|
||||
// Single stepping also enforces single instruction size blocks
|
||||
Set(FEXCore::Config::ConfigOption::CONFIG_MAXINST, "1");
|
||||
}
|
||||
@@ -402,7 +424,7 @@ bool Exists(ConfigOption Option) {
|
||||
return Meta->OptionExists(Option);
|
||||
}
|
||||
|
||||
std::optional<LayerValue*> All(ConfigOption Option) {
|
||||
std::optional<DefaultValues::Type::StringArrayType*> All(ConfigOption Option) {
|
||||
return Meta->All(Option);
|
||||
}
|
||||
|
||||
@@ -410,6 +432,11 @@ std::optional<fextl::string*> Get(ConfigOption Option) {
|
||||
return Meta->Get(Option);
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
std::optional<T> GetConv(ConfigOption Option) {
|
||||
return Meta->GetConv<T>(Option);
|
||||
}
|
||||
|
||||
void Set(ConfigOption Option, std::string_view Data) {
|
||||
Meta->Set(Option, Data);
|
||||
}
|
||||
@@ -418,31 +445,14 @@ void Erase(ConfigOption Option) {
|
||||
Meta->Erase(Option);
|
||||
}
|
||||
|
||||
void EraseSet(ConfigOption Option, std::string_view Data) {
|
||||
Meta->EraseSet(Option, Data);
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
T Value<T>::Get(FEXCore::Config::ConfigOption Option) {
|
||||
T Result;
|
||||
auto Value = FEXCore::Config::Get(Option);
|
||||
|
||||
if (!FEXCore::StrConv::Conv(**Value, &Result)) {
|
||||
LOGMAN_MSG_A_FMT("Attempted to convert invalid value");
|
||||
}
|
||||
return Result;
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
T Value<T>::GetIfExists(FEXCore::Config::ConfigOption Option, T Default) {
|
||||
T Result;
|
||||
auto Value = FEXCore::Config::Get(Option);
|
||||
|
||||
if (Value && FEXCore::StrConv::Conv(**Value, &Result)) {
|
||||
return Result;
|
||||
} else {
|
||||
return Default;
|
||||
auto Value = FEXCore::Config::GetConv<T>(Option);
|
||||
if (Value) {
|
||||
return *Value;
|
||||
}
|
||||
|
||||
return Default;
|
||||
}
|
||||
|
||||
template<>
|
||||
@@ -482,12 +492,13 @@ template Value<uint8_t>::Value(FEXCore::Config::ConfigOption _Option, uint8_t De
|
||||
template Value<uint64_t>::Value(FEXCore::Config::ConfigOption _Option, uint64_t Default);
|
||||
|
||||
template<typename T>
|
||||
void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string>* List) {
|
||||
void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, DefaultValues::Type::StringArrayType* List) {
|
||||
auto Value = FEXCore::Config::All(Option);
|
||||
List->clear();
|
||||
if (Value) {
|
||||
*List = **Value;
|
||||
}
|
||||
}
|
||||
template void Value<fextl::string>::GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string>* List);
|
||||
template void Value<DefaultValues::Type::StringArrayType>::GetListIfExists(FEXCore::Config::ConfigOption Option,
|
||||
DefaultValues::Type::StringArrayType* List);
|
||||
} // namespace FEXCore::Config
|
||||
@@ -59,6 +59,8 @@
|
||||
"DISABLEFLAGM": "disableflagm",
|
||||
"ENABLEFLAGM2": "enableflagm2",
|
||||
"DISABLEFLAGM2": "disableflagm2",
|
||||
"ENABLEFRINTTS": "enablefrintts",
|
||||
"DISABLEFRINTTS": "disablefrintts",
|
||||
"ENABLECRYPTO": "enablecrypto",
|
||||
"DISABLECRYPTO": "disablecrypto",
|
||||
"ENABLERPRES": "enablerpres",
|
||||
@@ -90,19 +92,6 @@
|
||||
"\t{enable,disable}preserveallabi: Will force enable or disable preserve_all abi even if the host doesn't support it"
|
||||
]
|
||||
},
|
||||
"CPUID": {
|
||||
"Type": "strenum",
|
||||
"Default": "FEXCore::Config::CPUID::OFF",
|
||||
"Enums": {
|
||||
"ENABLESHA": "enablesha",
|
||||
"DISABLESHA": "disablesha"
|
||||
},
|
||||
"Desc": [
|
||||
"Allows controlling of the CPU features are exposed in CPUID.",
|
||||
"\toff: Default CPU features queried from CPU features",
|
||||
"\t{enable,disable}sha: Will force enable or disable sha even if the host doesn't support it"
|
||||
]
|
||||
},
|
||||
"SmallTSCScale": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
@@ -433,6 +422,14 @@
|
||||
"Should work without issues in most cases."
|
||||
]
|
||||
},
|
||||
"VolatileMetadata": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"Use volatile metadata in PE files to inform TSO instructions when available.",
|
||||
"When metadata is unavailable falls back to the currently enabled TSO options."
|
||||
]
|
||||
},
|
||||
"X87ReducedPrecision": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
|
||||
@@ -186,6 +186,10 @@ public:
|
||||
|
||||
void AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t GuestThunkEntrypoint) override;
|
||||
|
||||
void AddForceTSOInformation(const IntervalList<uint64_t>& ValidRanges, fextl::set<uint64_t>&& Instructions) override;
|
||||
|
||||
void RemoveForceTSOInformation(uint64_t Address, uint64_t Size) override;
|
||||
|
||||
public:
|
||||
friend class FEXCore::HLE::SyscallHandler;
|
||||
#ifdef JIT_ARM64
|
||||
@@ -373,5 +377,7 @@ private:
|
||||
std::shared_mutex CustomIRMutex;
|
||||
std::atomic<bool> HasCustomIRHandlers {};
|
||||
fextl::unordered_map<uint64_t, std::tuple<CustomIREntrypointHandler, void*, void*>> CustomIRHandlers;
|
||||
IntervalList<uint64_t> ForceTSOValidRanges; // The ranges for which ForceTSOInstructions has populated data
|
||||
fextl::set<uint64_t> ForceTSOInstructions;
|
||||
};
|
||||
} // namespace FEXCore::Context
|
||||
@@ -0,0 +1,158 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Core/Addressing.h"
|
||||
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "FEXCore/Utils/MathUtils.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage) {
|
||||
Ref Tmp = A.Base;
|
||||
|
||||
if (A.Offset) {
|
||||
Ref Offset = IREmit->_Constant(A.Offset);
|
||||
Tmp = Tmp ? IREmit->_Add(GPRSize, Tmp, Offset) : Offset;
|
||||
}
|
||||
|
||||
if (A.Index) {
|
||||
if (A.IndexScale != 1) {
|
||||
uint32_t Log2 = FEXCore::ilog2(A.IndexScale);
|
||||
|
||||
if (Tmp) {
|
||||
Tmp = IREmit->_AddShift(GPRSize, Tmp, A.Index, ShiftType::LSL, Log2);
|
||||
} else {
|
||||
Tmp = IREmit->_Lshl(GPRSize, A.Index, IREmit->_Constant(Log2));
|
||||
}
|
||||
} else {
|
||||
Tmp = Tmp ? IREmit->_Add(GPRSize, Tmp, A.Index) : A.Index;
|
||||
}
|
||||
}
|
||||
|
||||
// For 64-bit AddrSize can be 32-bit or 64-bit
|
||||
// For 32-bit AddrSize can be 32-bit or 16-bit
|
||||
//
|
||||
// If the AddrSize is not the GPRSize then we need to clear the upper bits.
|
||||
if ((A.AddrSize < GPRSize) && !AllowUpperGarbage && Tmp) {
|
||||
Tmp = IREmit->_Bfe(GPRSize, IR::OpSizeAsBits(A.AddrSize), 0, Tmp);
|
||||
}
|
||||
|
||||
if (A.Segment && AddSegmentBase) {
|
||||
Tmp = Tmp ? IREmit->_Add(GPRSize, Tmp, A.Segment) : A.Segment;
|
||||
}
|
||||
|
||||
return Tmp ?: IREmit->_Constant(0);
|
||||
}
|
||||
|
||||
AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO, bool Vector,
|
||||
IR::OpSize AccessSize) {
|
||||
auto SoftwareAddressCalculation = [IREmit, &A, GPRSize]() -> AddressMode {
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(IREmit, A, GPRSize, true),
|
||||
.Index = IREmit->Invalid(),
|
||||
};
|
||||
};
|
||||
|
||||
const auto Is32Bit = GPRSize == OpSize::i32Bit;
|
||||
const auto GPRSizeMatchesAddrSize = A.AddrSize == GPRSize;
|
||||
const auto OffsetIndexToLargeFor32Bit = Is32Bit && (A.Offset <= -16384 || A.Offset >= 16384);
|
||||
if (!GPRSizeMatchesAddrSize || OffsetIndexToLargeFor32Bit) {
|
||||
// If address size doesn't match GPR size then no optimizations can occur.
|
||||
return SoftwareAddressCalculation();
|
||||
}
|
||||
|
||||
// Loadstore rules:
|
||||
// Non-TSO GPR:
|
||||
// * LDR/STR: [Reg]
|
||||
// * LDR/STR: [Reg + Reg, {Shift <AccessSize>}]
|
||||
// * Can't use with 32-bit
|
||||
// * LDR/STR: [Reg + [0,4095] * <AccessSize>]
|
||||
// * Imm must be smaller than 16k with 32-bit
|
||||
// * LDUR/STUR: [Reg + [-256, 255]]
|
||||
//
|
||||
// TSO GPR:
|
||||
// * ARMv8.0:
|
||||
// LDAR/STLR: [Reg]
|
||||
// * FEAT_LRCPC:
|
||||
// LDAPR: [Reg]
|
||||
// * FEAT_LRCPC2:
|
||||
// LDAPUR/STLUR: [Reg + [-256, 255]]
|
||||
//
|
||||
// Non-TSO Vector:
|
||||
// * LDR/STR: [Reg + [0,4095] * <AccessSize>]
|
||||
// * LDUR/STUR: [Reg + [-256,255]]
|
||||
//
|
||||
// TSO Vector:
|
||||
// * ARMv8.0:
|
||||
// Just DMB + previous
|
||||
// * FEAT_LRCPC3 (Unsupported by FEXCore currently):
|
||||
// LDAPUR/STLUR: [Reg + [-256,255]]
|
||||
|
||||
const auto AccessSizeAsImm = OpSizeToSize(AccessSize);
|
||||
const bool OffsetIsSIMM9 = A.Offset && A.Offset >= -256 && A.Offset <= 255;
|
||||
const bool OffsetIsUnsignedScaled = A.Offset > 0 && (A.Offset & (AccessSizeAsImm - 1)) == 0 && (A.Offset / AccessSizeAsImm) <= 4095;
|
||||
|
||||
auto InlineImmOffsetLoadstore = [IREmit, &GPRSize](AddressMode A) -> AddressMode {
|
||||
// Peel off the offset
|
||||
AddressMode B = A;
|
||||
B.Offset = 0;
|
||||
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(IREmit, B, GPRSize, true /* AddSegmentBase */, false),
|
||||
.Index = IREmit->_Constant(A.Offset),
|
||||
.IndexType = MEM_OFFSET_SXTX,
|
||||
.IndexScale = 1,
|
||||
};
|
||||
};
|
||||
|
||||
auto ScaledRegisterLoadstore = [IREmit, GPRSize](AddressMode A) -> AddressMode {
|
||||
if (A.Index && A.Segment) {
|
||||
A.Base = IREmit->_Add(GPRSize, A.Base, A.Segment);
|
||||
} else if (A.Segment) {
|
||||
A.Index = A.Segment;
|
||||
A.IndexScale = 1;
|
||||
}
|
||||
return A;
|
||||
};
|
||||
|
||||
if (AtomicTSO) {
|
||||
if (!Vector) {
|
||||
if (HostSupportsTSOImm9 && OffsetIsSIMM9) {
|
||||
return InlineImmOffsetLoadstore(A);
|
||||
}
|
||||
} else {
|
||||
// TODO: LRCPC3 support for vector Imm9.
|
||||
}
|
||||
} else {
|
||||
if (OffsetIsSIMM9 || OffsetIsUnsignedScaled) {
|
||||
return InlineImmOffsetLoadstore(A);
|
||||
} else if (!Is32Bit && A.Base && (A.Index || A.Segment) && !A.Offset && (A.IndexScale == 1 || A.IndexScale == AccessSizeAsImm)) {
|
||||
return ScaledRegisterLoadstore(A);
|
||||
}
|
||||
}
|
||||
|
||||
if (Vector || !AtomicTSO) {
|
||||
if ((A.Base || A.Segment) && A.Offset) {
|
||||
const bool Const_16K = A.Offset > -16384 && A.Offset < 16384 && GPRSizeMatchesAddrSize && Is32Bit;
|
||||
|
||||
if (!Is32Bit || Const_16K) {
|
||||
// Peel off the offset
|
||||
AddressMode B = A;
|
||||
B.Offset = 0;
|
||||
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(IREmit, B, GPRSize, true /* AddSegmentBase */, false),
|
||||
.Index = IREmit->_Constant(A.Offset),
|
||||
.IndexType = MEM_OFFSET_SXTX,
|
||||
.IndexScale = 1,
|
||||
};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Fallback on software address calculation
|
||||
return SoftwareAddressCalculation();
|
||||
}
|
||||
|
||||
|
||||
}; // namespace FEXCore::IR
|
||||
@@ -0,0 +1,27 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include "Interface/IR/IR.h"
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class IREmitter;
|
||||
|
||||
struct AddressMode {
|
||||
Ref Segment {nullptr};
|
||||
Ref Base {nullptr};
|
||||
Ref Index {nullptr};
|
||||
MemOffsetType IndexType = MEM_OFFSET_SXTX;
|
||||
uint8_t IndexScale = 1;
|
||||
int64_t Offset = 0;
|
||||
|
||||
// Size in bytes for the address calculation. 8 for an arm64 hardware mode.
|
||||
IR::OpSize AddrSize;
|
||||
bool NonTSO;
|
||||
};
|
||||
|
||||
Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage = false);
|
||||
AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO, bool Vector,
|
||||
IR::OpSize AccessSize);
|
||||
|
||||
}; // namespace FEXCore::IR
|
||||
@@ -24,6 +24,22 @@
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
// LLVM's preserve_all doc, this is used throughout this file and reproduced
|
||||
// here for reference:
|
||||
//
|
||||
// the callee preserve all general purpose registers,
|
||||
// except X0-X8 and X16-X18. Furthermore it also preserves lower 128 bits of
|
||||
// V8-V31 SIMD - floating point registers.
|
||||
//
|
||||
// Note that the call necessarily also clobbers x30, the link register (LR)
|
||||
// which is not considered general purpose.
|
||||
//
|
||||
// Meanwhile, for non-preserve_all, the AAPCS64 ABI says:
|
||||
//
|
||||
// A subroutine invocation must preserve the contents of the registers
|
||||
// r19-r29 and SP.
|
||||
|
||||
namespace x64 {
|
||||
#ifndef _M_ARM_64EC
|
||||
// All but x19 and x29 are caller saved
|
||||
@@ -50,6 +66,13 @@ namespace x64 {
|
||||
REG_AF,
|
||||
};
|
||||
|
||||
// I wish this could get constexpr generated from SRA's definition but impossible until libstdc++12, libc++15.
|
||||
// SRA GPRs that need to be spilled when calling a function with `preserve_all` ABI.
|
||||
constexpr std::array<ARMEmitter::Register, 7> PreserveAll_SRA = {
|
||||
ARMEmitter::Reg::r4, ARMEmitter::Reg::r5, ARMEmitter::Reg::r6, ARMEmitter::Reg::r7,
|
||||
ARMEmitter::Reg::r8, ARMEmitter::Reg::r16, ARMEmitter::Reg::r17,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::Register, 8> RA = {
|
||||
// All these callee saved
|
||||
ARMEmitter::Reg::r20, ARMEmitter::Reg::r21, ARMEmitter::Reg::r22, ARMEmitter::Reg::r23,
|
||||
@@ -58,6 +81,14 @@ namespace x64 {
|
||||
|
||||
constexpr unsigned RAPairs = 6;
|
||||
|
||||
// Dynamic GPRs
|
||||
constexpr std::array<ARMEmitter::Register, 2> PreserveAll_Dynamic = {
|
||||
ARMEmitter::Reg::r18,
|
||||
ARMEmitter::Reg::r30,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::Register, 2> NotPreserved_Dynamic = PreserveAll_Dynamic;
|
||||
|
||||
// All are caller saved
|
||||
constexpr std::array<ARMEmitter::VRegister, 16> SRAFPR = {
|
||||
ARMEmitter::VReg::v16, ARMEmitter::VReg::v17, ARMEmitter::VReg::v18, ARMEmitter::VReg::v19,
|
||||
@@ -65,6 +96,11 @@ namespace x64 {
|
||||
ARMEmitter::VReg::v24, ARMEmitter::VReg::v25, ARMEmitter::VReg::v26, ARMEmitter::VReg::v27,
|
||||
ARMEmitter::VReg::v28, ARMEmitter::VReg::v29, ARMEmitter::VReg::v30, ARMEmitter::VReg::v31};
|
||||
|
||||
// SRA FPRs that need to be spilled when calling a function with `preserve_all` ABI.
|
||||
constexpr std::array<ARMEmitter::Register, 0> PreserveAll_SRAFPR = {
|
||||
// None.
|
||||
};
|
||||
|
||||
// v8..v15 = (lower 64bits) Callee saved
|
||||
constexpr std::array<ARMEmitter::VRegister, 14> RAFPR = {
|
||||
// v0 ~ v1 are used as temps.
|
||||
@@ -74,6 +110,10 @@ namespace x64 {
|
||||
ARMEmitter::VReg::v7, ARMEmitter::VReg::v8, ARMEmitter::VReg::v9, ARMEmitter::VReg::v10, ARMEmitter::VReg::v11,
|
||||
ARMEmitter::VReg::v12, ARMEmitter::VReg::v13, ARMEmitter::VReg::v14, ARMEmitter::VReg::v15,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::VRegister, 6> PreserveAll_DynamicFPR = {
|
||||
ARMEmitter::VReg::v2, ARMEmitter::VReg::v3, ARMEmitter::VReg::v4, ARMEmitter::VReg::v5, ARMEmitter::VReg::v6, ARMEmitter::VReg::v7,
|
||||
};
|
||||
#else
|
||||
constexpr std::array<ARMEmitter::Register, 18> SRA = {
|
||||
ARMEmitter::Reg::r8,
|
||||
@@ -98,11 +138,22 @@ namespace x64 {
|
||||
REG_AF,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::Register, 7> PreserveAll_SRA = {
|
||||
ARMEmitter::Reg::r0, ARMEmitter::Reg::r1, ARMEmitter::Reg::r2, ARMEmitter::Reg::r3,
|
||||
ARMEmitter::Reg::r4, ARMEmitter::Reg::r5, ARMEmitter::Reg::r8,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::Register, 7> RA = {
|
||||
ARMEmitter::Reg::r6, ARMEmitter::Reg::r7, ARMEmitter::Reg::r14, ARMEmitter::Reg::r15,
|
||||
ARMEmitter::Reg::r16, ARMEmitter::Reg::r17, ARMEmitter::Reg::r30,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::Register, 5> PreserveAll_Dynamic = {
|
||||
ARMEmitter::Reg::r6, ARMEmitter::Reg::r7, ARMEmitter::Reg::r16, ARMEmitter::Reg::r17, ARMEmitter::Reg::r30,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::Register, 7> NotPreserved_Dynamic = RA;
|
||||
|
||||
constexpr unsigned RAPairs = 6;
|
||||
|
||||
constexpr std::array<ARMEmitter::VRegister, 16> SRAFPR = {
|
||||
@@ -112,18 +163,20 @@ namespace x64 {
|
||||
ARMEmitter::VReg::v12, ARMEmitter::VReg::v13, ARMEmitter::VReg::v14, ARMEmitter::VReg::v15,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::VRegister, 8> PreserveAll_SRAFPR = {
|
||||
ARMEmitter::VReg::v0, ARMEmitter::VReg::v1, ARMEmitter::VReg::v2, ARMEmitter::VReg::v3,
|
||||
ARMEmitter::VReg::v4, ARMEmitter::VReg::v5, ARMEmitter::VReg::v6, ARMEmitter::VReg::v7,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::VRegister, 14> RAFPR = {
|
||||
ARMEmitter::VReg::v18, ARMEmitter::VReg::v19, ARMEmitter::VReg::v20, ARMEmitter::VReg::v21, ARMEmitter::VReg::v22,
|
||||
ARMEmitter::VReg::v23, ARMEmitter::VReg::v24, ARMEmitter::VReg::v25, ARMEmitter::VReg::v26, ARMEmitter::VReg::v27,
|
||||
ARMEmitter::VReg::v28, ARMEmitter::VReg::v29, ARMEmitter::VReg::v30, ARMEmitter::VReg::v31};
|
||||
#endif
|
||||
|
||||
// I wish this could get constexpr generated from SRA's definition but impossible until libstdc++12, libc++15.
|
||||
// SRA GPRs that need to be spilled when calling a function with `preserve_all` ABI.
|
||||
constexpr std::array<ARMEmitter::Register, 7> PreserveAll_SRA = {
|
||||
ARMEmitter::Reg::r4, ARMEmitter::Reg::r5, ARMEmitter::Reg::r6, ARMEmitter::Reg::r7,
|
||||
ARMEmitter::Reg::r8, ARMEmitter::Reg::r16, ARMEmitter::Reg::r17,
|
||||
constexpr std::array<ARMEmitter::VRegister, 0> PreserveAll_DynamicFPR = {
|
||||
// None
|
||||
};
|
||||
#endif
|
||||
|
||||
constexpr uint32_t PreserveAll_SRAMask = {[]() -> uint32_t {
|
||||
uint32_t Mask {};
|
||||
@@ -147,16 +200,6 @@ namespace x64 {
|
||||
return Mask;
|
||||
}()};
|
||||
|
||||
// Dynamic GPRs
|
||||
constexpr std::array<ARMEmitter::Register, 1> PreserveAll_Dynamic = {
|
||||
// Only LR needs to get saved.
|
||||
ARMEmitter::Reg::r30};
|
||||
|
||||
// SRA FPRs that need to be spilled when calling a function with `preserve_all` ABI.
|
||||
constexpr std::array<ARMEmitter::Register, 0> PreserveAll_SRAFPR = {
|
||||
// None.
|
||||
};
|
||||
|
||||
constexpr uint32_t PreserveAll_SRAFPRMask = {[]() -> uint32_t {
|
||||
uint32_t Mask {};
|
||||
for (auto Reg : PreserveAll_SRAFPR) {
|
||||
@@ -165,13 +208,6 @@ namespace x64 {
|
||||
return Mask;
|
||||
}()};
|
||||
|
||||
// Dynamic FPRs
|
||||
// - v0-v7
|
||||
constexpr std::array<ARMEmitter::VRegister, 6> PreserveAll_DynamicFPR = {
|
||||
// v0 ~ v1 are temps
|
||||
ARMEmitter::VReg::v2, ARMEmitter::VReg::v3, ARMEmitter::VReg::v4, ARMEmitter::VReg::v5, ARMEmitter::VReg::v6, ARMEmitter::VReg::v7,
|
||||
};
|
||||
|
||||
// SRA FPRs that need to be spilled when the host supports SVE-256bit with `preserve_all` ABI.
|
||||
// This is /all/ of the SRA registers
|
||||
constexpr std::array<ARMEmitter::VRegister, 16> PreserveAll_SRAFPRSVE = SRAFPR;
|
||||
@@ -232,6 +268,11 @@ namespace x32 {
|
||||
ARMEmitter::Reg::r19,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::Register, 7> NotPreserved_Dynamic = {
|
||||
ARMEmitter::Reg::r12, ARMEmitter::Reg::r13, ARMEmitter::Reg::r14, ARMEmitter::Reg::r15,
|
||||
ARMEmitter::Reg::r16, ARMEmitter::Reg::r17, ARMEmitter::Reg::r30,
|
||||
};
|
||||
|
||||
constexpr unsigned RAPairs = 12;
|
||||
|
||||
// All are caller saved
|
||||
@@ -284,17 +325,7 @@ namespace x32 {
|
||||
constexpr std::array<ARMEmitter::Register, 3> PreserveAll_Dynamic = {ARMEmitter::Reg::r16, ARMEmitter::Reg::r17, ARMEmitter::Reg::r30};
|
||||
|
||||
// SRA FPRs that need to be spilled when calling a function with `preserve_all` ABI.
|
||||
constexpr std::array<ARMEmitter::Register, 0> PreserveAll_SRAFPR = {
|
||||
// None.
|
||||
};
|
||||
|
||||
constexpr uint32_t PreserveAll_SRAFPRMask = {[]() -> uint32_t {
|
||||
uint32_t Mask {};
|
||||
for (auto Reg : PreserveAll_SRAFPR) {
|
||||
Mask |= (1U << Reg.Idx());
|
||||
}
|
||||
return Mask;
|
||||
}()};
|
||||
constexpr uint32_t PreserveAll_SRAFPRMask = 0;
|
||||
|
||||
// Dynamic FPRs
|
||||
// - v0-v7
|
||||
@@ -355,18 +386,16 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr
|
||||
if (EmitterCTX->Config.Is64BitMode()) {
|
||||
StaticRegisters = x64::SRA;
|
||||
GeneralRegisters = x64::RA;
|
||||
GeneralRegistersNotPreserved = x64::NotPreserved_Dynamic;
|
||||
StaticFPRegisters = x64::SRAFPR;
|
||||
GeneralFPRegisters = x64::RAFPR;
|
||||
PairRegisters = x64::RAPairs;
|
||||
#ifdef _M_ARM_64EC
|
||||
ConfiguredDynamicRegisterBase = std::span(x64::RA.begin(), 7);
|
||||
#endif
|
||||
} else {
|
||||
ConfiguredDynamicRegisterBase = std::span(x32::RA.begin() + 6, 8);
|
||||
PairRegisters = x32::RAPairs;
|
||||
|
||||
StaticRegisters = x32::SRA;
|
||||
GeneralRegisters = x32::RA;
|
||||
GeneralRegistersNotPreserved = x32::NotPreserved_Dynamic;
|
||||
|
||||
StaticFPRegisters = x32::SRAFPR;
|
||||
GeneralFPRegisters = x32::RAFPR;
|
||||
@@ -676,10 +705,12 @@ void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs, uint3
|
||||
|
||||
// Now handle PF/AF
|
||||
if (PFAFSpillMask) {
|
||||
auto PFOffset = offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw);
|
||||
[[maybe_unused]] auto AFOffset = offsetof(FEXCore::Core::CpuStateFrame, State.af_raw);
|
||||
LOGMAN_THROW_A_FMT(PFAFSpillMask == PFAFMask, "PF/AF not spilled together");
|
||||
LOGMAN_THROW_A_FMT(AFOffset == PFOffset + 4, "PF/AF are together");
|
||||
|
||||
str(REG_PF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw));
|
||||
str(REG_AF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.af_raw));
|
||||
stp<ARMEmitter::IndexType::OFFSET>(REG_PF.W(), REG_AF.W(), STATE.R(), PFOffset);
|
||||
}
|
||||
|
||||
if (FPRs) {
|
||||
@@ -825,8 +856,7 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
if (PFAFFillMask) {
|
||||
LOGMAN_THROW_A_FMT(PFAFFillMask == PFAFMask, "PF/AF not filled together");
|
||||
|
||||
ldr(REG_PF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw));
|
||||
ldr(REG_AF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.af_raw));
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(REG_PF.W(), REG_AF.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -930,9 +960,9 @@ void Arm64Emitter::PopGeneralRegisters(std::span<const ARMEmitter::Register> Reg
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::PushDynamicRegsAndLR(ARMEmitter::Register TmpReg) {
|
||||
void Arm64Emitter::PushDynamicRegs(ARMEmitter::Register TmpReg) {
|
||||
const auto CanUseSVE256 = EmitterCTX->HostFeatures.SupportsSVE256;
|
||||
const auto GPRSize = (ConfiguredDynamicRegisterBase.size() + 1) * Core::CPUState::GPR_REG_SIZE;
|
||||
const auto GPRSize = GeneralRegistersNotPreserved.size() * Core::CPUState::GPR_REG_SIZE;
|
||||
const auto FPRRegSize = CanUseSVE256 ? 32 : 16;
|
||||
const auto FPRSize = GeneralFPRegisters.size() * FPRRegSize;
|
||||
const uint64_t SPOffset = AlignUp(GPRSize + FPRSize, 16);
|
||||
@@ -948,25 +978,17 @@ void Arm64Emitter::PushDynamicRegsAndLR(ARMEmitter::Register TmpReg) {
|
||||
PushVectorRegisters(TmpReg, CanUseSVE256, GeneralFPRegisters);
|
||||
|
||||
// Push the general registers.
|
||||
PushGeneralRegisters(TmpReg, ConfiguredDynamicRegisterBase);
|
||||
|
||||
#ifndef _M_ARM_64EC
|
||||
str(ARMEmitter::XReg::lr, TmpReg, 0);
|
||||
#endif
|
||||
PushGeneralRegisters(TmpReg, GeneralRegistersNotPreserved);
|
||||
}
|
||||
|
||||
void Arm64Emitter::PopDynamicRegsAndLR() {
|
||||
void Arm64Emitter::PopDynamicRegs() {
|
||||
const auto CanUseSVE256 = EmitterCTX->HostFeatures.SupportsSVE256;
|
||||
|
||||
// Pop vectors first
|
||||
PopVectorRegisters(CanUseSVE256, GeneralFPRegisters);
|
||||
|
||||
// Pop GPRs second
|
||||
PopGeneralRegisters(ConfiguredDynamicRegisterBase);
|
||||
|
||||
#ifndef _M_ARM_64EC
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
#endif
|
||||
PopGeneralRegisters(GeneralRegistersNotPreserved);
|
||||
}
|
||||
|
||||
void Arm64Emitter::SpillForPreserveAllABICall(ARMEmitter::Register TmpReg, bool FPRs) {
|
||||
|
||||
@@ -104,9 +104,9 @@ protected:
|
||||
|
||||
FEXCore::Context::ContextImpl* EmitterCTX;
|
||||
|
||||
std::span<const ARMEmitter::Register> ConfiguredDynamicRegisterBase {};
|
||||
std::span<const ARMEmitter::Register> StaticRegisters {};
|
||||
std::span<const ARMEmitter::Register> GeneralRegisters {};
|
||||
std::span<const ARMEmitter::Register> GeneralRegistersNotPreserved {};
|
||||
std::span<const ARMEmitter::VRegister> StaticFPRegisters {};
|
||||
std::span<const ARMEmitter::VRegister> GeneralFPRegisters {};
|
||||
uint32_t PairRegisters = 0;
|
||||
@@ -141,8 +141,8 @@ protected:
|
||||
void PopVectorRegisters(bool SVERegs, std::span<const ARMEmitter::VRegister> VRegs);
|
||||
void PopGeneralRegisters(std::span<const ARMEmitter::Register> Regs);
|
||||
|
||||
void PushDynamicRegsAndLR(ARMEmitter::Register TmpReg);
|
||||
void PopDynamicRegsAndLR();
|
||||
void PushDynamicRegs(ARMEmitter::Register TmpReg);
|
||||
void PopDynamicRegs();
|
||||
|
||||
void PushCalleeSavedRegisters();
|
||||
void PopCalleeSavedRegisters();
|
||||
@@ -150,12 +150,12 @@ protected:
|
||||
// Spills and fills SRA/Dynamic registers that are required for Arm64 `preserve_all` ABI.
|
||||
// This ABI changes most registers to be callee saved.
|
||||
// Caller Saved:
|
||||
// - X0-X8, X16-X18.
|
||||
// - X0-X8, X16-X18, X30.
|
||||
// - v0-v7
|
||||
// - For 256-bit SVE hosts: top 128-bits of v8-v31
|
||||
//
|
||||
// Callee Saved:
|
||||
// - X9-X15, X19-X31
|
||||
// - X9-X15, X19-X29, X31
|
||||
// - Low 128-bits of v8-v31
|
||||
void SpillForPreserveAllABICall(ARMEmitter::Register TmpReg, bool FPRs = true);
|
||||
void FillForPreserveAllABICall(bool FPRs = true);
|
||||
@@ -165,7 +165,7 @@ protected:
|
||||
SpillForPreserveAllABICall(TmpReg, FPRs);
|
||||
} else {
|
||||
SpillStaticRegs(TmpReg, FPRs);
|
||||
PushDynamicRegsAndLR(TmpReg);
|
||||
PushDynamicRegs(TmpReg);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -173,7 +173,7 @@ protected:
|
||||
if (SupportsPreserveAllABI) {
|
||||
FillForPreserveAllABICall(FPRs);
|
||||
} else {
|
||||
PopDynamicRegsAndLR();
|
||||
PopDynamicRegs();
|
||||
FillStaticRegs(FPRs);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -298,7 +298,7 @@ namespace CPU {
|
||||
// Fill in telemetry values
|
||||
for (size_t i = 0; i < FEXCore::Telemetry::TYPE_LAST; ++i) {
|
||||
auto& Telem = FEXCore::Telemetry::GetTelemetryValue(static_cast<FEXCore::Telemetry::TelemetryType>(i));
|
||||
Common.TelemetryValueAddresses[i] = reinterpret_cast<uint64_t>(Telem.GetAddr());
|
||||
Common.TelemetryValueAddresses[i] = reinterpret_cast<uint64_t>(&Telem);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
@@ -395,25 +395,7 @@ void CPUIDEmu::SetupFeatures() {
|
||||
XCR0 |= XCR0_AVX;
|
||||
}
|
||||
|
||||
// Override features if the user has specifically called for it.
|
||||
FEX_CONFIG_OPT(CPUIDFeatures, CPUID);
|
||||
if (!CPUIDFeatures()) {
|
||||
// Early exit if no features are overriden.
|
||||
return;
|
||||
}
|
||||
|
||||
#define ENABLE_DISABLE_OPTION(FeatureName, name, enum_name) \
|
||||
do { \
|
||||
const bool Disable##name = (CPUIDFeatures() & FEXCore::Config::CPUID::DISABLE##enum_name) != 0; \
|
||||
const bool Enable##name = (CPUIDFeatures() & FEXCore::Config::CPUID::ENABLE##enum_name) != 0; \
|
||||
LogMan::Throw::AFmt(!(Disable##name && Enable##name), "Disabling and Enabling CPU feature (" #name ") is mutually exclusive"); \
|
||||
const bool AlreadyEnabled = Features.FeatureName; \
|
||||
const bool Result = (AlreadyEnabled | Enable##name) & !Disable##name; \
|
||||
Features.FeatureName = Result; \
|
||||
} while (0)
|
||||
|
||||
ENABLE_DISABLE_OPTION(SHA, SHA, SHA);
|
||||
#undef ENABLE_DISABLE_OPTION
|
||||
Features.SHA = CTX->HostFeatures.SupportsSHA;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0h(uint32_t Leaf) const {
|
||||
|
||||
@@ -380,7 +380,7 @@ bool ContextImpl::InitCore() {
|
||||
SignalDelegation->SetConfig(SignalConfig);
|
||||
|
||||
#ifndef _WIN32
|
||||
#elif !defined(_M_ARM64EC)
|
||||
#elif !defined(_M_ARM_64EC)
|
||||
// WOW64 always needs the interrupt fault check to be enabled.
|
||||
Config.NeedsPendingInterruptFaultCheck = true;
|
||||
#endif
|
||||
@@ -565,6 +565,16 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
for (size_t j = 0; j < CodeBlocks->size(); ++j) {
|
||||
const FEXCore::Frontend::Decoder::DecodedBlocks& Block = CodeBlocks->at(j);
|
||||
|
||||
bool BlockInForceTSOValidRange = false;
|
||||
auto InstForceTSOIt = ForceTSOInstructions.end();
|
||||
if (ForceTSOValidRanges.Contains({Block.Entry, Block.Entry + Block.Size})) {
|
||||
if (auto It = ForceTSOInstructions.lower_bound(Block.Entry); *It < Block.Entry + Block.Size) {
|
||||
InstForceTSOIt = It;
|
||||
BlockInForceTSOValidRange = true;
|
||||
}
|
||||
}
|
||||
|
||||
// Set the block entry point
|
||||
Thread->OpDispatcher->SetNewBlockIfChanged(Block.Entry);
|
||||
|
||||
@@ -581,6 +591,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < InstsInBlock; ++i) {
|
||||
uint64_t InstAddress = Block.Entry + BlockInstructionsLength;
|
||||
const FEXCore::X86Tables::X86InstInfo* TableInfo {nullptr};
|
||||
const FEXCore::X86Tables::DecodedInst* DecodedInfo {nullptr};
|
||||
|
||||
@@ -603,7 +614,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
Thread->OpDispatcher->FlushRegisterCache(true);
|
||||
|
||||
if (ExtendedDebugInfo || Thread->OpDispatcher->CanHaveSideEffects(TableInfo, DecodedInfo)) {
|
||||
Thread->OpDispatcher->_GuestOpcode(Block.Entry + BlockInstructionsLength - GuestRIP);
|
||||
Thread->OpDispatcher->_GuestOpcode(InstAddress - GuestRIP);
|
||||
}
|
||||
|
||||
if (Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL) {
|
||||
@@ -620,7 +631,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
Thread->OpDispatcher->SetCurrentCodeBlock(CodeWasChangedBlock);
|
||||
Thread->OpDispatcher->_ThreadRemoveCodeEntry();
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, Block.Entry + BlockInstructionsLength - GuestRIP));
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, InstAddress - GuestRIP));
|
||||
|
||||
auto NextOpBlock = Thread->OpDispatcher->CreateNewCodeBlockAfter(CurrentBlock);
|
||||
|
||||
@@ -632,17 +643,27 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
auto Fn = TableInfo->OpcodeDispatcher;
|
||||
Thread->OpDispatcher->ResetHandledLock();
|
||||
Thread->OpDispatcher->ResetDecodeFailure();
|
||||
IR::ForceTSOMode ForceTSO =
|
||||
BlockInForceTSOValidRange ?
|
||||
(InstForceTSOIt != ForceTSOInstructions.end() && *InstForceTSOIt == InstAddress ? IR::ForceTSOMode::ForceEnabled :
|
||||
IR::ForceTSOMode::ForceDisabled) :
|
||||
IR::ForceTSOMode::NoOverride;
|
||||
Thread->OpDispatcher->SetForceTSO(ForceTSO);
|
||||
std::invoke(Fn, Thread->OpDispatcher, DecodedInfo);
|
||||
if (Thread->OpDispatcher->HadDecodeFailure()) {
|
||||
HadDispatchError = true;
|
||||
} else {
|
||||
if (Thread->OpDispatcher->HasHandledLock() != IsLocked) {
|
||||
HadDispatchError = true;
|
||||
LogMan::Msg::EFmt("Missing LOCK HANDLER at 0x{:x}{{'{}'}}", Block.Entry + BlockInstructionsLength, TableInfo->Name ?: "UND");
|
||||
LogMan::Msg::EFmt("Missing LOCK HANDLER at 0x{:x}{{'{}'}}", InstAddress, TableInfo->Name ?: "UND");
|
||||
}
|
||||
BlockInstructionsLength += DecodedInfo->InstSize;
|
||||
TotalInstructionsLength += DecodedInfo->InstSize;
|
||||
++TotalInstructions;
|
||||
|
||||
// Walk InstForceTSOIt forward past the handled instruction
|
||||
InstForceTSOIt =
|
||||
std::find_if(InstForceTSOIt, ForceTSOInstructions.end(), [&](auto Val) { return Val >= Block.Entry + BlockInstructionsLength; });
|
||||
}
|
||||
} else {
|
||||
// Invalid instruction
|
||||
@@ -971,6 +992,19 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::AddForceTSOInformation(const IntervalList<uint64_t>& ValidRanges, fextl::set<uint64_t>&& Instructions) {
|
||||
LogMan::Throw::AFmt(CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to be unique_locked here");
|
||||
ForceTSOValidRanges.Insert(ValidRanges);
|
||||
ForceTSOInstructions.merge(Instructions);
|
||||
}
|
||||
|
||||
void ContextImpl::RemoveForceTSOInformation(uint64_t Address, uint64_t Size) {
|
||||
LogMan::Throw::AFmt(CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to be unique_locked here");
|
||||
|
||||
ForceTSOValidRanges.Remove({Address, Address + Size});
|
||||
ForceTSOInstructions.erase(ForceTSOInstructions.lower_bound(Address), ForceTSOInstructions.upper_bound(Address + Size));
|
||||
}
|
||||
|
||||
void ContextImpl::RemoveCustomIREntrypoint(uintptr_t Entrypoint) {
|
||||
LOGMAN_THROW_A_FMT(Config.Is64BitMode || !(Entrypoint >> 32), "64-bit Entrypoint in 32-bit mode {:x}", Entrypoint);
|
||||
|
||||
|
||||
@@ -505,7 +505,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
auto EmitLongALUOpHandler = [&](auto R, auto Offset) {
|
||||
auto Address = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR(TMP4);
|
||||
PushDynamicRegs(TMP4);
|
||||
SpillStaticRegs(TMP4);
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
@@ -529,7 +529,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
FillStaticRegs();
|
||||
|
||||
// Fix the stack and any values that were stepped on
|
||||
PopDynamicRegsAndLR();
|
||||
PopDynamicRegs();
|
||||
|
||||
// Go back to our code block
|
||||
ret();
|
||||
|
||||
@@ -471,7 +471,9 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
|
||||
size_t CurrentSrc = 0;
|
||||
|
||||
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_1ST_SRC) != 0) {
|
||||
const auto VEXOperand = Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_SRC_MASK;
|
||||
|
||||
if (VEXOperand == FEXCore::X86Tables::InstFlags::FLAGS_VEX_1ST_SRC) {
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
|
||||
|
||||
@@ -496,7 +498,7 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
++CurrentSrc;
|
||||
}
|
||||
|
||||
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_2ND_SRC) != 0) {
|
||||
if (VEXOperand == FEXCore::X86Tables::InstFlags::FLAGS_VEX_2ND_SRC) {
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = MapVEXToReg(Options.vvvv, HasXMMSrc);
|
||||
@@ -515,7 +517,7 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
++CurrentSrc;
|
||||
}
|
||||
|
||||
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_DST) != 0) {
|
||||
if (VEXOperand == FEXCore::X86Tables::InstFlags::FLAGS_VEX_DST) {
|
||||
CurrentDest->Type = DecodedOperand::OpType::GPR;
|
||||
CurrentDest->Data.GPR.HighBits = false;
|
||||
CurrentDest->Data.GPR.GPR = MapVEXToReg(Options.vvvv, HasXMMDst);
|
||||
@@ -690,7 +692,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
return NormalOp(LocalInfo, Op, options);
|
||||
}
|
||||
} else if (Info->Type == FEXCore::X86Tables::TYPE_GROUP_EVEX) {
|
||||
FEXCORE_TELEMETRY_SET(EVEXOpTelem, 1);
|
||||
FEXCORE_TELEMETRY_SET(TYPE_USES_EVEX_OPS, 1);
|
||||
// EVEX unsupported
|
||||
return false;
|
||||
}
|
||||
@@ -912,12 +914,6 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (DecodeInst->Dest.IsGPR()) {
|
||||
LOGMAN_THROW_A_FMT(DecodeInst->Dest.Data.GPR.GPR != FEXCore::X86State::REG_INVALID, "Destination GPR was invalid");
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
void Decoder::BranchTargetInMultiblockRange() {
|
||||
|
||||
@@ -90,10 +90,10 @@ private:
|
||||
Utils::FixedSizePooledAllocation<FEXCore::X86Tables::DecodedInst*, 5000, 500> PoolObject;
|
||||
size_t DecodedSize {};
|
||||
|
||||
const uint8_t* InstStream;
|
||||
const uint8_t* InstStream {};
|
||||
|
||||
static constexpr size_t MAX_INST_SIZE = 15;
|
||||
uint8_t InstructionSize;
|
||||
uint8_t InstructionSize {};
|
||||
std::array<uint8_t, MAX_INST_SIZE> Instruction;
|
||||
FEXCore::X86Tables::DecodedInst* DecodeInst;
|
||||
|
||||
@@ -124,7 +124,5 @@ private:
|
||||
};
|
||||
|
||||
const uint8_t* AdjustAddrForSpecialRegion(const uint8_t* _InstStream, uint64_t EntryPoint, uint64_t RIP);
|
||||
|
||||
FEXCORE_TELEMETRY_INIT(EVEXOpTelem, TYPE_USES_EVEX_OPS);
|
||||
};
|
||||
} // namespace FEXCore::Frontend
|
||||
@@ -1,6 +1,6 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include "Common/SoftFloat.h"
|
||||
#include "Common/SoftFloat-3e/softfloat.h"
|
||||
|
||||
#include "Interface/Core/Interpreter/Fallbacks/FallbackOpHandler.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <algorithm>
|
||||
|
||||
@@ -1386,7 +1386,10 @@ DEF_OP(Bfi) {
|
||||
bfi(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
} else if (Dst != Src) {
|
||||
// If the destination isn't the source then we can move the DstSrc and insert directly.
|
||||
mov(EmitSize, Dst, SrcDst);
|
||||
//
|
||||
// The move is 64-bit to allow register renaming, the upper bits don't
|
||||
// matter because of the bfi's EmitSize.
|
||||
mov(ARMEmitter::Size::i64Bit, Dst, SrcDst);
|
||||
bfi(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
} else {
|
||||
// Destination didn't match the dst source register.
|
||||
@@ -1555,7 +1558,7 @@ DEF_OP(VExtractToGPR) {
|
||||
|
||||
const auto Offset = ElementSizeBits * Op->Index;
|
||||
[[maybe_unused]] const auto Is256Bit = Offset >= SSERegBitSize;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
|
||||
@@ -101,11 +101,16 @@ DEF_OP(CAS) {
|
||||
auto Expected = GetReg(Op->Expected.ID());
|
||||
auto Desired = GetReg(Op->Desired.ID());
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Dst = GetReg(Node);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mov(EmitSize, TMP2, Expected);
|
||||
casal(SubEmitSize, TMP2, Desired, MemSrc);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
if (Expected == Dst && Dst != MemSrc && Dst != Desired) {
|
||||
casal(SubEmitSize, Dst, Desired, MemSrc);
|
||||
} else {
|
||||
mov(EmitSize, TMP2, Expected);
|
||||
casal(SubEmitSize, TMP2, Desired, MemSrc);
|
||||
mov(EmitSize, Dst, TMP2.R());
|
||||
}
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
ARMEmitter::ForwardLabel LoopNotExpected;
|
||||
@@ -122,11 +127,11 @@ DEF_OP(CAS) {
|
||||
b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
|
||||
stlxr(SubEmitSize, TMP3, Desired, MemSrc);
|
||||
cbnz(EmitSize, TMP3, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), Expected);
|
||||
mov(EmitSize, Dst, Expected);
|
||||
b(&LoopExpected);
|
||||
|
||||
Bind(&LoopNotExpected);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
mov(EmitSize, Dst, TMP2.R());
|
||||
// exclusive monitor needs to be cleared here
|
||||
// Might have hit the case where ldaxr was hit but stlxr wasn't
|
||||
clrex();
|
||||
@@ -286,7 +291,6 @@ DEF_OP(AtomicSwap) {
|
||||
const auto SubEmitSize = OpSize == IR::OpSize::i64Bit ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == IR::OpSize::i32Bit ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == IR::OpSize::i16Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == IR::OpSize::i8Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
|
||||
@@ -150,7 +150,7 @@ DEF_OP(Syscall) {
|
||||
// X2: Pointer to SyscallArguments
|
||||
|
||||
FEXCore::IR::SyscallFlags Flags = Op->Flags;
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
PushDynamicRegs(TMP1);
|
||||
|
||||
uint32_t GPRSpillMask = ~0U;
|
||||
uint32_t FPRSpillMask = ~0U;
|
||||
@@ -201,7 +201,7 @@ DEF_OP(Syscall) {
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
PopDynamicRegs();
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NORETURNEDRESULT) != FEXCore::IR::SyscallFlags::NORETURNEDRESULT) {
|
||||
// Move result to its destination register.
|
||||
@@ -314,7 +314,7 @@ DEF_OP(Thunk) {
|
||||
|
||||
SpillStaticRegs(TMP1); // spill to ctx before ra64 spill
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
PushDynamicRegs(TMP1);
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetReg(Op->ArgPtr.ID()));
|
||||
|
||||
@@ -326,7 +326,7 @@ DEF_OP(Thunk) {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
PopDynamicRegs();
|
||||
|
||||
FillStaticRegs(); // load from ctx after ra64 refill
|
||||
}
|
||||
@@ -378,7 +378,7 @@ DEF_OP(ValidateCode) {
|
||||
}
|
||||
|
||||
DEF_OP(ThreadRemoveCodeEntry) {
|
||||
PushDynamicRegsAndLR(TMP4);
|
||||
PushDynamicRegs(TMP4);
|
||||
SpillStaticRegs(TMP4);
|
||||
|
||||
// Arguments are passed as follows:
|
||||
@@ -397,7 +397,7 @@ DEF_OP(ThreadRemoveCodeEntry) {
|
||||
FillStaticRegs();
|
||||
|
||||
// Fix the stack and any values that were stepped on
|
||||
PopDynamicRegsAndLR();
|
||||
PopDynamicRegs();
|
||||
}
|
||||
|
||||
DEF_OP(CPUID) {
|
||||
@@ -406,7 +406,7 @@ DEF_OP(CPUID) {
|
||||
mov(ARMEmitter::Size::i64Bit, TMP2, GetReg(Op->Function.ID()));
|
||||
mov(ARMEmitter::Size::i64Bit, TMP3, GetReg(Op->Leaf.ID()));
|
||||
|
||||
PushDynamicRegsAndLR(TMP4);
|
||||
PushDynamicRegs(TMP4);
|
||||
SpillStaticRegs(TMP4);
|
||||
|
||||
// x0 = CPUID Handler
|
||||
@@ -433,7 +433,7 @@ DEF_OP(CPUID) {
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
PopDynamicRegs();
|
||||
|
||||
// Results are in x0, x1
|
||||
// Results want to be 4xi32 scalars
|
||||
@@ -446,7 +446,7 @@ DEF_OP(CPUID) {
|
||||
DEF_OP(XGetBV) {
|
||||
auto Op = IROp->C<IR::IROp_XGetBV>();
|
||||
|
||||
PushDynamicRegsAndLR(TMP4);
|
||||
PushDynamicRegs(TMP4);
|
||||
SpillStaticRegs(TMP4);
|
||||
|
||||
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, GetReg(Op->Function.ID()));
|
||||
@@ -467,7 +467,7 @@ DEF_OP(XGetBV) {
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
PopDynamicRegs();
|
||||
|
||||
// Results are in x0, need to split into i32 parts
|
||||
mov(ARMEmitter::Size::i32Bit, GetReg(Op->OutEAX.ID()), TMP1);
|
||||
|
||||
@@ -6,6 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
@@ -16,7 +17,7 @@ DEF_OP(VInsGPR) {
|
||||
const auto DestIdx = Op->DestIdx;
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp);
|
||||
const auto ElementsPer128Bit = IR::NumElements(IR::OpSize::i128Bit, ElementSize);
|
||||
@@ -104,6 +105,16 @@ DEF_OP(VCastFromGPR) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VLoadTwoGPRs) {
|
||||
const auto Op = IROp->C<IR::IROp_VLoadTwoGPRs>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto SrcLower = GetReg(Op->Lower.ID());
|
||||
const auto SrcUpper = GetReg(Op->Upper.ID());
|
||||
fmov(ARMEmitter::Size::i64Bit, Dst.D(), SrcLower);
|
||||
fmov(ARMEmitter::Size::i64Bit, Dst.D(), SrcUpper, true);
|
||||
}
|
||||
|
||||
DEF_OP(VDupFromGPR) {
|
||||
const auto Op = IROp->C<IR::IROp_VDupFromGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -112,7 +123,7 @@ DEF_OP(VDupFromGPR) {
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
@@ -206,7 +217,7 @@ DEF_OP(Vector_SToF) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -239,7 +250,7 @@ DEF_OP(Vector_FToZS) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -270,7 +281,7 @@ DEF_OP(Vector_FToS) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
|
||||
@@ -301,7 +312,7 @@ DEF_OP(Vector_FToF) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Conv = (IR::OpSizeToSize(ElementSize) << 8) | IR::OpSizeToSize(Op->SrcElementSize);
|
||||
|
||||
@@ -404,7 +415,7 @@ DEF_OP(Vector_FToI) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -459,13 +470,68 @@ DEF_OP(Vector_FToI) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToISized) {
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToISized>();
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = IROp->Size == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit, "256-bit not wired up, though we could change that");
|
||||
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFRINTTS, "Need FRINTTS for Vector_FToISized");
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
|
||||
if (ElementSize == IROp->Size) {
|
||||
// See above
|
||||
#define ROUNDING_FN(name) \
|
||||
if (ElementSize == IR::OpSize::i32Bit) { \
|
||||
name(Dst.S(), Vector.S()); \
|
||||
} else if (ElementSize == IR::OpSize::i64Bit) { \
|
||||
name(Dst.D(), Vector.D()); \
|
||||
} else { \
|
||||
FEX_UNREACHABLE; \
|
||||
}
|
||||
|
||||
if (Op->IntSize == IR::OpSize::i64Bit) {
|
||||
if (Op->HostRound) {
|
||||
ROUNDING_FN(frint64x);
|
||||
} else {
|
||||
ROUNDING_FN(frint64z);
|
||||
}
|
||||
} else {
|
||||
if (Op->HostRound) {
|
||||
ROUNDING_FN(frint32x);
|
||||
} else {
|
||||
ROUNDING_FN(frint32z);
|
||||
}
|
||||
}
|
||||
|
||||
#undef ROUNDING_FN
|
||||
} else {
|
||||
if (Op->IntSize == IR::OpSize::i64Bit) {
|
||||
if (Op->HostRound) {
|
||||
frint64x(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
} else {
|
||||
frint64z(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
}
|
||||
} else {
|
||||
if (Op->HostRound) {
|
||||
frint32x(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
} else {
|
||||
frint32z(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_F64ToI32) {
|
||||
const auto Op = IROp->C<IR::IROp_Vector_F64ToI32>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Round = Op->Round;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
|
||||
@@ -248,6 +248,46 @@ DEF_OP(VSha1SU1) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSha256H) {
|
||||
auto Op = IROp->C<IR::IROp_VSha256H>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src3 = GetVReg(Op->Src3.ID());
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha256h(Dst, Src2, Src3);
|
||||
} else if (Dst != Src2 && Dst != Src3) {
|
||||
mov(Dst.Q(), Src1.Q());
|
||||
sha256h(Dst, Src2, Src3);
|
||||
} else {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
sha256h(VTMP1, Src2, Src3);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSha256H2) {
|
||||
auto Op = IROp->C<IR::IROp_VSha256H2>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src3 = GetVReg(Op->Src3.ID());
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha256h2(Dst, Src2, Src3);
|
||||
} else if (Dst != Src2 && Dst != Src3) {
|
||||
mov(Dst.Q(), Src1.Q());
|
||||
sha256h2(Dst, Src2, Src3);
|
||||
} else {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
sha256h2(VTMP1, Src2, Src3);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSha256U0) {
|
||||
auto Op = IROp->C<IR::IROp_VSha256U0>();
|
||||
|
||||
@@ -271,7 +311,7 @@ DEF_OP(VSha256U1) {
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
|
||||
if (Dst != Src1 && Dst != Src1) {
|
||||
if (Dst != Src1 && Dst != Src2) {
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), 0);
|
||||
sha256su1(Dst, Src1, Src2);
|
||||
} else {
|
||||
|
||||
@@ -57,10 +57,10 @@ private:
|
||||
const bool HostSupportsRPRES {};
|
||||
const bool HostSupportsAFP {};
|
||||
|
||||
ARMEmitter::BiDirectionalLabel* PendingTargetLabel;
|
||||
FEXCore::Context::ContextImpl* CTX;
|
||||
const FEXCore::IR::IRListView* IR;
|
||||
uint64_t Entry;
|
||||
ARMEmitter::BiDirectionalLabel* PendingTargetLabel {};
|
||||
FEXCore::Context::ContextImpl* CTX {};
|
||||
const FEXCore::IR::IRListView* IR {};
|
||||
uint64_t Entry {};
|
||||
CPUBackend::CompiledCode CodeData {};
|
||||
|
||||
fextl::map<IR::NodeID, ARMEmitter::BiDirectionalLabel> JumpTargets;
|
||||
@@ -257,9 +257,9 @@ private:
|
||||
|
||||
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
|
||||
void EmitDetectionString();
|
||||
IR::RegisterAllocationPass* RAPass;
|
||||
const IR::RegisterAllocationData* RAData;
|
||||
FEXCore::Core::DebugData* DebugData;
|
||||
IR::RegisterAllocationPass* RAPass {};
|
||||
const IR::RegisterAllocationData* RAData {};
|
||||
FEXCore::Core::DebugData* DebugData {};
|
||||
|
||||
void ResetStack();
|
||||
/**
|
||||
|
||||
@@ -642,10 +642,10 @@ ARMEmitter::SVEMemOperand Arm64JITCore::GenerateSVEMemOperand(IR::OpSize AccessS
|
||||
}
|
||||
|
||||
const auto SignedConst = static_cast<int64_t>(Const);
|
||||
const auto SignedAVXSize = static_cast<int64_t>(Core::CPUState::XMM_AVX_REG_SIZE);
|
||||
const auto SignedSVESize = static_cast<int64_t>(HostSupportsSVE256 ? Core::CPUState::XMM_AVX_REG_SIZE : Core::CPUState::XMM_SSE_REG_SIZE);
|
||||
|
||||
const auto IsCleanlyDivisible = (SignedConst % SignedAVXSize) == 0;
|
||||
const auto Index = SignedConst / SignedAVXSize;
|
||||
const auto IsCleanlyDivisible = (SignedConst % SignedSVESize) == 0;
|
||||
const auto Index = SignedConst / SignedSVESize;
|
||||
|
||||
// SVE's immediate variants of load stores are quite limited in terms
|
||||
// of immediate range. They also operate on a by-vector-length basis.
|
||||
@@ -837,7 +837,7 @@ DEF_OP(VLoadVectorMasked) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
const auto CMPPredicate = ARMEmitter::PReg::p0;
|
||||
@@ -892,7 +892,15 @@ DEF_OP(VLoadVectorMasked) {
|
||||
auto WorkingReg = TMP1;
|
||||
auto TempMemReg = MemReg;
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, TempDst.Q(), 0);
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid(), "Complex addressing requested and not supported!");
|
||||
uint64_t Const {};
|
||||
if (Op->Offset.IsInvalid()) {
|
||||
// Intentional no-op.
|
||||
} else if (IsInlineConstant(Op->Offset, &Const)) {
|
||||
TempMemReg = TMP2;
|
||||
add(ARMEmitter::Size::i64Bit, TMP2, MemReg, Const);
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Complex addressing requested and not supported!");
|
||||
}
|
||||
|
||||
const uint64_t ElementSizeInBits = IR::OpSizeAsBits(IROp->ElementSize);
|
||||
for (size_t i = 0; i < NumElements; ++i) {
|
||||
@@ -932,7 +940,7 @@ DEF_OP(VStoreVectorMasked) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
const auto CMPPredicate = ARMEmitter::PReg::p0;
|
||||
@@ -984,7 +992,16 @@ DEF_OP(VStoreVectorMasked) {
|
||||
// Use VTMP1 as the temporary destination
|
||||
auto WorkingReg = TMP1;
|
||||
auto TempMemReg = MemReg;
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid(), "Complex addressing requested and not supported!");
|
||||
|
||||
uint64_t Const {};
|
||||
if (Op->Offset.IsInvalid()) {
|
||||
// Intentional no-op.
|
||||
} else if (IsInlineConstant(Op->Offset, &Const)) {
|
||||
TempMemReg = TMP2;
|
||||
add(ARMEmitter::Size::i64Bit, TMP2, MemReg, Const);
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Complex addressing requested and not supported!");
|
||||
}
|
||||
|
||||
const uint64_t ElementSizeInBits = IR::OpSizeAsBits(IROp->ElementSize);
|
||||
for (size_t i = 0; i < NumElements; ++i) {
|
||||
@@ -1021,6 +1038,7 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
|
||||
ARMEmitter::VRegister VectorIndexLow, std::optional<ARMEmitter::VRegister> VectorIndexHigh,
|
||||
ARMEmitter::VRegister MaskReg, IR::OpSize VectorIndexSize, size_t DataElementOffsetStart,
|
||||
size_t IndexElementOffsetStart, uint8_t OffsetScale) {
|
||||
LOGMAN_THROW_A_FMT(ElementSize >= IR::OpSize::i8Bit && ElementSize <= IR::OpSize::i64Bit, "Invalid element size");
|
||||
|
||||
const auto PerformSMove = [this](IR::OpSize ElementSize, const ARMEmitter::Register Dst, const ARMEmitter::VRegister Vector, int index) {
|
||||
switch (ElementSize) {
|
||||
@@ -1150,7 +1168,7 @@ DEF_OP(VLoadVectorGatherMasked) {
|
||||
/// - AddrBase also doesn't need to exist
|
||||
/// - If the instruction is using 64-bit vector indexing or 32-bit addresses where the top-bit isn't set then this is valid!
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto IncomingDst = GetVReg(Op->Incoming.ID());
|
||||
@@ -1381,7 +1399,7 @@ DEF_OP(VBroadcastFromMem) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -2506,7 +2524,7 @@ DEF_OP(VStoreNonTemporal) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
|
||||
|
||||
const auto Value = GetVReg(Op->Value.ID());
|
||||
@@ -2548,7 +2566,7 @@ DEF_OP(VLoadNonTemporal) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
|
||||
@@ -166,7 +166,7 @@ DEF_OP(PopRoundingMode) {
|
||||
DEF_OP(Print) {
|
||||
auto Op = IROp->C<IR::IROp_Print>();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
PushDynamicRegs(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
if (IsGPR(Op->Value.ID())) {
|
||||
@@ -189,7 +189,7 @@ DEF_OP(Print) {
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
PopDynamicRegsAndLR();
|
||||
PopDynamicRegs();
|
||||
}
|
||||
|
||||
DEF_OP(ProcessorID) {
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -27,7 +27,6 @@ $end_info$
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <bit>
|
||||
#include <cstdint>
|
||||
#include <tuple>
|
||||
|
||||
@@ -1000,6 +999,7 @@ void OpDispatchBuilder::TESTOp(OpcodeArgs, uint32_t SrcIndex) {
|
||||
Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
const auto Size = OpSizeFromDst(Op);
|
||||
LOGMAN_THROW_A_FMT(Size >= IR::OpSize::i8Bit && Size <= IR::OpSize::i64Bit, "Invalid size");
|
||||
|
||||
uint64_t Const;
|
||||
bool AlwaysNonnegative = false;
|
||||
@@ -1092,8 +1092,8 @@ void OpDispatchBuilder::CQOOp(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::XCHGOp(OpcodeArgs) {
|
||||
// Load both the source and the destination
|
||||
if (Op->OP == 0x90 && GetSrcSize(Op) >= 4 && Op->Src[0].IsGPR() && Op->Src[0].Data.GPR.GPR == FEXCore::X86State::REG_RAX &&
|
||||
Op->Dest.IsGPR() && Op->Dest.Data.GPR.GPR == FEXCore::X86State::REG_RAX) {
|
||||
if (Op->OP == 0x90 && Op->Src[0].IsGPR() && Op->Src[0].Data.GPR.GPR == FEXCore::X86State::REG_RAX && Op->Dest.IsGPR() &&
|
||||
Op->Dest.Data.GPR.GPR == FEXCore::X86State::REG_RAX) {
|
||||
// This is one heck of a sucky special case
|
||||
// If we are the 0x90 XCHG opcode (Meaning source is GPR RAX)
|
||||
// and destination register is ALSO RAX
|
||||
@@ -1103,6 +1103,14 @@ void OpDispatchBuilder::XCHGOp(OpcodeArgs) {
|
||||
// But this would result in a zext on 64bit, which would ruin the no-op nature of the instruction
|
||||
// So x86-64 spec mandates this special case that even though it is a 32bit instruction and
|
||||
// is supposed to zext the result, it is a true no-op
|
||||
//
|
||||
// x86 spec text here:
|
||||
//
|
||||
// XCHG (E)AX, (E)AX (encoded instruction byte is 90H) is an alias for
|
||||
// NOP regardless of data size prefixes, including REX.W.
|
||||
//
|
||||
// Note that also includes 16-bit so we don't gate this on size. The
|
||||
// sequence (66 90) is a valid two-byte nop that we also ignore.
|
||||
if (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX) {
|
||||
// If this instruction has a REP prefix then this is architecturally
|
||||
// defined to be a `PAUSE` instruction. On older processors this ends up
|
||||
@@ -1422,14 +1430,15 @@ void OpDispatchBuilder::SHLDOp(OpcodeArgs) {
|
||||
// Calculate flags early.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto Size = GetSrcBitSize(Op);
|
||||
|
||||
// Allow garbage on the Src if it will be ignored by the Lshr below
|
||||
Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = Size >= 32});
|
||||
Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags);
|
||||
|
||||
// Allow garbage on the shift, we're masking it anyway.
|
||||
Ref Shift = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
const auto Size = GetSrcBitSize(Op);
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op.
|
||||
if (Size == 64) {
|
||||
Shift = _And(OpSize::i64Bit, Shift, _InlineConstant(0x3F));
|
||||
@@ -1538,7 +1547,7 @@ void OpDispatchBuilder::SHRDImmediateOp(OpcodeArgs) {
|
||||
Ref ShiftRight = _Constant(Shift);
|
||||
auto ShiftLeft = _Constant(Size - Shift);
|
||||
|
||||
auto Tmp1 = _Lshr(Size == 64 ? OpSize::i64Bit : OpSize::i32Bit, Dest, ShiftRight);
|
||||
auto Tmp1 = _Lshr(OpSize::i32Bit, Dest, ShiftRight);
|
||||
auto Tmp2 = _Lshl(OpSize::i64Bit, Src, ShiftLeft);
|
||||
|
||||
Res = _Or(OpSize::i64Bit, Tmp1, Tmp2);
|
||||
@@ -2176,7 +2185,7 @@ void OpDispatchBuilder::RCRSmallerOp(OpcodeArgs) {
|
||||
|
||||
StoreResult(GPRClass, Op, Res, OpSize::iInvalid);
|
||||
|
||||
uint64_t SrcConst;
|
||||
uint64_t SrcConst = 0;
|
||||
bool IsSrcConst = IsValueConstant(WrapNode(Src), &SrcConst);
|
||||
SrcConst &= 0x1f;
|
||||
|
||||
@@ -2400,7 +2409,10 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
unsigned LshrSize = std::max<uint8_t>(IR::OpSizeToSize(OpSize::i32Bit), Size / 8);
|
||||
auto BitSelect = (Size == (LshrSize * 8)) ? Src : _And(OpSize::i64Bit, Src, _Constant(Mask));
|
||||
|
||||
// OF/SF/ZF/AF/PF undefined.
|
||||
// OF/SF/AF/PF undefined. ZF must be preserved. We choose to preserve OF/SF
|
||||
// too since we just use an rmif to insert into CF directly. We could
|
||||
// optimize perhaps.
|
||||
//
|
||||
// Set CF before the action to save a move, except for complements where we
|
||||
// can reuse the invert.
|
||||
if (Action != BTAction::BTComplement) {
|
||||
@@ -2408,7 +2420,8 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
Value = _Lshr(IR::SizeToOpSize(LshrSize), Value, BitSelect);
|
||||
}
|
||||
|
||||
SetCFDirect_InvalidateNZV(Value, ConstantShift, Value);
|
||||
SetRFLAG(Value, X86State::RFLAG_CF_RAW_LOC, ConstantShift, true);
|
||||
CFInverted = false;
|
||||
}
|
||||
|
||||
switch (Action) {
|
||||
@@ -2441,7 +2454,9 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
Value = Dest;
|
||||
}
|
||||
|
||||
SetCFInverted_InvalidateNZV(Value, ConstantShift, true);
|
||||
SetRFLAG(Value, X86State::RFLAG_CF_RAW_LOC, ConstantShift, true);
|
||||
CFInverted = true;
|
||||
|
||||
StoreResult(GPRClass, Op, Dest, OpSize::iInvalid);
|
||||
break;
|
||||
}
|
||||
@@ -2473,7 +2488,7 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
|
||||
if (DestIsLockedMem(Op)) {
|
||||
HandledLock = true;
|
||||
Value = _AtomicFetchCLR(OpSize::i8Bit, BitMask, LoadEffectiveAddress(Address, true));
|
||||
Value = _AtomicFetchCLR(OpSize::i8Bit, BitMask, LoadEffectiveAddress(this, Address, CTX->GetGPROpSize(), true));
|
||||
} else {
|
||||
Value = _LoadMemAutoTSO(GPRClass, OpSize::i8Bit, Address, OpSize::i8Bit);
|
||||
|
||||
@@ -2488,7 +2503,7 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
|
||||
if (DestIsLockedMem(Op)) {
|
||||
HandledLock = true;
|
||||
Value = _AtomicFetchOr(OpSize::i8Bit, BitMask, LoadEffectiveAddress(Address, true));
|
||||
Value = _AtomicFetchOr(OpSize::i8Bit, BitMask, LoadEffectiveAddress(this, Address, CTX->GetGPROpSize(), true));
|
||||
} else {
|
||||
Value = _LoadMemAutoTSO(GPRClass, OpSize::i8Bit, Address, OpSize::i8Bit);
|
||||
|
||||
@@ -2503,7 +2518,7 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
|
||||
if (DestIsLockedMem(Op)) {
|
||||
HandledLock = true;
|
||||
Value = _AtomicFetchXor(OpSize::i8Bit, BitMask, LoadEffectiveAddress(Address, true));
|
||||
Value = _AtomicFetchXor(OpSize::i8Bit, BitMask, LoadEffectiveAddress(this, Address, CTX->GetGPROpSize(), true));
|
||||
} else {
|
||||
Value = _LoadMemAutoTSO(GPRClass, OpSize::i8Bit, Address, OpSize::i8Bit);
|
||||
|
||||
@@ -2706,6 +2721,7 @@ void OpDispatchBuilder::MULOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::NOTOp(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
const auto SizeBits = IR::OpSizeAsBits(Size);
|
||||
LOGMAN_THROW_A_FMT(Size >= IR::OpSize::i8Bit && Size <= IR::OpSize::i64Bit, "Invalid size");
|
||||
|
||||
Ref MaskConst {};
|
||||
if (Size == OpSize::i64Bit) {
|
||||
@@ -3313,7 +3329,7 @@ void OpDispatchBuilder::CMPSOp(OpcodeArgs) {
|
||||
SetCurrentCodeBlock(BeforeLoop);
|
||||
StartNewBlock();
|
||||
|
||||
ForeachDirection([this, Op, Size, REPE](int PtrDir) {
|
||||
ForeachDirection([this, Op, Size, REPE](int32_t PtrDir) {
|
||||
IRPair<IROp_CondJump> InnerJump;
|
||||
auto JumpIntoLoop = Jump();
|
||||
|
||||
@@ -3346,11 +3362,11 @@ void OpDispatchBuilder::CMPSOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RCX, TailCounter);
|
||||
|
||||
// Offset the pointer
|
||||
Dest_RDI = _Add(OpSize::i64Bit, Dest_RDI, _Constant(PtrDir * IR::OpSizeToSize(Size)));
|
||||
Dest_RDI = _Add(OpSize::i64Bit, Dest_RDI, _Constant(PtrDir * static_cast<int32_t>(IR::OpSizeToSize(Size))));
|
||||
StoreGPRRegister(X86State::REG_RDI, Dest_RDI);
|
||||
|
||||
// Offset second pointer
|
||||
Dest_RSI = _Add(OpSize::i64Bit, Dest_RSI, _Constant(PtrDir * IR::OpSizeToSize(Size)));
|
||||
Dest_RSI = _Add(OpSize::i64Bit, Dest_RSI, _Constant(PtrDir * static_cast<int32_t>(IR::OpSizeToSize(Size))));
|
||||
StoreGPRRegister(X86State::REG_RSI, Dest_RSI);
|
||||
|
||||
// If TailCounter != 0, compare sources.
|
||||
@@ -3412,7 +3428,7 @@ void OpDispatchBuilder::LODSOp(OpcodeArgs) {
|
||||
// Calculate flags early. because end of block
|
||||
CalculateDeferredFlags();
|
||||
|
||||
ForeachDirection([this, Op, Size](int PtrDir) {
|
||||
ForeachDirection([this, Op, Size](int32_t PtrDir) {
|
||||
// XXX: Theoretically LODS could be optimized to
|
||||
// RSI += {-}(RCX * Size)
|
||||
// RAX = [RSI - Size]
|
||||
@@ -3456,7 +3472,7 @@ void OpDispatchBuilder::LODSOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RCX, TailCounter);
|
||||
|
||||
// Offset the pointer
|
||||
TailDest_RSI = _Add(OpSize::i64Bit, TailDest_RSI, _Constant(PtrDir * IR::OpSizeToSize(Size)));
|
||||
TailDest_RSI = _Add(OpSize::i64Bit, TailDest_RSI, _Constant(PtrDir * static_cast<int32_t>(IR::OpSizeToSize(Size))));
|
||||
StoreGPRRegister(X86State::REG_RSI, TailDest_RSI);
|
||||
|
||||
// Jump back to the start, we have more work to do
|
||||
@@ -3496,7 +3512,7 @@ void OpDispatchBuilder::SCASOp(OpcodeArgs) {
|
||||
// Calculate flags early. because end of block
|
||||
CalculateDeferredFlags();
|
||||
|
||||
ForeachDirection([this, Op, Size](int Dir) {
|
||||
ForeachDirection([this, Op, Size](int32_t Dir) {
|
||||
bool REPE = Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX;
|
||||
|
||||
auto JumpStart = Jump();
|
||||
@@ -3540,7 +3556,7 @@ void OpDispatchBuilder::SCASOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RCX, TailCounter);
|
||||
|
||||
// Offset the pointer
|
||||
TailDest_RDI = _Add(OpSize::i64Bit, TailDest_RDI, _Constant(Dir * IR::OpSizeToSize(Size)));
|
||||
TailDest_RDI = _Add(OpSize::i64Bit, TailDest_RDI, _Constant(Dir * static_cast<int32_t>(IR::OpSizeToSize(Size))));
|
||||
StoreGPRRegister(X86State::REG_RDI, TailDest_RDI);
|
||||
|
||||
CalculateDeferredFlags();
|
||||
@@ -3788,7 +3804,7 @@ void OpDispatchBuilder::CMPXCHGOp(OpcodeArgs) {
|
||||
|
||||
if (GPRSize == OpSize::i64Bit && Size == OpSize::i32Bit) {
|
||||
Src1 = LoadSource_WithOpSize(GPRClass, Op, Op->Dest, GPRSize, Op->Flags, {.AllowUpperGarbage = true});
|
||||
Src1Lower = _Bfe(GPRSize, IR::OpSizeAsBits(Size), 0, Src1);
|
||||
Src1Lower = Trivial ? Src1 : _Bfe(GPRSize, IR::OpSizeAsBits(Size), 0, Src1);
|
||||
} else {
|
||||
Src1 = LoadSource_WithOpSize(GPRClass, Op, Op->Dest, Size, Op->Flags, {.AllowUpperGarbage = true});
|
||||
Src1Lower = Src1;
|
||||
@@ -3825,15 +3841,9 @@ void OpDispatchBuilder::CMPXCHGOp(OpcodeArgs) {
|
||||
Ref Src2 = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
HandledLock = Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK;
|
||||
|
||||
Ref Src3 {};
|
||||
Ref Src3Lower {};
|
||||
if (GPRSize == OpSize::i64Bit && Size == OpSize::i32Bit) {
|
||||
Src3 = LoadGPRRegister(X86State::REG_RAX);
|
||||
Src3Lower = _Bfe(OpSize::i32Bit, 32, 0, Src3);
|
||||
} else {
|
||||
Src3 = LoadGPRRegister(X86State::REG_RAX, Size);
|
||||
Src3Lower = Src3;
|
||||
}
|
||||
auto Src3 = LoadGPRRegister(X86State::REG_RAX);
|
||||
auto Src3Lower = _Bfe(OpSize::i64Bit, OpSizeAsBits(Size), 0, Src3);
|
||||
|
||||
// If this is a memory location then we want the pointer to it
|
||||
Ref Src1 = MakeSegmentAddress(Op, Op->Dest);
|
||||
|
||||
@@ -3841,7 +3851,7 @@ void OpDispatchBuilder::CMPXCHGOp(OpcodeArgs) {
|
||||
// if (DataSrc == Src3) { *Src1 == Src2; } Src2 = DataSrc
|
||||
// This will write to memory! Careful!
|
||||
// Third operand must be a calculated guest memory address
|
||||
Ref CASResult = _CAS(Size, Src3Lower, Src2, Src1);
|
||||
Ref CASResult = _CAS(Size, Src3, Src2, Src1);
|
||||
Ref RAXResult = CASResult;
|
||||
|
||||
CalculateFlags_SUB(OpSizeFromSrc(Op), Src3Lower, CASResult);
|
||||
@@ -4012,7 +4022,7 @@ Ref OpDispatchBuilder::GetSegment(uint32_t Flags, uint32_t DefaultPrefix, bool O
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX:
|
||||
SegmentResult = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, gs_cached));
|
||||
break;
|
||||
default: break; // Do nothing
|
||||
default: return nullptr;
|
||||
}
|
||||
|
||||
CheckLegacySegmentRead(SegmentResult, Prefix);
|
||||
@@ -4142,155 +4152,6 @@ void OpDispatchBuilder::UpdatePrefixFromSegment(Ref Segment, uint32_t SegmentReg
|
||||
}
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::LoadEffectiveAddress(AddressMode A, bool AddSegmentBase, bool AllowUpperGarbage) {
|
||||
const auto GPRSize = CTX->GetGPROpSize();
|
||||
Ref Tmp = A.Base;
|
||||
|
||||
if (A.Offset) {
|
||||
Ref Offset = _Constant(A.Offset);
|
||||
Tmp = Tmp ? _Add(GPRSize, Tmp, Offset) : Offset;
|
||||
}
|
||||
|
||||
if (A.Index) {
|
||||
if (A.IndexScale != 1) {
|
||||
LOGMAN_THROW_A_FMT((A.IndexScale & (A.IndexScale - 1)) == 0, "power of two");
|
||||
uint32_t Log2 = FEXCore::ilog2(A.IndexScale);
|
||||
|
||||
if (Tmp) {
|
||||
Tmp = _AddShift(GPRSize, Tmp, A.Index, ShiftType::LSL, Log2);
|
||||
} else {
|
||||
Tmp = _Lshl(GPRSize, A.Index, _Constant(Log2));
|
||||
}
|
||||
} else {
|
||||
Tmp = Tmp ? _Add(GPRSize, Tmp, A.Index) : A.Index;
|
||||
}
|
||||
}
|
||||
|
||||
// For 64-bit AddrSize can be 32-bit or 64-bit
|
||||
// For 32-bit AddrSize can be 32-bit or 16-bit
|
||||
//
|
||||
// If the AddrSize is not the GPRSize then we need to clear the upper bits.
|
||||
if ((A.AddrSize < GPRSize) && !AllowUpperGarbage && Tmp) {
|
||||
Tmp = _Bfe(GPRSize, IR::OpSizeAsBits(A.AddrSize), 0, Tmp);
|
||||
}
|
||||
|
||||
if (A.Segment && AddSegmentBase) {
|
||||
Tmp = Tmp ? _Add(GPRSize, Tmp, A.Segment) : A.Segment;
|
||||
}
|
||||
|
||||
return Tmp ?: _Constant(0);
|
||||
}
|
||||
|
||||
AddressMode OpDispatchBuilder::SelectAddressMode(AddressMode A, bool AtomicTSO, bool Vector, IR::OpSize AccessSize) {
|
||||
auto SoftwareAddressCalculation = [this, &A]() -> AddressMode {
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(A, true),
|
||||
.Index = InvalidNode,
|
||||
};
|
||||
};
|
||||
|
||||
const auto GPRSize = CTX->GetGPROpSize();
|
||||
const auto Is32Bit = GPRSize == OpSize::i32Bit;
|
||||
const auto GPRSizeMatchesAddrSize = A.AddrSize == GPRSize;
|
||||
const auto OffsetIndexToLargeFor32Bit = Is32Bit && (A.Offset <= -16384 || A.Offset >= 16384);
|
||||
if (!GPRSizeMatchesAddrSize || OffsetIndexToLargeFor32Bit) {
|
||||
// If address size doesn't match GPR size then no optimizations can occur.
|
||||
return SoftwareAddressCalculation();
|
||||
}
|
||||
|
||||
// Loadstore rules:
|
||||
// Non-TSO GPR:
|
||||
// * LDR/STR: [Reg]
|
||||
// * LDR/STR: [Reg + Reg, {Shift <AccessSize>}]
|
||||
// * Can't use with 32-bit
|
||||
// * LDR/STR: [Reg + [0,4095] * <AccessSize>]
|
||||
// * Imm must be smaller than 16k with 32-bit
|
||||
// * LDUR/STUR: [Reg + [-256, 255]]
|
||||
//
|
||||
// TSO GPR:
|
||||
// * ARMv8.0:
|
||||
// LDAR/STLR: [Reg]
|
||||
// * FEAT_LRCPC:
|
||||
// LDAPR: [Reg]
|
||||
// * FEAT_LRCPC2:
|
||||
// LDAPUR/STLUR: [Reg + [-256, 255]]
|
||||
//
|
||||
// Non-TSO Vector:
|
||||
// * LDR/STR: [Reg + [0,4095] * <AccessSize>]
|
||||
// * LDUR/STUR: [Reg + [-256,255]]
|
||||
//
|
||||
// TSO Vector:
|
||||
// * ARMv8.0:
|
||||
// Just DMB + previous
|
||||
// * FEAT_LRCPC3 (Unsupported by FEXCore currently):
|
||||
// LDAPUR/STLUR: [Reg + [-256,255]]
|
||||
|
||||
const auto AccessSizeAsImm = OpSizeToSize(AccessSize);
|
||||
const bool OffsetIsSIMM9 = A.Offset && A.Offset >= -256 && A.Offset <= 255;
|
||||
const bool OffsetIsUnsignedScaled = A.Offset > 0 && (A.Offset & (AccessSizeAsImm - 1)) == 0 && (A.Offset / AccessSizeAsImm) <= 4095;
|
||||
|
||||
auto InlineImmOffsetLoadstore = [this](AddressMode A) -> AddressMode {
|
||||
// Peel off the offset
|
||||
AddressMode B = A;
|
||||
B.Offset = 0;
|
||||
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(B, true /* AddSegmentBase */, false),
|
||||
.Index = _InlineConstant(A.Offset),
|
||||
.IndexType = MEM_OFFSET_SXTX,
|
||||
.IndexScale = 1,
|
||||
};
|
||||
};
|
||||
|
||||
auto ScaledRegisterLoadstore = [this, &GPRSize](AddressMode A) -> AddressMode {
|
||||
if (A.Index && A.Segment) {
|
||||
A.Base = _Add(GPRSize, A.Base, A.Segment);
|
||||
} else if (A.Segment) {
|
||||
A.Index = A.Segment;
|
||||
A.IndexScale = 1;
|
||||
}
|
||||
return A;
|
||||
};
|
||||
|
||||
if (AtomicTSO) {
|
||||
if (!Vector) {
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && OffsetIsSIMM9) {
|
||||
return InlineImmOffsetLoadstore(A);
|
||||
}
|
||||
} else {
|
||||
// TODO: LRCPC3 support for vector Imm9.
|
||||
}
|
||||
} else {
|
||||
if (OffsetIsSIMM9 || OffsetIsUnsignedScaled) {
|
||||
return InlineImmOffsetLoadstore(A);
|
||||
} else if (!Is32Bit && A.Base && (A.Index || A.Segment) & !A.Offset && (A.IndexScale == 1 || A.IndexScale == AccessSizeAsImm)) {
|
||||
return ScaledRegisterLoadstore(A);
|
||||
}
|
||||
}
|
||||
|
||||
if (Vector || !AtomicTSO) {
|
||||
if ((A.Base || A.Segment) && A.Offset) {
|
||||
const bool Const_16K = A.Offset > -16384 && A.Offset < 16384 && GPRSizeMatchesAddrSize && Is32Bit;
|
||||
|
||||
if (!Is32Bit || Const_16K) {
|
||||
// Peel off the offset
|
||||
AddressMode B = A;
|
||||
B.Offset = 0;
|
||||
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(B, true /* AddSegmentBase */, false),
|
||||
.Index = _Constant(A.Offset),
|
||||
.IndexType = MEM_OFFSET_SXTX,
|
||||
.IndexScale = 1,
|
||||
};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Fallback on software address calculation
|
||||
return SoftwareAddressCalculation();
|
||||
}
|
||||
|
||||
AddressMode OpDispatchBuilder::DecodeAddress(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand,
|
||||
MemoryAccessType AccessType, bool IsLoad) {
|
||||
const auto GPRSize = CTX->GetGPROpSize();
|
||||
@@ -4388,7 +4249,7 @@ Ref OpDispatchBuilder::LoadSource_WithOpSize(RegisterClassType Class, const X86T
|
||||
|
||||
if ((IsOperandMem(Operand, true) && LoadData) || ForceLoad) {
|
||||
if (OpSize == OpSize::f80Bit) {
|
||||
Ref MemSrc = LoadEffectiveAddress(A, true);
|
||||
Ref MemSrc = LoadEffectiveAddress(this, A, CTX->GetGPROpSize(), true);
|
||||
if (CTX->HostFeatures.SupportsSVE128 || CTX->HostFeatures.SupportsSVE256) {
|
||||
return _LoadMemX87SVEOptPredicate(OpSize::i128Bit, OpSize::i16Bit, MemSrc);
|
||||
} else {
|
||||
@@ -4400,7 +4261,7 @@ Ref OpDispatchBuilder::LoadSource_WithOpSize(RegisterClassType Class, const X86T
|
||||
|
||||
return _LoadMemAutoTSO(Class, OpSize, A, Align == OpSize::iInvalid ? OpSize : Align);
|
||||
} else {
|
||||
return LoadEffectiveAddress(A, false, AllowUpperGarbage);
|
||||
return LoadEffectiveAddress(this, A, CTX->GetGPROpSize(), false, AllowUpperGarbage);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4521,7 +4382,7 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
AddressMode A = DecodeAddress(Op, Operand, AccessType, false /* IsLoad */);
|
||||
|
||||
if (OpSize == OpSize::f80Bit) {
|
||||
Ref MemStoreDst = LoadEffectiveAddress(A, true);
|
||||
Ref MemStoreDst = LoadEffectiveAddress(this, A, CTX->GetGPROpSize(), true);
|
||||
if (CTX->HostFeatures.SupportsSVE128 || CTX->HostFeatures.SupportsSVE256) {
|
||||
_StoreMemX87SVEOptPredicate(OpSize::i128Bit, OpSize::i16Bit, Src, MemStoreDst);
|
||||
} else {
|
||||
@@ -5579,9 +5440,9 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
|
||||
// 1 = Invalid
|
||||
|
||||
{OPDReg(0xD9, 2) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FSTF64, OpSize::i32Bit>},
|
||||
{OPDReg(0xD9, 2) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FST, OpSize::i32Bit>},
|
||||
|
||||
{OPDReg(0xD9, 3) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FSTF64, OpSize::i32Bit>},
|
||||
{OPDReg(0xD9, 3) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FST, OpSize::i32Bit>},
|
||||
|
||||
{OPDReg(0xD9, 4) | 0x00, 8, &OpDispatchBuilder::X87LDENVF64},
|
||||
|
||||
@@ -5676,7 +5537,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
|
||||
// 6 = Invalid
|
||||
|
||||
{OPDReg(0xDB, 7) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FSTF64, OpSize::f80Bit>},
|
||||
{OPDReg(0xDB, 7) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FST, OpSize::f80Bit>},
|
||||
|
||||
|
||||
{OPD(0xDB, 0xC0), 8, &OpDispatchBuilder::X87FCMOV},
|
||||
@@ -5727,9 +5588,9 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
|
||||
{OPDReg(0xDD, 1) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FISTF64, true>},
|
||||
|
||||
{OPDReg(0xDD, 2) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FSTF64, OpSize::i64Bit>},
|
||||
{OPDReg(0xDD, 2) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FST, OpSize::i64Bit>},
|
||||
|
||||
{OPDReg(0xDD, 3) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FSTF64, OpSize::i64Bit>},
|
||||
{OPDReg(0xDD, 3) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FST, OpSize::i64Bit>},
|
||||
|
||||
{OPDReg(0xDD, 4) | 0x00, 8, &OpDispatchBuilder::X87FRSTOR},
|
||||
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Core/Addressing.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
@@ -46,6 +47,12 @@ enum class BTAction {
|
||||
BTComplement,
|
||||
};
|
||||
|
||||
enum class ForceTSOMode {
|
||||
NoOverride,
|
||||
ForceDisabled,
|
||||
ForceEnabled,
|
||||
};
|
||||
|
||||
struct LoadSourceOptions {
|
||||
// Alignment of the load in bytes. iInvalid signifies opsize aligned.
|
||||
IR::OpSize Align = OpSize::iInvalid;
|
||||
@@ -72,19 +79,6 @@ struct LoadSourceOptions {
|
||||
bool AllowUpperGarbage = false;
|
||||
};
|
||||
|
||||
struct AddressMode {
|
||||
Ref Segment {nullptr};
|
||||
Ref Base {nullptr};
|
||||
Ref Index {nullptr};
|
||||
MemOffsetType IndexType = MEM_OFFSET_SXTX;
|
||||
uint8_t IndexScale = 1;
|
||||
int64_t Offset = 0;
|
||||
|
||||
// Size in bytes for the address calculation. 8 for an arm64 hardware mode.
|
||||
IR::OpSize AddrSize;
|
||||
bool NonTSO;
|
||||
};
|
||||
|
||||
class OpDispatchBuilder final : public IREmitter {
|
||||
friend class FEXCore::IR::Pass;
|
||||
friend class FEXCore::IR::PassManager;
|
||||
@@ -273,6 +267,13 @@ public:
|
||||
return HandledLock;
|
||||
}
|
||||
|
||||
void SetForceTSO(ForceTSOMode Mode) {
|
||||
ForceTSO = Mode;
|
||||
}
|
||||
ForceTSOMode GetForceTSO() const {
|
||||
return ForceTSO;
|
||||
}
|
||||
|
||||
void SetDumpIR(bool DumpIR) {
|
||||
ShouldDump = DumpIR;
|
||||
}
|
||||
@@ -753,7 +754,6 @@ public:
|
||||
void FLDF64_Const(OpcodeArgs, uint64_t Num);
|
||||
void FLDF64(OpcodeArgs, IR::OpSize Width);
|
||||
void FMULF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpResult ResInST0);
|
||||
void FSTF64(OpcodeArgs, IR::OpSize Width);
|
||||
void FSUBF64(OpcodeArgs, IR::OpSize Width, bool Integer, bool Reverse, OpResult ResInST0);
|
||||
void FTSTF64(OpcodeArgs);
|
||||
void X87FLDCWF64(OpcodeArgs);
|
||||
@@ -1215,6 +1215,7 @@ public:
|
||||
uint64_t NextBit = (1ull << (Index - 1));
|
||||
uint32_t Offset = CacheIndexToContextOffset(Index);
|
||||
auto Class = CacheIndexClass(Index);
|
||||
LOGMAN_THROW_A_FMT(Offset != ~0U, "Invalid offset");
|
||||
|
||||
// Use stp where possible to store multiple values at a time. This accelerates AVX.
|
||||
// TODO: this is all really confusing because of backwards iteration,
|
||||
@@ -1332,6 +1333,7 @@ private:
|
||||
bool HandledLock {false};
|
||||
bool DecodeFailure {false};
|
||||
bool NeedsBlockEnd {false};
|
||||
ForceTSOMode ForceTSO {ForceTSOMode::NoOverride};
|
||||
// Used during new op bringup
|
||||
bool ShouldDump {false};
|
||||
|
||||
@@ -1492,9 +1494,6 @@ private:
|
||||
|
||||
Ref GetRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset = 0);
|
||||
|
||||
Ref LoadEffectiveAddress(AddressMode A, bool AddSegmentBase, bool AllowUpperGarbage = false);
|
||||
AddressMode SelectAddressMode(AddressMode A, bool AtomicTSO, bool Vector, IR::OpSize AccessSize);
|
||||
|
||||
bool IsOperandMem(const X86Tables::DecodedOperand& Operand, bool Load) {
|
||||
// Literals are immediates as sources but memory addresses as destinations.
|
||||
return !(Load && Operand.IsLiteral()) && !Operand.IsGPR();
|
||||
@@ -1735,27 +1734,6 @@ private:
|
||||
CFInverted = true;
|
||||
}
|
||||
|
||||
// As above but with
|
||||
//
|
||||
// x - 1
|
||||
//
|
||||
// If x = 0, hardware C is not set. If x = 1, hardware C is set.
|
||||
void SetCFInverted_InvalidateNZV(Ref Value, unsigned ValueOffset = 0, bool MustMask = false) {
|
||||
if (CTX->HostFeatures.SupportsFlagM) {
|
||||
// This turns into a single rmif
|
||||
SetCFInverted(Value, ValueOffset, MustMask);
|
||||
} else {
|
||||
// Do math on flagm
|
||||
if (ValueOffset || MustMask) {
|
||||
Value = _Bfe(OpSize::i64Bit, 1, ValueOffset, Value);
|
||||
}
|
||||
|
||||
HandleNZCVWrite();
|
||||
_SubNZCV(OpSize::i32Bit, Value, _InlineConstant(1));
|
||||
CFInverted = true;
|
||||
}
|
||||
}
|
||||
|
||||
void SetCFInverted(Ref Value, unsigned ValueOffset = 0, bool MustMask = false) {
|
||||
SetRFLAG(Value, X86State::RFLAG_CF_RAW_LOC, ValueOffset, MustMask);
|
||||
CFInverted = true;
|
||||
@@ -1837,12 +1815,12 @@ private:
|
||||
static const int AVXHigh0Index = 48;
|
||||
static const int AVXHigh15Index = 63;
|
||||
|
||||
int CacheIndexToContextOffset(int Index) {
|
||||
uint32_t CacheIndexToContextOffset(int Index) {
|
||||
switch (Index) {
|
||||
case MM0Index ... MM7Index: return offsetof(FEXCore::Core::CPUState, mm[Index - MM0Index]);
|
||||
case AVXHigh0Index ... AVXHigh15Index: return offsetof(FEXCore::Core::CPUState, avx_high[Index - AVXHigh0Index][0]);
|
||||
case AbridgedFTWIndex: return offsetof(FEXCore::Core::CPUState, AbridgedFTW);
|
||||
default: return -1;
|
||||
default: return ~0U;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2379,11 +2357,15 @@ private:
|
||||
bool BlockSetRIP {false};
|
||||
|
||||
bool Multiblock {};
|
||||
uint64_t Entry;
|
||||
uint64_t Entry {};
|
||||
IROp_IRHeader* CurrentHeader {};
|
||||
|
||||
bool IsTSOEnabled(FEXCore::IR::RegisterClassType Class) {
|
||||
if (Class == FPRClass) {
|
||||
if (ForceTSO == ForceTSOMode::ForceEnabled) {
|
||||
return true;
|
||||
} else if (ForceTSO == ForceTSOMode::ForceDisabled) {
|
||||
return false;
|
||||
} else if (Class == FPRClass) {
|
||||
return CTX->IsVectorAtomicTSOEnabled();
|
||||
} else {
|
||||
return CTX->IsAtomicTSOEnabled();
|
||||
@@ -2408,7 +2390,7 @@ private:
|
||||
|
||||
Ref _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, AddressMode A, IR::OpSize Align = IR::OpSize::i8Bit) {
|
||||
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
A = SelectAddressMode(A, AtomicTSO, Class != GPRClass, Size);
|
||||
A = SelectAddressMode(this, A, CTX->GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, AtomicTSO, Class != GPRClass, Size);
|
||||
|
||||
if (AtomicTSO) {
|
||||
return _LoadMemTSO(Class, Size, A.Base, A.Index, Align, A.IndexType, A.IndexScale);
|
||||
@@ -2428,7 +2410,7 @@ private:
|
||||
A.Offset = 0;
|
||||
}
|
||||
|
||||
Out.Base = LoadEffectiveAddress(A, true, false);
|
||||
Out.Base = LoadEffectiveAddress(this, A, CTX->GetGPROpSize(), true, false);
|
||||
return Out;
|
||||
}
|
||||
|
||||
@@ -2459,7 +2441,7 @@ private:
|
||||
|
||||
Ref _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, AddressMode A, Ref Value, IR::OpSize Align = IR::OpSize::i8Bit) {
|
||||
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
A = SelectAddressMode(A, AtomicTSO, Class != GPRClass, Size);
|
||||
A = SelectAddressMode(this, A, CTX->GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, AtomicTSO, Class != GPRClass, Size);
|
||||
|
||||
if (AtomicTSO) {
|
||||
return _StoreMemTSO(Class, Size, Value, A.Base, A.Index, Align, A.IndexType, A.IndexScale);
|
||||
|
||||
@@ -1326,10 +1326,12 @@ void OpDispatchBuilder::AVX128_MOVMSK(OpcodeArgs) {
|
||||
};
|
||||
|
||||
Ref GPR {};
|
||||
if (SrcSize == OpSize::i128Bit && ElementSize == OpSize::i64Bit) {
|
||||
GPR = Mask8Byte(Src.Low);
|
||||
} else if (SrcSize == OpSize::i128Bit && ElementSize == OpSize::i32Bit) {
|
||||
GPR = Mask4Byte(Src.Low);
|
||||
if (Is128Bit) {
|
||||
if (ElementSize == OpSize::i64Bit) {
|
||||
GPR = Mask8Byte(Src.Low);
|
||||
} else {
|
||||
GPR = Mask4Byte(Src.Low);
|
||||
}
|
||||
} else if (ElementSize == OpSize::i32Bit) {
|
||||
auto GPRLow = Mask4Byte(Src.Low);
|
||||
auto GPRHigh = Mask4Byte(Src.High);
|
||||
@@ -1670,7 +1672,6 @@ void OpDispatchBuilder::AVX128_VEXTRACT128(OpcodeArgs) {
|
||||
const auto DstIsXMM = Op->Dest.IsGPR();
|
||||
const auto Selector = Op->Src[1].Literal() & 0b1;
|
||||
|
||||
///< TODO: Once we support loading only upper-half of the ymm register we can load the half depending on selection literal.
|
||||
auto Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, true);
|
||||
|
||||
RefPair Result {};
|
||||
@@ -2032,7 +2033,18 @@ void OpDispatchBuilder::AVX128_VPALIGNR(OpcodeArgs) {
|
||||
return Src2;
|
||||
}
|
||||
|
||||
return _VExtr(OpSize::i128Bit, OpSize::i8Bit, Src1, Src2, Index);
|
||||
if (Index == 16) {
|
||||
return Src1;
|
||||
}
|
||||
|
||||
auto SanitizedIndex = Index;
|
||||
if (Index > 16) {
|
||||
Src2 = Src1;
|
||||
Src1 = LoadZeroVector(OpSize::i128Bit);
|
||||
SanitizedIndex -= 16;
|
||||
}
|
||||
|
||||
return _VExtr(OpSize::i128Bit, OpSize::i8Bit, Src1, Src2, SanitizedIndex);
|
||||
});
|
||||
}
|
||||
|
||||
@@ -2052,9 +2064,7 @@ void OpDispatchBuilder::AVX128_VMASKMOVImpl(OpcodeArgs, IR::OpSize ElementSize,
|
||||
auto Data = AVX128_LoadSource_WithOpSize(Op, DataOp, Op->Flags, !Is128Bit);
|
||||
_VStoreVectorMasked(OpSize::i128Bit, ElementSize, Mask.Low, Data.Low, Address, Invalid(), MEM_OFFSET_SXTX, 1);
|
||||
if (!Is128Bit) {
|
||||
///< TODO: This can be cleaner if AVX128_LoadSource_WithOpSize could return both constructed addresses.
|
||||
auto AddressHigh = _Add(OpSize::i64Bit, Address, _Constant(16));
|
||||
_VStoreVectorMasked(OpSize::i128Bit, ElementSize, Mask.High, Data.High, AddressHigh, Invalid(), MEM_OFFSET_SXTX, 1);
|
||||
_VStoreVectorMasked(OpSize::i128Bit, ElementSize, Mask.High, Data.High, Address, _InlineConstant(16), MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
} else {
|
||||
auto Address = MakeAddress(DataOp);
|
||||
@@ -2065,9 +2075,7 @@ void OpDispatchBuilder::AVX128_VMASKMOVImpl(OpcodeArgs, IR::OpSize ElementSize,
|
||||
if (Is128Bit) {
|
||||
Result.High = LoadZeroVector(OpSize::i128Bit);
|
||||
} else {
|
||||
///< TODO: This can be cleaner if AVX128_LoadSource_WithOpSize could return both constructed addresses.
|
||||
auto AddressHigh = _Add(OpSize::i64Bit, Address, _Constant(16));
|
||||
Result.High = _VLoadVectorMasked(OpSize::i128Bit, ElementSize, Mask.High, AddressHigh, Invalid(), MEM_OFFSET_SXTX, 1);
|
||||
Result.High = _VLoadVectorMasked(OpSize::i128Bit, ElementSize, Mask.High, Address, _InlineConstant(16), MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
}
|
||||
|
||||
@@ -296,63 +296,88 @@ Ref OpDispatchBuilder::BitwiseAtLeastTwo(Ref A, Ref B, Ref C) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
const auto Ch = [this](Ref E, Ref F, Ref G) -> Ref {
|
||||
return _Xor(OpSize::i32Bit, _And(OpSize::i32Bit, E, F), _Andn(OpSize::i32Bit, G, E));
|
||||
};
|
||||
const auto Sigma0 = [this](Ref A) -> Ref {
|
||||
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 2)), A, ShiftType::ROR, 13),
|
||||
A, ShiftType::ROR, 22);
|
||||
};
|
||||
const auto Sigma1 = [this](Ref E) -> Ref {
|
||||
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, E, _Constant(OpSize::i32Bit, 6)), E, ShiftType::ROR, 11),
|
||||
E, ShiftType::ROR, 25);
|
||||
};
|
||||
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
// Hardcoded to XMM0
|
||||
auto XMM0 = LoadXMMRegister(0);
|
||||
|
||||
auto E0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 1);
|
||||
auto F0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 0);
|
||||
auto G0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
|
||||
Ref Q0 = _Add(OpSize::i32Bit, Ch(E0, F0, G0), Sigma1(E0));
|
||||
Ref Result;
|
||||
if (CTX->HostFeatures.SupportsSHA) {
|
||||
auto shuffle_abcd = [this](Ref Src1, Ref Src2) -> Ref {
|
||||
// Generates a suitable SHA256 `abcd` configuration from x86 format.
|
||||
auto Tmp = _VZip2(OpSize::i128Bit, OpSize::i64Bit, Src2, Src1);
|
||||
return _VRev64(OpSize::i128Bit, OpSize::i32Bit, Tmp);
|
||||
};
|
||||
|
||||
auto WK0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, XMM0, 0);
|
||||
Q0 = _Add(OpSize::i32Bit, Q0, WK0);
|
||||
auto shuffle_efgh = [this](Ref Src1, Ref Src2) -> Ref {
|
||||
// Generates a suitable SHA256 `efgh` configuration from x86 format.
|
||||
auto Tmp = _VZip(OpSize::i128Bit, OpSize::i64Bit, Src2, Src1);
|
||||
return _VRev64(OpSize::i128Bit, OpSize::i32Bit, Tmp);
|
||||
};
|
||||
|
||||
auto H0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0);
|
||||
Q0 = _Add(OpSize::i32Bit, Q0, H0);
|
||||
auto ABCD = shuffle_abcd(Dest, Src);
|
||||
auto EFGH = shuffle_efgh(Dest, Src);
|
||||
|
||||
auto A0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
|
||||
auto B0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 2);
|
||||
auto C0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto A1 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Q0, BitwiseAtLeastTwo(A0, B0, C0)), Sigma0(A0));
|
||||
// x86 uses only the bottom 64-bits of the key, so duplicate to match ARM64 semantics.
|
||||
auto Key = _VDupElement(OpSize::i128Bit, OpSize::i64Bit, XMM0, 0);
|
||||
|
||||
auto D0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2);
|
||||
auto E1 = _Add(OpSize::i32Bit, Q0, D0);
|
||||
auto A = _VSha256H(ABCD, EFGH, Key);
|
||||
auto B = _VSha256H2(EFGH, ABCD, Key);
|
||||
Result = shuffle_abcd(A, B);
|
||||
} else {
|
||||
const auto Ch = [this](Ref E, Ref F, Ref G) -> Ref {
|
||||
return _Xor(OpSize::i32Bit, _And(OpSize::i32Bit, E, F), _Andn(OpSize::i32Bit, G, E));
|
||||
};
|
||||
const auto Sigma0 = [this](Ref A) -> Ref {
|
||||
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 2)), A, ShiftType::ROR, 13),
|
||||
A, ShiftType::ROR, 22);
|
||||
};
|
||||
const auto Sigma1 = [this](Ref E) -> Ref {
|
||||
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, E, _Constant(OpSize::i32Bit, 6)), E, ShiftType::ROR, 11),
|
||||
E, ShiftType::ROR, 25);
|
||||
};
|
||||
|
||||
Ref Q1 = _Add(OpSize::i32Bit, Ch(E1, E0, F0), Sigma1(E1));
|
||||
auto E0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 1);
|
||||
auto F0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 0);
|
||||
auto G0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
|
||||
Ref Q0 = _Add(OpSize::i32Bit, Ch(E0, F0, G0), Sigma1(E0));
|
||||
|
||||
auto WK1 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, XMM0, 1);
|
||||
Q1 = _Add(OpSize::i32Bit, Q1, WK1);
|
||||
auto WK0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, XMM0, 0);
|
||||
Q0 = _Add(OpSize::i32Bit, Q0, WK0);
|
||||
|
||||
// Rematerialize G0. Costs a move but saves spilling, coming out ahead.
|
||||
G0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
|
||||
Q1 = _Add(OpSize::i32Bit, Q1, G0);
|
||||
auto H0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0);
|
||||
Q0 = _Add(OpSize::i32Bit, Q0, H0);
|
||||
|
||||
auto A2 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Q1, BitwiseAtLeastTwo(A1, A0, B0)), Sigma0(A1));
|
||||
auto A0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
|
||||
auto B0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 2);
|
||||
auto C0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto A1 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Q0, BitwiseAtLeastTwo(A0, B0, C0)), Sigma0(A0));
|
||||
|
||||
// Rematerialize C0. As with G0.
|
||||
C0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto E2 = _Add(OpSize::i32Bit, Q1, C0);
|
||||
auto D0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2);
|
||||
auto E1 = _Add(OpSize::i32Bit, Q0, D0);
|
||||
|
||||
auto Res3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, A2);
|
||||
auto Res2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, Res3, A1);
|
||||
auto Res1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, Res2, E2);
|
||||
auto Res0 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, Res1, E1);
|
||||
Ref Q1 = _Add(OpSize::i32Bit, Ch(E1, E0, F0), Sigma1(E1));
|
||||
|
||||
StoreResult(FPRClass, Op, Res0, OpSize::iInvalid);
|
||||
auto WK1 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, XMM0, 1);
|
||||
Q1 = _Add(OpSize::i32Bit, Q1, WK1);
|
||||
|
||||
// Rematerialize G0. Costs a move but saves spilling, coming out ahead.
|
||||
G0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
|
||||
Q1 = _Add(OpSize::i32Bit, Q1, G0);
|
||||
|
||||
auto A2 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Q1, BitwiseAtLeastTwo(A1, A0, B0)), Sigma0(A1));
|
||||
|
||||
// Rematerialize C0. As with G0.
|
||||
C0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto E2 = _Add(OpSize::i32Bit, Q1, C0);
|
||||
|
||||
auto Res3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, A2);
|
||||
auto Res2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, Res3, A1);
|
||||
auto Res1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, Res2, E2);
|
||||
Result = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, Res1, E1);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
|
||||
|
||||
@@ -135,6 +135,7 @@ Ref OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateOF(IR::OpSize SrcSize, Ref Res, Ref Src1, Ref Src2, bool Sub) {
|
||||
LOGMAN_THROW_A_FMT(SrcSize >= IR::OpSize::i8Bit && SrcSize <= IR::OpSize::i64Bit, "Invalid size");
|
||||
const auto OpSize = SrcSize == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
const uint64_t SignBit = IR::OpSizeAsBits(SrcSize) - 1;
|
||||
Ref Anded = nullptr;
|
||||
|
||||
@@ -20,7 +20,6 @@ constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDisp
|
||||
{0x32, 2, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{0x34, 3, &OpDispatchBuilder::UnimplementedOp},
|
||||
|
||||
{0x3F, 1, &OpDispatchBuilder::ThunkOp},
|
||||
{0x40, 16, &OpDispatchBuilder::CMOVOp},
|
||||
{0x6E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVBetweenGPR_FPR, OpDispatchBuilder::VectorOpType::MMX>},
|
||||
{0x6F, 1, &OpDispatchBuilder::MOVQMMXOp},
|
||||
@@ -144,8 +143,11 @@ constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDisp
|
||||
{0xFD, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, OpSize::i16Bit>},
|
||||
{0xFE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, OpSize::i32Bit>},
|
||||
|
||||
#ifndef _WIN32
|
||||
// FEX reserved instructions
|
||||
{0x37, 1, &OpDispatchBuilder::CallbackReturnOp},
|
||||
{0x3F, 1, &OpDispatchBuilder::ThunkOp},
|
||||
#endif
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_SecondaryRepModTables[] = {
|
||||
|
||||
@@ -2104,17 +2104,28 @@ Ref OpDispatchBuilder::CVTFPR_To_GPRImpl(OpcodeArgs, Ref Src, IR::OpSize SrcElem
|
||||
// Source Element size is determined by instruction
|
||||
const auto GPRSize = OpSizeFromDst(Op);
|
||||
|
||||
if (HostRoundingMode) {
|
||||
Src = _Vector_FToI(SrcElementSize, SrcElementSize, Src, Round_Host);
|
||||
}
|
||||
Ref Converted = _Float_ToGPR_ZS(GPRSize, SrcElementSize, Src);
|
||||
if (CTX->HostFeatures.SupportsFRINTTS) {
|
||||
// When we have FRINTTS, this is a two-step process. First, we round to the
|
||||
// right integer (where _Vector_FToISized matches x86 semantics), then just
|
||||
// convert that to a GPR.
|
||||
Src = _Vector_FToISized(SrcElementSize, SrcElementSize, Src, HostRoundingMode, GPRSize);
|
||||
return _Float_ToGPR_ZS(GPRSize, SrcElementSize, Src);
|
||||
} else {
|
||||
// When we lack hardware support, we need a bit of a convoluted sequence of
|
||||
// fixups before before and after conversion to emulate x86 semantics.
|
||||
if (HostRoundingMode) {
|
||||
Src = _Vector_FToI(SrcElementSize, SrcElementSize, Src, Round_Host);
|
||||
}
|
||||
|
||||
bool Dst32 = GPRSize == OpSize::i32Bit;
|
||||
Ref MaxI = Dst32 ? _Constant(0x80000000) : _Constant(0x8000000000000000);
|
||||
Ref MaxF = LoadAndCacheNamedVectorConstant(SrcElementSize, (SrcElementSize == OpSize::i32Bit) ?
|
||||
(Dst32 ? NAMED_VECTOR_CVTMAX_F32_I32 : NAMED_VECTOR_CVTMAX_F32_I64) :
|
||||
(Dst32 ? NAMED_VECTOR_CVTMAX_F64_I32 : NAMED_VECTOR_CVTMAX_F64_I64));
|
||||
return _Select(GPRSize, SrcElementSize, CondClassType {FEXCore::IR::COND_FGT}, MaxF, Src, Converted, MaxI);
|
||||
Ref Converted = _Float_ToGPR_ZS(GPRSize, SrcElementSize, Src);
|
||||
|
||||
bool Dst32 = GPRSize == OpSize::i32Bit;
|
||||
Ref MaxI = Dst32 ? _Constant(0x80000000) : _Constant(0x8000000000000000);
|
||||
Ref MaxF = LoadAndCacheNamedVectorConstant(SrcElementSize, (SrcElementSize == OpSize::i32Bit) ?
|
||||
(Dst32 ? NAMED_VECTOR_CVTMAX_F32_I32 : NAMED_VECTOR_CVTMAX_F32_I64) :
|
||||
(Dst32 ? NAMED_VECTOR_CVTMAX_F64_I32 : NAMED_VECTOR_CVTMAX_F64_I64));
|
||||
return _Select(GPRSize, SrcElementSize, CondClassType {FEXCore::IR::COND_FGT}, MaxF, Src, Converted, MaxI);
|
||||
}
|
||||
}
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
|
||||
@@ -2169,25 +2180,39 @@ template void OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, false>(
|
||||
|
||||
Ref OpDispatchBuilder::Vector_CVT_Float_To_Int32Impl(OpcodeArgs, IR::OpSize DstSize, Ref Src, IR::OpSize SrcSize, IR::OpSize SrcElementSize,
|
||||
bool HostRoundingMode, bool ZeroUpperHalf) {
|
||||
if (HostRoundingMode) {
|
||||
Src = _Vector_FToI(SrcSize, SrcElementSize, Src, Round_Host);
|
||||
}
|
||||
|
||||
OpSize OverflowConstSize = ZeroUpperHalf && SrcElementSize == OpSize::i64Bit ? DstSize / 2 : DstSize;
|
||||
Ref MaxI = LoadAndCacheNamedVectorConstant(OverflowConstSize, NAMED_VECTOR_CVTMAX_I32);
|
||||
Ref Converted {}, Cmp {};
|
||||
if (SrcElementSize == OpSize::i64Bit) {
|
||||
Ref MaxF = LoadAndCacheNamedVectorConstant(SrcSize, NAMED_VECTOR_CVTMAX_F64_I32);
|
||||
Converted = _Vector_F64ToI32(DstSize, Src, Round_Towards_Zero, ZeroUpperHalf);
|
||||
|
||||
Cmp = _VFCMPGT(SrcSize, OpSize::i64Bit, MaxF, Src);
|
||||
Cmp = _VUShrNI(DstSize, OpSize::i64Bit, Cmp, 32);
|
||||
if (CTX->HostFeatures.SupportsFRINTTS && SrcSize != OpSize::i256Bit) {
|
||||
// If we have FRINTS, this is the usual 2-step
|
||||
Src = _Vector_FToISized(SrcSize, SrcElementSize, Src, HostRoundingMode, OpSize::i32Bit);
|
||||
Ref Dst = _Vector_FToZS(SrcSize, SrcElementSize, Src);
|
||||
if (SrcElementSize == OpSize::i32Bit) {
|
||||
// Return 32-bit result as-is
|
||||
return Dst;
|
||||
} else {
|
||||
// Down step from 64-bit ints to 32-bit ints
|
||||
return _VUShrNI(DstSize, SrcElementSize, Dst, 0);
|
||||
}
|
||||
} else {
|
||||
Ref MaxF = LoadAndCacheNamedVectorConstant(DstSize, NAMED_VECTOR_CVTMAX_F32_I32);
|
||||
Converted = _Vector_FToZS(DstSize, OpSize::i32Bit, Src);
|
||||
Cmp = _VFCMPGT(DstSize, OpSize::i32Bit, MaxF, Src);
|
||||
// Otherwise, we have to do all the fixups, but vectorized.
|
||||
if (HostRoundingMode) {
|
||||
Src = _Vector_FToI(SrcSize, SrcElementSize, Src, Round_Host);
|
||||
}
|
||||
|
||||
OpSize OverflowConstSize = ZeroUpperHalf && SrcElementSize == OpSize::i64Bit ? DstSize / 2 : DstSize;
|
||||
Ref MaxI = LoadAndCacheNamedVectorConstant(OverflowConstSize, NAMED_VECTOR_CVTMAX_I32);
|
||||
Ref Converted {}, Cmp {};
|
||||
if (SrcElementSize == OpSize::i64Bit) {
|
||||
Ref MaxF = LoadAndCacheNamedVectorConstant(SrcSize, NAMED_VECTOR_CVTMAX_F64_I32);
|
||||
Converted = _Vector_F64ToI32(DstSize, Src, Round_Towards_Zero, ZeroUpperHalf);
|
||||
|
||||
Cmp = _VFCMPGT(SrcSize, OpSize::i64Bit, MaxF, Src);
|
||||
Cmp = _VUShrNI(DstSize, OpSize::i64Bit, Cmp, 32);
|
||||
} else {
|
||||
Ref MaxF = LoadAndCacheNamedVectorConstant(DstSize, NAMED_VECTOR_CVTMAX_F32_I32);
|
||||
Converted = _Vector_FToZS(DstSize, OpSize::i32Bit, Src);
|
||||
Cmp = _VFCMPGT(DstSize, OpSize::i32Bit, MaxF, Src);
|
||||
}
|
||||
return _VBSL(DstSize, Cmp, Converted, MaxI);
|
||||
}
|
||||
return _VBSL(DstSize, Cmp, Converted, MaxI);
|
||||
}
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
|
||||
@@ -3949,6 +3974,7 @@ void OpDispatchBuilder::PTestOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VTESTOpImpl(OpSize SrcSize, IR::OpSize ElementSize, Ref Src1, Ref Src2) {
|
||||
LOGMAN_THROW_A_FMT(ElementSize >= IR::OpSize::i8Bit && ElementSize <= IR::OpSize::i64Bit, "Invalid size");
|
||||
const auto ElementSizeInBits = IR::OpSizeAsBits(ElementSize);
|
||||
const auto MaskConstant = uint64_t {1} << (ElementSizeInBits - 1);
|
||||
|
||||
@@ -4650,8 +4676,7 @@ void OpDispatchBuilder::VPBLENDDOp(OpcodeArgs) {
|
||||
return;
|
||||
}
|
||||
if (Selector == 0xFF && Is256Bit) {
|
||||
Ref Result = Is256Bit ? Src2 : _VMov(OpSize::i128Bit, Src2);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResult(FPRClass, Op, Src2, OpSize::iInvalid);
|
||||
return;
|
||||
}
|
||||
// The only bits we care about from the 8-bit immediate for 128-bit operations
|
||||
@@ -5072,7 +5097,7 @@ void OpDispatchBuilder::VPGATHER(OpcodeArgs) {
|
||||
// Only loads two 32-bit elements in to the lower 64-bits of the first destination.
|
||||
// Bits [255:65] all become zero.
|
||||
Result = _VMov(OpSize::i64Bit, Result);
|
||||
} else if (Is128Bit) {
|
||||
} else {
|
||||
Result = _VMov(OpSize::i128Bit, Result);
|
||||
}
|
||||
} else {
|
||||
|
||||
@@ -9,6 +9,7 @@ $end_info$
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/Core/Addressing.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
@@ -124,29 +125,17 @@ void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
auto zeroed_exponent = _Select(COND_EQ, absolute, zero, zero, adjusted_exponent);
|
||||
auto upper = _Or(OpSize::i64Bit, sign, zeroed_exponent);
|
||||
|
||||
Ref ConvertedData = _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, shifted);
|
||||
ConvertedData = _VInsElement(OpSize::i128Bit, OpSize::i64Bit, 1, 0, ConvertedData, _VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, upper));
|
||||
Ref ConvertedData = _VLoadTwoGPRs(shifted, upper);
|
||||
_PushStack(ConvertedData, Data, ReadWidth, false);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FST(OpcodeArgs, IR::OpSize Width) {
|
||||
// Ref Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
// FIXME: Is TSO relevant for x87?
|
||||
const auto SourceSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
AddressMode A = DecodeAddress(Op, Op->Dest, MemoryAccessType::DEFAULT, false);
|
||||
|
||||
// Index scale is a power of 2?
|
||||
LOGMAN_THROW_A_FMT(A.IndexScale > 0 && (A.IndexScale & (A.IndexScale - 1)) == 0, "Invalid index scale");
|
||||
A = SelectAddressMode(this, A, CTX->GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, false, false, Width);
|
||||
_StoreStackMem(SourceSize, Width, A.Base, A.Index, OpSize::iInvalid, A.IndexType, A.IndexScale, /*Float=*/true);
|
||||
|
||||
Ref Addr = A.Base ? A.Base : _Constant(0);
|
||||
if (A.Index) {
|
||||
Ref ScaledIndex = A.Index;
|
||||
if (A.IndexScale > 1) {
|
||||
ScaledIndex = _Lshl(A.AddrSize, ScaledIndex, _Constant(std::log2(A.IndexScale)));
|
||||
}
|
||||
Addr = _Add(A.AddrSize, Addr, ScaledIndex);
|
||||
}
|
||||
|
||||
_StoreStackMem(OpSize::i128Bit, Width, Addr, _Constant(A.Offset), /*Float=*/true);
|
||||
if (Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) {
|
||||
_PopStackDestroy();
|
||||
}
|
||||
@@ -550,8 +539,7 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
|
||||
auto low = _Constant(~0ULL);
|
||||
auto high = _Constant(0xFFFF);
|
||||
Ref Mask = _VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, low);
|
||||
Mask = _VInsGPR(OpSize::i128Bit, OpSize::i64Bit, 1, Mask, high);
|
||||
Ref Mask = _VLoadTwoGPRs(low, high);
|
||||
const auto StoreSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
for (int i = 0; i < 7; ++i) {
|
||||
Ref Reg = _LoadMem(FPRClass, OpSize::i128Bit, Mem, _Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
|
||||
@@ -103,27 +103,6 @@ void OpDispatchBuilder::FILDF64(OpcodeArgs) {
|
||||
_PushStack(ConvertedData, Data, ReadWidth, false);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FSTF64(OpcodeArgs, IR::OpSize Width) {
|
||||
AddressMode A = DecodeAddress(Op, Op->Dest, MemoryAccessType::DEFAULT, false);
|
||||
|
||||
// Index scale is a power of 2?
|
||||
LOGMAN_THROW_A_FMT(A.IndexScale > 0 && (A.IndexScale & (A.IndexScale - 1)) == 0, "Invalid index scale");
|
||||
|
||||
Ref Addr = A.Base ? A.Base : _Constant(0);
|
||||
if (A.Index) {
|
||||
Ref ScaledIndex = A.Index;
|
||||
if (A.IndexScale > 1) {
|
||||
ScaledIndex = _Lshl(A.AddrSize, ScaledIndex, _Constant(std::log2(A.IndexScale)));
|
||||
}
|
||||
Addr = _Add(A.AddrSize, Addr, ScaledIndex);
|
||||
}
|
||||
|
||||
_StoreStackMem(OpSize::i64Bit, Width, Addr, _Constant(A.Offset), /*Float=*/true);
|
||||
if (Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) {
|
||||
_PopStackDestroy();
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FISTF64(OpcodeArgs, bool Truncate) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
|
||||
@@ -260,6 +260,7 @@ auto BaseOpsLambda = []() consteval {
|
||||
{0xFE, 1, X86InstInfo{"PADDD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0xFF, 1, X86InstInfo{"UD0", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
|
||||
#ifndef _WIN32
|
||||
// FEX reserved instructions
|
||||
// Unused x86 encoding instruction.
|
||||
|
||||
@@ -267,6 +268,7 @@ auto BaseOpsLambda = []() consteval {
|
||||
|
||||
// This was originally used by VIA to jump to its alternative instruction set. Used for OP_THUNK
|
||||
{0x3F, 1, X86InstInfo{"ALTINST", TYPE_INST, FLAGS_BLOCK_END | FLAGS_NO_OVERLAY | FLAGS_SETS_RIP, 0, nullptr}},
|
||||
#endif
|
||||
};
|
||||
|
||||
GenerateTable(&Table.at(0), TwoByteOpTable, std::size(TwoByteOpTable));
|
||||
|
||||
@@ -343,14 +343,14 @@ constexpr InstFlagType FLAGS_MODRM = (1ULL << 16);
|
||||
// x87
|
||||
constexpr InstFlagType FLAGS_POP = (1ULL << 20);
|
||||
|
||||
// Whether or not the instruction has a VEX prefix for the first source operand
|
||||
constexpr InstFlagType FLAGS_VEX_1ST_SRC = (1ULL << 21);
|
||||
// Whether or not the instruction has a VEX prefix for the second source operand
|
||||
constexpr InstFlagType FLAGS_VEX_2ND_SRC = (1ULL << 22);
|
||||
// Whether or not the instruction has a VEX prefix for the destination
|
||||
constexpr InstFlagType FLAGS_VEX_DST = (1ULL << 23);
|
||||
// Whether or not the instruction has a VEX prefix for the dest, first, or second source.
|
||||
constexpr InstFlagType FLAGS_VEX_SRC_MASK = (0b11ULL << 21);
|
||||
constexpr InstFlagType FLAGS_VEX_NO_OPERAND = (0b00ULL << 21);
|
||||
constexpr InstFlagType FLAGS_VEX_DST = (0b01ULL << 21);
|
||||
constexpr InstFlagType FLAGS_VEX_1ST_SRC = (0b10ULL << 21);
|
||||
constexpr InstFlagType FLAGS_VEX_2ND_SRC = (0b11ULL << 21);
|
||||
// Whether or not the instruction has a VSIB byte
|
||||
constexpr InstFlagType FLAGS_VEX_VSIB = (1ULL << 24);
|
||||
constexpr InstFlagType FLAGS_VEX_VSIB = (1ULL << 23);
|
||||
|
||||
constexpr InstFlagType FLAGS_SIZE_DST_OFF = 58;
|
||||
constexpr InstFlagType FLAGS_SIZE_SRC_OFF = FLAGS_SIZE_DST_OFF + 3;
|
||||
@@ -440,7 +440,7 @@ struct X86InstInfo {
|
||||
}
|
||||
};
|
||||
|
||||
static_assert(std::is_trivial<X86InstInfo>::value, "X86InstInfo needs to be trivial");
|
||||
static_assert(std::is_trivially_copyable_v<X86InstInfo>);
|
||||
|
||||
constexpr size_t MAX_PRIMARY_TABLE_SIZE = 256;
|
||||
constexpr size_t MAX_SECOND_TABLE_SIZE = 256;
|
||||
|
||||
@@ -159,7 +159,7 @@ struct NodeWrapperBase final {
|
||||
operator==(const NodeWrapperBase<Type>&, const NodeWrapperBase<Type>&) = default;
|
||||
};
|
||||
|
||||
static_assert(std::is_trivial_v<NodeWrapperBase<OrderedNode>>);
|
||||
static_assert(std::is_trivially_copyable_v<NodeWrapperBase<OrderedNode>>);
|
||||
|
||||
static_assert(sizeof(NodeWrapperBase<OrderedNode>) == sizeof(uint32_t));
|
||||
|
||||
@@ -355,7 +355,7 @@ private:
|
||||
}
|
||||
};
|
||||
|
||||
static_assert(std::is_trivial_v<OrderedNode>);
|
||||
static_assert(std::is_trivially_constructible_v<OrderedNode>);
|
||||
static_assert(std::is_trivially_copyable_v<OrderedNode>);
|
||||
static_assert(offsetof(OrderedNode, Header) == 0);
|
||||
static_assert(sizeof(OrderedNode) == (sizeof(OrderedNodeHeader) + sizeof(uint32_t)));
|
||||
@@ -439,7 +439,7 @@ struct TypeDefinition final {
|
||||
operator==(const TypeDefinition&, const TypeDefinition&) = default;
|
||||
};
|
||||
|
||||
static_assert(std::is_trivial_v<TypeDefinition>);
|
||||
static_assert(std::is_trivially_copyable_v<TypeDefinition>);
|
||||
|
||||
struct FenceType final {
|
||||
using value_type = uint8_t;
|
||||
@@ -642,7 +642,9 @@ static inline OpSize operator/(IR::OpSize Size, T Divisor) {
|
||||
}
|
||||
|
||||
static inline uint8_t NumElements(IR::OpSize RegisterSize, IR::OpSize ElementSize) {
|
||||
LOGMAN_THROW_A_FMT(RegisterSize != IR::OpSize::iInvalid && ElementSize != IR::OpSize::iInvalid, "Invalid Size");
|
||||
LOGMAN_THROW_A_FMT(RegisterSize != IR::OpSize::iInvalid && ElementSize != IR::OpSize::iInvalid && RegisterSize != IR::OpSize::iUnsized &&
|
||||
ElementSize != IR::OpSize::iUnsized,
|
||||
"Invalid Size");
|
||||
return IR::OpSizeToSize(RegisterSize) / IR::OpSizeToSize(ElementSize);
|
||||
}
|
||||
|
||||
|
||||
@@ -749,8 +749,8 @@
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"EmitValidation": [
|
||||
"Offset % IR::OpSizeToSize(RegisterSize) == 0",
|
||||
"RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i256Bit"
|
||||
"RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i256Bit",
|
||||
"Offset % IR::OpSizeToSize(RegisterSize) == 0"
|
||||
]
|
||||
},
|
||||
"VStoreNonTemporalPair OpSize:#RegisterSize, FPR:$ValueLow, FPR:$ValueHigh, GPR:$Addr, i8:$Offset": {
|
||||
@@ -761,8 +761,8 @@
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"EmitValidation": [
|
||||
"Offset % IR::OpSizeToSize(RegisterSize) == 0",
|
||||
"RegisterSize == FEXCore::IR::OpSize::i128Bit"
|
||||
"RegisterSize == FEXCore::IR::OpSize::i128Bit",
|
||||
"Offset % IR::OpSizeToSize(RegisterSize) == 0"
|
||||
]
|
||||
},
|
||||
"FPR = VLoadNonTemporal OpSize:#RegisterSize, GPR:$Addr, i8:$Offset": {
|
||||
@@ -773,8 +773,8 @@
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"EmitValidation": [
|
||||
"Offset % IR::OpSizeToSize(RegisterSize) == 0",
|
||||
"RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i256Bit"
|
||||
"RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i256Bit",
|
||||
"Offset % IR::OpSizeToSize(RegisterSize) == 0"
|
||||
]
|
||||
}
|
||||
},
|
||||
@@ -789,7 +789,7 @@
|
||||
"Dest = %Expected",
|
||||
"if (deref(%Addr) != %Expected) Dest = deref(%Addr)"
|
||||
],
|
||||
|
||||
"TiedSource": 0,
|
||||
"DestSize": "Size",
|
||||
"ImplicitFlagClobber": true,
|
||||
"EmitValidation": [
|
||||
@@ -1507,7 +1507,8 @@
|
||||
"DestSize": "Size",
|
||||
"TiedSource": 0,
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit",
|
||||
"(Width + lsb) <= IR::OpSizeAsBits(Size)"
|
||||
]
|
||||
},
|
||||
"GPR = Bfxil OpSize:#Size, u8:$Width, u8:$lsb, GPR:$Dest, GPR:$Src": {
|
||||
@@ -1519,7 +1520,8 @@
|
||||
"DestSize": "Size",
|
||||
"TiedSource": 0,
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit",
|
||||
"(Width + lsb) <= IR::OpSizeAsBits(Size)"
|
||||
]
|
||||
},
|
||||
"GPR = Bfe OpSize:#Size, u8:$Width, u8:$lsb, GPR:$Src": {
|
||||
@@ -1529,7 +1531,8 @@
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit",
|
||||
"(Width + lsb) <= IR::OpSizeAsBits(Size)"
|
||||
]
|
||||
},
|
||||
"GPR = Sbfe OpSize:#Size, u8:$Width, u8:$lsb, GPR:$Src": {
|
||||
@@ -1539,7 +1542,8 @@
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit",
|
||||
"(Width + lsb) <= IR::OpSizeAsBits(Size)"
|
||||
]
|
||||
},
|
||||
"GPR = NZCVSelect OpSize:#ResultSize, CondClass:$Cond, GPR:$TrueVal, GPR:$FalseVal": {
|
||||
@@ -2047,29 +2051,49 @@
|
||||
"FPR = VShlI OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
"ElementSize": "ElementSize",
|
||||
"EmitValidation": [
|
||||
"ElementSize >= FEXCore::IR::OpSize::i8Bit && ElementSize <= FEXCore::IR::OpSize::i64Bit",
|
||||
"BitShift > 0"
|
||||
]
|
||||
},
|
||||
"FPR = VUShrI OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
"ElementSize": "ElementSize",
|
||||
"EmitValidation": [
|
||||
"ElementSize >= FEXCore::IR::OpSize::i8Bit && ElementSize <= FEXCore::IR::OpSize::i64Bit",
|
||||
"BitShift > 0"
|
||||
]
|
||||
},
|
||||
"FPR = VUShraI OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$DestVector, FPR:$Vector, u8:$BitShift": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
"ElementSize": "ElementSize",
|
||||
"EmitValidation": [
|
||||
"ElementSize >= FEXCore::IR::OpSize::i8Bit && ElementSize <= FEXCore::IR::OpSize::i64Bit",
|
||||
"BitShift > 0 && BitShift <= IR::OpSizeAsBits(ElementSize)"
|
||||
]
|
||||
},
|
||||
"FPR = VSShrI OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
"ElementSize": "ElementSize",
|
||||
"EmitValidation": [
|
||||
"ElementSize >= FEXCore::IR::OpSize::i8Bit && ElementSize <= FEXCore::IR::OpSize::i64Bit",
|
||||
"BitShift > 0"
|
||||
]
|
||||
},
|
||||
|
||||
"FPR = VUShrNI OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift": {
|
||||
"TiedSource": 0,
|
||||
"Desc": "Unsigned shifts right each element and then narrows to the next lower element size",
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize >> 1"
|
||||
"ElementSize": "ElementSize >> 1",
|
||||
"EmitValidation": [
|
||||
"ElementSize >= FEXCore::IR::OpSize::i16Bit && ElementSize <= FEXCore::IR::OpSize::i64Bit",
|
||||
"BitShift <= IR::OpSizeAsBits(ElementSize)"
|
||||
]
|
||||
},
|
||||
|
||||
"FPR = VUShrNI2 OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper, u8:$BitShift": {
|
||||
@@ -2078,7 +2102,11 @@
|
||||
"Inserts results in to the high elements of the first argument"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize >> 1"
|
||||
"ElementSize": "ElementSize >> 1",
|
||||
"EmitValidation": [
|
||||
"ElementSize >= FEXCore::IR::OpSize::i16Bit && ElementSize <= FEXCore::IR::OpSize::i64Bit",
|
||||
"BitShift > 0 && BitShift <= IR::OpSizeAsBits(ElementSize)"
|
||||
]
|
||||
},
|
||||
"FPR = VSXTL OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": "Sign extends elements from the source element size to the next size up",
|
||||
@@ -2311,11 +2339,13 @@
|
||||
|
||||
"FPR = VFMin OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
"ElementSize": "ElementSize",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VFMax OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
"ElementSize": "ElementSize",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VMul OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
@@ -2417,6 +2447,7 @@
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VInsGPR OpSize:#RegisterSize, OpSize:#ElementSize, u8:$DestIdx, FPR:$DestVector, GPR:$Src": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
@@ -2428,6 +2459,7 @@
|
||||
"TmpVector <RegisterSize *2> = concat(Upper:Lower)",
|
||||
"Dest = TmpVector >> (ElementSize * Index * 8); // Or can be thought of `concat(&TmpVector[Index], i128)`"
|
||||
],
|
||||
"TiedSource": 1,
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
@@ -2507,6 +2539,7 @@
|
||||
"If the bit in the field is 1 then the corresponding bit is pulled from VectorTrue",
|
||||
"If the bit in the field is 0 then the corresponding bit is pulled from VectorFalse"
|
||||
],
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize"
|
||||
},
|
||||
|
||||
@@ -2590,6 +2623,12 @@
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VLoadTwoGPRs GPR:$Lower, GPR:$Upper": {
|
||||
"Desc": ["Moves two 64-bit registers to a vector register optimally"],
|
||||
"DestSize": "OpSize::i128Bit",
|
||||
"ElementSize": "OpSize::i64Bit"
|
||||
},
|
||||
|
||||
"FPR = Float_FromGPR_S OpSize:#DstElementSize, OpSize:$SrcElementSize, GPR:$Src": {
|
||||
"Desc": ["Scalar op: Converts signed GPR to Scalar float",
|
||||
"Zeroes the upper bits of the vector register"
|
||||
@@ -2658,6 +2697,14 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = Vector_FToISized OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, i1:$HostRound, OpSize:$IntSize": {
|
||||
"Desc": ["Vector op: Rounds float to sized integral",
|
||||
"Either host rounding or round-to-zero",
|
||||
"Rounding mode determined by argument"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = Vector_F64ToI32 OpSize:#RegisterSize, FPR:$Vector, RoundType:$Round, i1:$EnsureZeroUpperHalf": {
|
||||
"Desc": ["Vector op: Rounds 64-bit float to 32-bit integral with round mode",
|
||||
"Matches CVTPD2DQ/CVTTPD2DQ behaviour"
|
||||
@@ -2724,6 +2771,16 @@
|
||||
"Desc": "Does vector scalar VSha256U1 instruction",
|
||||
"DestSize": "FEXCore::IR::OpSize::i128Bit"
|
||||
},
|
||||
"FPR = VSha256H FPR:$Src1, FPR:$Src2, FPR:$Src3": {
|
||||
"Desc": "Does vector scalar VSha256H instruction",
|
||||
"DestSize": "FEXCore::IR::OpSize::i128Bit",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VSha256H2 FPR:$Src1, FPR:$Src2, FPR:$Src3": {
|
||||
"Desc": "Does vector scalar VSha256H2 instruction",
|
||||
"DestSize": "FEXCore::IR::OpSize::i128Bit",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"GPR = CRC32 GPR:$Src1, GPR:$Src2, OpSize:$SrcSize": {
|
||||
"Desc": ["CRC32 using polynomial 0x1EDC6F41"
|
||||
],
|
||||
@@ -2846,7 +2903,7 @@
|
||||
"HasSideEffects": true,
|
||||
"X87": true
|
||||
},
|
||||
"StoreStackMem OpSize:$SourceSize, OpSize:$StoreSize, GPR:$Addr, GPR:$Offset, i1:$Float": {
|
||||
"StoreStackMem OpSize:$SourceSize, OpSize:$StoreSize, GPR:$Addr, GPR:$Offset, OpSize:$Align, MemOffsetType:$OffsetType, u8:$OffsetScale, i1:$Float": {
|
||||
"Desc": [
|
||||
"Takes the top value off the x87 stack and stores it to memory.",
|
||||
"SourceSize is 128bit for F80 values, 64-bit for low precision.",
|
||||
|
||||
@@ -37,7 +37,7 @@ static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView* IR, uint64_t Arg) {
|
||||
*out << "#0x" << std::hex << Arg;
|
||||
*out << "#0x" << std::hex << Arg << std::dec;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
|
||||
@@ -60,6 +60,7 @@ public:
|
||||
#include <FEXCore/IR/IRDefines.inc>
|
||||
IRPair<IROp_Constant> _Constant(IR::OpSize Size, uint64_t Constant) {
|
||||
auto Op = AllocateOp<IROp_Constant, IROps::OP_CONSTANT>();
|
||||
LOGMAN_THROW_A_FMT(Size >= IR::OpSize::i8Bit && Size <= IR::OpSize::i64Bit, "Invalid size");
|
||||
uint64_t Mask = ~0ULL >> (64 - IR::OpSizeAsBits(Size));
|
||||
Op.first->Constant = (Constant & Mask);
|
||||
Op.first->Header.Size = Size;
|
||||
@@ -356,10 +357,10 @@ protected:
|
||||
// These could be combined with a little bit of work to be more efficient with memory usage. Isn't a big deal
|
||||
DualIntrusiveAllocatorThreadPool DualListData;
|
||||
|
||||
Ref InvalidNode;
|
||||
Ref InvalidNode {};
|
||||
Ref CurrentCodeBlock {};
|
||||
fextl::vector<Ref> CodeBlocks;
|
||||
uint64_t Entry;
|
||||
uint64_t Entry {};
|
||||
};
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -98,11 +98,11 @@ protected:
|
||||
DualIntrusiveAllocator(size_t Size)
|
||||
: MemorySize {Size} {}
|
||||
|
||||
uintptr_t Data;
|
||||
uintptr_t List;
|
||||
uintptr_t Data {};
|
||||
uintptr_t List {};
|
||||
size_t DataCurrentOffset {0};
|
||||
size_t ListCurrentOffset {0};
|
||||
size_t MemorySize;
|
||||
size_t MemorySize {};
|
||||
};
|
||||
|
||||
class DualIntrusiveAllocatorMalloc final : public DualIntrusiveAllocator {
|
||||
|
||||
@@ -70,7 +70,7 @@ void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl* ctx) {
|
||||
FEX_CONFIG_OPT(DisablePasses, O0);
|
||||
|
||||
if (!DisablePasses()) {
|
||||
InsertPass(CreateX87StackOptimizationPass(ctx->HostFeatures));
|
||||
InsertPass(CreateX87StackOptimizationPass(ctx->HostFeatures, ctx->GetGPROpSize()));
|
||||
InsertPass(CreateConstProp(ctx->HostFeatures.SupportsTSOImm9, &ctx->CPUID));
|
||||
InsertPass(CreateDeadFlagCalculationEliminination());
|
||||
}
|
||||
|
||||
@@ -80,7 +80,7 @@ public:
|
||||
void Finalize();
|
||||
|
||||
protected:
|
||||
FEXCore::HLE::SyscallHandler* SyscallHandler;
|
||||
FEXCore::HLE::SyscallHandler* SyscallHandler {};
|
||||
|
||||
private:
|
||||
using PassArrayType = fextl::vector<fextl::unique_ptr<Pass>>;
|
||||
|
||||
@@ -20,7 +20,7 @@ class RegisterAllocationData;
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool SupportsTSOImm9, const FEXCore::CPUIDEmu* CPUID);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadFlagCalculationEliminination();
|
||||
fextl::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass();
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateX87StackOptimizationPass(const FEXCore::HostFeatures&);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateX87StackOptimizationPass(const FEXCore::HostFeatures&, OpSize GPROpSize);
|
||||
|
||||
namespace Validation {
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateIRValidation();
|
||||
|
||||
@@ -24,6 +24,7 @@ $end_info$
|
||||
namespace FEXCore::IR {
|
||||
|
||||
uint64_t getMask(IROp_Header* Op) {
|
||||
LOGMAN_THROW_A_FMT(Op->Size >= IR::OpSize::i8Bit && Op->Size <= IR::OpSize::i64Bit, "Invalid mask size");
|
||||
uint64_t NumBits = IR::OpSizeAsBits(Op->Size);
|
||||
return (~0ULL) >> (64 - NumBits);
|
||||
}
|
||||
@@ -92,8 +93,8 @@ private:
|
||||
return InlineIf(IREmit, CurrentIR, CodeNode, IROp, Index, Filter);
|
||||
}
|
||||
|
||||
void InlineMemImmediate(IREmitter* IREmit, const IRListView& IR, Ref CodeNode, IROp_Header* IROp, OrderedNodeWrapper Offset,
|
||||
MemOffsetType OffsetType, const size_t Offset_Index, uint8_t& OffsetScale, bool TSO) {
|
||||
void InlineMemImmediate(IREmitter* IREmit, const IRListView& IR, Ref CodeNode, IR::RegisterClassType RegisterClass, IROp_Header* IROp,
|
||||
OrderedNodeWrapper Offset, MemOffsetType OffsetType, const size_t Offset_Index, uint8_t& OffsetScale, bool TSO) {
|
||||
uint64_t Imm {};
|
||||
if (OffsetType != MEM_OFFSET_SXTX || !IREmit->IsValueConstant(Offset, &Imm)) {
|
||||
return;
|
||||
@@ -107,6 +108,9 @@ private:
|
||||
IsSIMM9 &= (SupportsTSOImm9 || !TSO);
|
||||
|
||||
// Extended offsets for regular loadstore only.
|
||||
LOGMAN_THROW_A_FMT(IROp->Size >= IR::OpSize::i8Bit && IROp->Size <= (RegisterClass == GPRClass ? IR::OpSize::i64Bit : IR::OpSize::i256Bit),
|
||||
"Invalid "
|
||||
"size");
|
||||
bool IsExtended = (Imm & (IR::OpSizeToSize(IROp->Size) - 1)) == 0 && Imm / IR::OpSizeToSize(IROp->Size) <= 4095;
|
||||
IsExtended &= !TSO;
|
||||
|
||||
@@ -419,6 +423,7 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
uint64_t Constant;
|
||||
if (IREmit->IsValueConstant(Op->Src, &Constant)) {
|
||||
LOGMAN_THROW_A_FMT(IROp->Size >= IR::OpSize::i8Bit && IROp->Size <= IR::OpSize::i64Bit, "Invalid size");
|
||||
// SBFE of a constant can be converted to a constant.
|
||||
uint64_t SourceMask = Op->Width == 64 ? ~0ULL : ((1ULL << Op->Width) - 1);
|
||||
uint64_t DestSizeInBits = IR::OpSizeAsBits(IROp->Size);
|
||||
@@ -603,11 +608,41 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
|
||||
case OP_ADC:
|
||||
case OP_ADCWITHFLAGS:
|
||||
case OP_STORECONTEXT:
|
||||
case OP_RMIFNZCV: {
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
break;
|
||||
}
|
||||
case OP_STORECONTEXT: {
|
||||
// For i128Bit, we won't see a normal Constant to inline, but as a special
|
||||
// case we can replace with a 2x64-bit store which can use inline zeroes.
|
||||
if (IROp->Size == OpSize::i128Bit) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
auto Header = IREmit->GetOpHeader(IROp->Args[0]);
|
||||
const auto MAX_STP_OFFSET = (252 * 4);
|
||||
|
||||
if (Op->Offset <= MAX_STP_OFFSET && Header->Op == OP_LOADNAMEDVECTORCONSTANT) {
|
||||
auto Const = Header->C<IR::IROp_LoadNamedVectorConstant>();
|
||||
|
||||
if (Const->Constant == IR::NamedVectorConstant::NAMED_VECTOR_ZERO) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
Ref Zero = IREmit->_Constant(0);
|
||||
Ref STP = IREmit->_StoreContextPair(IR::OpSize::i64Bit, GPRClass, Zero, Zero, Op->Offset);
|
||||
IREmit->Remove(CodeNode);
|
||||
|
||||
// XXX: This works around InlineConstant not having an associated
|
||||
// register class, else we'd just do InlineConstant above.
|
||||
Ref InlineZero = IREmit->_InlineConstant(0);
|
||||
IREmit->ReplaceNodeArgument(STP, 0, InlineZero);
|
||||
IREmit->ReplaceNodeArgument(STP, 1, InlineZero);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_CONDADDNZCV:
|
||||
case OP_CONDSUBNZCV: {
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
@@ -661,28 +696,28 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
|
||||
case OP_LOADMEM: {
|
||||
auto Op = IROp->CW<IR::IROp_LoadMem>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, Op->Class, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
break;
|
||||
}
|
||||
case OP_STOREMEM: {
|
||||
auto Op = IROp->CW<IR::IROp_StoreMem>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, Op->Class, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, Op->Value_Index);
|
||||
break;
|
||||
}
|
||||
case OP_PREFETCH: {
|
||||
auto Op = IROp->CW<IR::IROp_Prefetch>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, GPRClass, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
break;
|
||||
}
|
||||
case OP_LOADMEMTSO: {
|
||||
auto Op = IROp->CW<IR::IROp_LoadMemTSO>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, true);
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, Op->Class, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, true);
|
||||
break;
|
||||
}
|
||||
case OP_STOREMEMTSO: {
|
||||
auto Op = IROp->CW<IR::IROp_StoreMemTSO>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, true);
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, Op->Class, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, true);
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, Op->Value_Index);
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -25,8 +25,8 @@ public:
|
||||
|
||||
private:
|
||||
|
||||
BitSet<uint64_t> NodeIsLive;
|
||||
OrderedNode* EntryBlock;
|
||||
BitSet<uint64_t> NodeIsLive {};
|
||||
OrderedNode* EntryBlock {};
|
||||
fextl::unordered_map<IR::NodeID, BlockInfo> OffsetToBlockMap;
|
||||
size_t MaxNodes {};
|
||||
|
||||
|
||||
@@ -248,8 +248,8 @@ private:
|
||||
return nullptr;
|
||||
};
|
||||
|
||||
PhysicalRegister DecodeSRAReg(const IROp_Header* IROp, Ref Node) {
|
||||
RegisterClassType Class;
|
||||
PhysicalRegister DecodeSRAReg(const IROp_Header* IROp) {
|
||||
RegisterClassType Class {};
|
||||
uint8_t Reg {};
|
||||
|
||||
uint8_t FlagOffset = Classes[GPRFixedClass.Val].Count - 2;
|
||||
@@ -260,7 +260,6 @@ private:
|
||||
Class = Op->Class;
|
||||
Reg = Op->Reg;
|
||||
} else if (IROp->Op == OP_STOREREGISTER) {
|
||||
LOGMAN_THROW_A_FMT(IROp->Op == OP_STOREREGISTER, "node is SRA");
|
||||
const IROp_StoreRegister* Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
|
||||
Class = Op->Class;
|
||||
@@ -520,7 +519,7 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
// each register, used below. Since we initialized Class->Available,
|
||||
// RegToSSA is otherwise undefined so we can stash our temps there.
|
||||
if (auto Node = DecodeSRANode(IROp, CodeNode); Node != nullptr) {
|
||||
auto Reg = DecodeSRAReg(IROp, Node);
|
||||
auto Reg = DecodeSRAReg(IROp);
|
||||
|
||||
PreferredReg[IR->GetID(Node).Value] = Reg;
|
||||
GetClass(Reg)->RegToSSA[Reg.Reg] = CodeNode;
|
||||
@@ -559,7 +558,7 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
// assumed by the forward pass. Do not reset it.
|
||||
|
||||
// SourcesNextUses is read backwards, this tracks the index
|
||||
unsigned SourceIndex = SourcesNextUses.size();
|
||||
int64_t SourceIndex = SourcesNextUses.size();
|
||||
|
||||
// Forward pass: Assign registers, spilling as we go.
|
||||
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
|
||||
@@ -567,7 +566,7 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
|
||||
// Static registers must be consistent at SRA load/store. Evict to ensure.
|
||||
if (auto Node = DecodeSRANode(IROp, CodeNode); Node != nullptr) {
|
||||
auto Reg = DecodeSRAReg(IROp, Node);
|
||||
auto Reg = DecodeSRAReg(IROp);
|
||||
RegisterClass* Class = &Classes[Reg.Class];
|
||||
|
||||
if (!(Class->Available & (1u << Reg.Reg))) {
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "FEXCore/Utils/LogManager.h"
|
||||
#include "Interface/Core/Interpreter/Fallbacks/FallbackOpHandler.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
@@ -5,8 +6,9 @@
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "FEXCore/Utils/Profiler.h"
|
||||
#include "FEXCore/Utils/MathUtils.h"
|
||||
#include "FEXCore/Core/HostFeatures.h"
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/Addressing.h"
|
||||
|
||||
#include <array>
|
||||
#include <cstddef>
|
||||
@@ -20,7 +22,7 @@
|
||||
// and apply the operations in a block of code. Once the block finishes, we emit the necessary operations
|
||||
// that we recorded onto the virtual stack. This allows us to save a lot of code movement
|
||||
// to and from stack registers, top management and valid flags. It also allows us to
|
||||
// perform memcpy optimizations like the one performed in STORESTACKMEMORY.
|
||||
// perform memcpy optimizations like the one performed in STORESTACKMEM.
|
||||
//
|
||||
// By default we run on the fast path - i.e. we assume all values are in the stack and we have a complete
|
||||
// stack overview. However, if we encounter a value that's not in the virtual stack - maybe it was added
|
||||
@@ -147,8 +149,9 @@ private:
|
||||
|
||||
class X87StackOptimization final : public Pass {
|
||||
public:
|
||||
X87StackOptimization(const FEXCore::HostFeatures& Features)
|
||||
: Features(Features) {
|
||||
X87StackOptimization(const FEXCore::HostFeatures& Features, OpSize GPROpSize)
|
||||
: Features(Features)
|
||||
, GPROpSize(GPROpSize) {
|
||||
FEX_CONFIG_OPT(ReducedPrecision, X87REDUCEDPRECISION);
|
||||
ReducedPrecisionMode = ReducedPrecision;
|
||||
}
|
||||
@@ -156,11 +159,97 @@ public:
|
||||
|
||||
private:
|
||||
const FEXCore::HostFeatures& Features;
|
||||
const OpSize GPROpSize;
|
||||
bool ReducedPrecisionMode;
|
||||
|
||||
// Helpers
|
||||
Ref RotateRight8(uint32_t V, Ref Amount);
|
||||
|
||||
void F80SplitStore_Helper(const IROp_StoreStackMem* Op, Ref StackNode) {
|
||||
Ref AddrNode = IR->GetNode(Op->Addr);
|
||||
Ref Offset = IR->GetNode(Op->Offset);
|
||||
OpSize Align = Op->Align;
|
||||
MemOffsetType OffsetType = Op->OffsetType;
|
||||
uint8_t OffsetScale = Op->OffsetScale;
|
||||
|
||||
IREmit->_StoreMem(FPRClass, OpSize::i64Bit, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
auto Upper = IREmit->_VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, StackNode, 1);
|
||||
|
||||
// Store the Upper part of the register (the remaining 2 bytes) into memory.
|
||||
AddressMode A {.Base = AddrNode,
|
||||
.Index = Op->Offset.IsInvalid() ? nullptr : Offset,
|
||||
.IndexType = MEM_OFFSET_SXTX,
|
||||
.IndexScale = OffsetScale,
|
||||
.Offset = 8,
|
||||
.AddrSize = OpSize::i64Bit};
|
||||
A = SelectAddressMode(IREmit, A, GPROpSize, Features.SupportsTSOImm9, false, false, OpSize::i16Bit);
|
||||
IREmit->_StoreMem(GPRClass, OpSize::i16Bit, Upper, A.Base, A.Index, OpSize::i64Bit, MEM_OFFSET_SXTX, A.IndexScale);
|
||||
}
|
||||
|
||||
void StoreStackMem_Helper(const IROp_StoreStackMem* Op, Ref StackNode) {
|
||||
Ref AddrNode = IR->GetNode(Op->Addr);
|
||||
Ref Offset = IR->GetNode(Op->Offset);
|
||||
OpSize Align = Op->Align;
|
||||
MemOffsetType OffsetType = Op->OffsetType;
|
||||
uint8_t OffsetScale = Op->OffsetScale;
|
||||
|
||||
// Normal Precision Mode
|
||||
switch (Op->StoreSize) {
|
||||
case OpSize::i32Bit:
|
||||
case OpSize::i64Bit: {
|
||||
StackNode = IREmit->_F80CVT(Op->StoreSize, StackNode);
|
||||
IREmit->_StoreMem(FPRClass, Op->StoreSize, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
break;
|
||||
}
|
||||
|
||||
case OpSize::f80Bit: {
|
||||
if (Features.SupportsSVE128 || Features.SupportsSVE256) {
|
||||
AddressMode A {.Base = AddrNode,
|
||||
.Index = Op->Offset.IsInvalid() ? nullptr : Offset,
|
||||
.IndexType = MEM_OFFSET_SXTX,
|
||||
.IndexScale = OffsetScale,
|
||||
.AddrSize = OpSize::i64Bit};
|
||||
AddrNode = LoadEffectiveAddress(IREmit, A, GPROpSize, false);
|
||||
IREmit->_StoreMemX87SVEOptPredicate(OpSize::i128Bit, OpSize::i16Bit, StackNode, AddrNode);
|
||||
} else { // 80bit requires split-store
|
||||
F80SplitStore_Helper(Op, StackNode);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: ERROR_AND_DIE_FMT("Unsupported x87 size");
|
||||
}
|
||||
}
|
||||
|
||||
// Performs a store to memory from a value the stack passed in as StackNode.
|
||||
// This is the version dealing with the reduced precision case.
|
||||
void StoreStackMem_Reduced_Helper(const IROp_StoreStackMem* Op, Ref StackNode) {
|
||||
Ref AddrNode = IR->GetNode(Op->Addr);
|
||||
Ref Offset = IR->GetNode(Op->Offset);
|
||||
OpSize Align = Op->Align;
|
||||
MemOffsetType OffsetType = Op->OffsetType;
|
||||
uint8_t OffsetScale = Op->OffsetScale;
|
||||
|
||||
switch (Op->StoreSize) {
|
||||
case OpSize::i32Bit: {
|
||||
StackNode = IREmit->_Float_FToF(OpSize::i32Bit, OpSize::i64Bit, StackNode);
|
||||
[[fallthrough]];
|
||||
}
|
||||
case OpSize::i64Bit: {
|
||||
IREmit->_StoreMem(FPRClass, Op->StoreSize, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
break;
|
||||
}
|
||||
|
||||
// 80bit requires split-store
|
||||
case OpSize::f80Bit: {
|
||||
StackNode = IREmit->_F80CVTTo(StackNode, OpSize::i64Bit);
|
||||
F80SplitStore_Helper(Op, StackNode);
|
||||
break;
|
||||
}
|
||||
default: ERROR_AND_DIE_FMT("Unsupported x87 size");
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
// Helper to check if a Ref is a Zero constant
|
||||
bool IsZero(Ref Node) {
|
||||
auto Header = IR->GetOp<IR::IROp_Header>(Node);
|
||||
@@ -172,7 +261,6 @@ private:
|
||||
return Const->Constant == 0;
|
||||
}
|
||||
|
||||
|
||||
// Handles a Unary operation.
|
||||
// Takes the op we are handling, the Node for the reduced precision case and the node for the normal case.
|
||||
// Depending on the type of Op64, we might need to pass a couple of extra constant arguments, this happens
|
||||
@@ -242,12 +330,12 @@ private:
|
||||
|
||||
// Cache for Constants
|
||||
// ConstantPoll[i] has IREmit->_Constant(i);
|
||||
std::array<Ref, 8> ConstantPool;
|
||||
std::array<Ref, 8> ConstantPool {};
|
||||
Ref GetConstant(ssize_t Offset);
|
||||
|
||||
// Cached value for Top
|
||||
// If slowpath is false, then TopCache is nullptr.
|
||||
std::array<Ref, 8> TopOffsetCache;
|
||||
std::array<Ref, 8> TopOffsetCache {};
|
||||
// Are we on the slow path?
|
||||
// Once we enter the slow path, we never come out.
|
||||
// This just simplifies the code atm. If there's a need to return to the fast path in the future
|
||||
@@ -257,7 +345,7 @@ private:
|
||||
bool SlowPath = false;
|
||||
// Keeping IREmitter not to pass arguments around
|
||||
IREmitter* IREmit = nullptr;
|
||||
IRListView* IR;
|
||||
IRListView* IR = nullptr;
|
||||
};
|
||||
|
||||
inline void X87StackOptimization::InvalidateCaches() {
|
||||
@@ -800,6 +888,9 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
Ref StackNode = SlowPath ? LoadStackValueAtOffset_Slow() : Value->StackDataNode;
|
||||
Ref AddrNode = CurrentIR.GetNode(Op->Addr);
|
||||
Ref Offset = CurrentIR.GetNode(Op->Offset);
|
||||
OpSize Align = Op->Align;
|
||||
MemOffsetType OffsetType = Op->OffsetType;
|
||||
uint8_t OffsetScale = Op->OffsetScale;
|
||||
|
||||
// On the fast path we can optimize memory copies.
|
||||
// If we are doing:
|
||||
@@ -811,51 +902,17 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
// or similar. As long as the source size and dest size are one and the same.
|
||||
// This will avoid any conversions between source and stack element size and conversion back.
|
||||
if (!SlowPath && Value->Source && Value->Source->first == Op->StoreSize && Value->InterpretAsFloat) {
|
||||
IREmit->_StoreMem(Value->InterpretAsFloat ? FPRClass : GPRClass, Op->StoreSize, Value->Source->second, AddrNode, Offset,
|
||||
OpSize::iInvalid, MEM_OFFSET_SXTX, 1);
|
||||
} else {
|
||||
if (ReducedPrecisionMode) {
|
||||
switch (Op->StoreSize) {
|
||||
case OpSize::i32Bit:
|
||||
case OpSize::i64Bit: {
|
||||
if (Op->StoreSize == OpSize::i32Bit) {
|
||||
StackNode = IREmit->_Float_FToF(OpSize::i32Bit, OpSize::i64Bit, StackNode);
|
||||
}
|
||||
IREmit->_StoreMem(FPRClass, Op->StoreSize, StackNode, AddrNode, Offset, OpSize::iInvalid, MEM_OFFSET_SXTX, 1);
|
||||
break;
|
||||
}
|
||||
case OpSize::f80Bit: {
|
||||
StackNode = IREmit->_F80CVTTo(StackNode, OpSize::i64Bit);
|
||||
IREmit->_StoreMem(FPRClass, OpSize::i64Bit, StackNode, AddrNode, Offset, OpSize::iInvalid, MEM_OFFSET_SXTX, 1);
|
||||
auto Upper = IREmit->_VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, StackNode, 1);
|
||||
auto NewOffset = IREmit->_Add(OpSize::i64Bit, Offset, GetConstant(8));
|
||||
IREmit->_StoreMem(GPRClass, OpSize::i16Bit, Upper, AddrNode, NewOffset, OpSize::i64Bit, MEM_OFFSET_SXTX, 1);
|
||||
break;
|
||||
}
|
||||
default: ERROR_AND_DIE_FMT("Unsupported x87 size");
|
||||
}
|
||||
} else { // !ReducedPrecisionMode
|
||||
if (Op->StoreSize != OpSize::f80Bit) { // if it's not 80bits then convert
|
||||
StackNode = IREmit->_F80CVT(Op->StoreSize, StackNode);
|
||||
}
|
||||
if (Op->StoreSize == OpSize::f80Bit) {
|
||||
if (Features.SupportsSVE128 || Features.SupportsSVE256) {
|
||||
if (!IsZero(Offset)) {
|
||||
AddrNode = IREmit->_Add(OpSize::i64Bit, AddrNode, Offset);
|
||||
}
|
||||
IREmit->_StoreMemX87SVEOptPredicate(OpSize::i128Bit, OpSize::i16Bit, StackNode, AddrNode);
|
||||
} else {
|
||||
// For X87 extended doubles, split before storing
|
||||
IREmit->_StoreMem(FPRClass, OpSize::i64Bit, StackNode, AddrNode, Offset, OpSize::iInvalid, MEM_OFFSET_SXTX, 1);
|
||||
auto Upper = IREmit->_VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, StackNode, 1);
|
||||
auto NewOffset = IREmit->_Add(OpSize::i64Bit, Offset, GetConstant(8));
|
||||
IREmit->_StoreMem(GPRClass, OpSize::i16Bit, Upper, AddrNode, NewOffset, OpSize::i64Bit, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
} else {
|
||||
IREmit->_StoreMem(FPRClass, Op->StoreSize, StackNode, AddrNode, Offset, OpSize::iInvalid, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
}
|
||||
IREmit->_StoreMem(Value->InterpretAsFloat ? FPRClass : GPRClass, Op->StoreSize, Value->Source->second, AddrNode, Offset, Align,
|
||||
OffsetType, OffsetScale);
|
||||
break;
|
||||
}
|
||||
|
||||
if (ReducedPrecisionMode) {
|
||||
StoreStackMem_Reduced_Helper(Op, StackNode);
|
||||
break;
|
||||
}
|
||||
|
||||
StoreStackMem_Helper(Op, StackNode);
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -987,7 +1044,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
|
||||
case OP_INCSTACKTOP: {
|
||||
if (SlowPath) {
|
||||
UpdateTopForPush_Slow();
|
||||
UpdateTopForPop_Slow();
|
||||
} else {
|
||||
StackData.rotate(false);
|
||||
}
|
||||
@@ -996,7 +1053,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
|
||||
case OP_DECSTACKTOP: {
|
||||
if (SlowPath) {
|
||||
UpdateTopForPop_Slow();
|
||||
UpdateTopForPush_Slow();
|
||||
} else {
|
||||
StackData.rotate(true);
|
||||
}
|
||||
@@ -1045,7 +1102,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
return;
|
||||
}
|
||||
|
||||
fextl::unique_ptr<Pass> CreateX87StackOptimizationPass(const FEXCore::HostFeatures& Features) {
|
||||
return fextl::make_unique<X87StackOptimization>(Features);
|
||||
fextl::unique_ptr<Pass> CreateX87StackOptimizationPass(const FEXCore::HostFeatures& Features, OpSize GPROpSize) {
|
||||
return fextl::make_unique<X87StackOptimization>(Features, GPROpSize);
|
||||
}
|
||||
} // namespace FEXCore::IR
|
||||
@@ -119,14 +119,6 @@ inline uint32_t GetRmReg(uint32_t Instr) {
|
||||
return (Instr >> RM_OFFSET) & REGISTER_MASK;
|
||||
}
|
||||
|
||||
|
||||
FEXCORE_TELEMETRY_STATIC_INIT(SplitLock, TYPE_HAS_SPLIT_LOCKS);
|
||||
FEXCORE_TELEMETRY_STATIC_INIT(SplitLock16B, TYPE_16BYTE_SPLIT);
|
||||
FEXCORE_TELEMETRY_STATIC_INIT(Cas16Tear, TYPE_CAS_16BIT_TEAR);
|
||||
FEXCORE_TELEMETRY_STATIC_INIT(Cas32Tear, TYPE_CAS_32BIT_TEAR);
|
||||
FEXCORE_TELEMETRY_STATIC_INIT(Cas64Tear, TYPE_CAS_64BIT_TEAR);
|
||||
FEXCORE_TELEMETRY_STATIC_INIT(Cas128Tear, TYPE_CAS_128BIT_TEAR);
|
||||
|
||||
static void ClearICache(void* Begin, std::size_t Length) {
|
||||
__builtin___clear_cache(static_cast<char*>(Begin), static_cast<char*>(Begin) + Length);
|
||||
}
|
||||
@@ -367,7 +359,7 @@ static bool RunCASPAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg1, uint3
|
||||
|
||||
// Check for Split lock across a cacheline
|
||||
if ((Addr & 63) > 56) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock, 1);
|
||||
FEXCORE_TELEMETRY_SET(TYPE_HAS_SPLIT_LOCKS, 1);
|
||||
if (StrictSplitLockMutex && !Lock.has_value()) {
|
||||
Lock.emplace(StrictSplitLockMutex);
|
||||
}
|
||||
@@ -375,7 +367,7 @@ static bool RunCASPAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg1, uint3
|
||||
|
||||
uint64_t AlignmentMask = 0b1111;
|
||||
if ((Addr & AlignmentMask) > 8) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock16B, 1);
|
||||
FEXCORE_TELEMETRY_SET(TYPE_16BYTE_SPLIT, 1);
|
||||
if (StrictSplitLockMutex && !Lock.has_value()) {
|
||||
Lock.emplace(StrictSplitLockMutex);
|
||||
}
|
||||
@@ -431,7 +423,7 @@ static bool RunCASPAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg1, uint3
|
||||
} else {
|
||||
// CAS managed to tear, we can't really solve this
|
||||
// Continue down the path to let the guest know values weren't expected
|
||||
FEXCORE_TELEMETRY_SET(Cas128Tear, 1);
|
||||
FEXCORE_TELEMETRY_SET(TYPE_CAS_128BIT_TEAR, 1);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -667,7 +659,7 @@ static uint16_t DoCAS16(uint16_t DesiredSrc, uint16_t ExpectedSrc, uint64_t Addr
|
||||
std::optional<FEXCore::Utils::SpinWaitLock::UniqueSpinMutex<uint32_t>> Lock {};
|
||||
|
||||
if ((Addr & 63) == 63) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock, 1);
|
||||
FEXCORE_TELEMETRY_SET(TYPE_HAS_SPLIT_LOCKS, 1);
|
||||
if (StrictSplitLockMutex && !Lock.has_value()) {
|
||||
Lock.emplace(StrictSplitLockMutex);
|
||||
}
|
||||
@@ -676,7 +668,7 @@ static uint16_t DoCAS16(uint16_t DesiredSrc, uint16_t ExpectedSrc, uint64_t Addr
|
||||
// 16 bit
|
||||
uint64_t AlignmentMask = 0b1111;
|
||||
if ((Addr & AlignmentMask) == 15) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock16B, 1);
|
||||
FEXCORE_TELEMETRY_SET(TYPE_16BYTE_SPLIT, 1);
|
||||
if (StrictSplitLockMutex && !Lock.has_value()) {
|
||||
Lock.emplace(StrictSplitLockMutex);
|
||||
}
|
||||
@@ -714,12 +706,11 @@ static uint16_t DoCAS16(uint16_t DesiredSrc, uint16_t ExpectedSrc, uint64_t Addr
|
||||
// CAS managed to tear, we can't really solve this
|
||||
// Continue down the path to let the guest know values weren't expected
|
||||
Tear = true;
|
||||
FEXCORE_TELEMETRY_SET(Cas16Tear, 1);
|
||||
FEXCORE_TELEMETRY_SET(TYPE_CAS_16BIT_TEAR, 1);
|
||||
}
|
||||
}
|
||||
|
||||
ActualLower = ExpectedLower;
|
||||
ActualUpper = ExpectedUpper;
|
||||
}
|
||||
|
||||
// If the bits changed that we were wanting to change then we have failed and can return
|
||||
@@ -943,7 +934,7 @@ static uint32_t DoCAS32(uint32_t DesiredSrc, uint32_t ExpectedSrc, uint64_t Addr
|
||||
std::optional<FEXCore::Utils::SpinWaitLock::UniqueSpinMutex<uint32_t>> Lock {};
|
||||
|
||||
if ((Addr & 63) > 60) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock, 1);
|
||||
FEXCORE_TELEMETRY_SET(TYPE_HAS_SPLIT_LOCKS, 1);
|
||||
if (StrictSplitLockMutex && !Lock.has_value()) {
|
||||
Lock.emplace(StrictSplitLockMutex);
|
||||
}
|
||||
@@ -952,7 +943,7 @@ static uint32_t DoCAS32(uint32_t DesiredSrc, uint32_t ExpectedSrc, uint64_t Addr
|
||||
// 32 bit
|
||||
uint64_t AlignmentMask = 0b1111;
|
||||
if ((Addr & AlignmentMask) > 12) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock16B, 1);
|
||||
FEXCORE_TELEMETRY_SET(TYPE_16BYTE_SPLIT, 1);
|
||||
if (StrictSplitLockMutex && !Lock.has_value()) {
|
||||
Lock.emplace(StrictSplitLockMutex);
|
||||
}
|
||||
@@ -1001,7 +992,7 @@ static uint32_t DoCAS32(uint32_t DesiredSrc, uint32_t ExpectedSrc, uint64_t Addr
|
||||
// CAS managed to tear, we can't really solve this
|
||||
// Continue down the path to let the guest know values weren't expected
|
||||
Tear = true;
|
||||
FEXCORE_TELEMETRY_SET(Cas32Tear, 1);
|
||||
FEXCORE_TELEMETRY_SET(TYPE_CAS_32BIT_TEAR, 1);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1175,7 +1166,7 @@ static uint64_t DoCAS64(uint64_t DesiredSrc, uint64_t ExpectedSrc, uint64_t Addr
|
||||
std::optional<FEXCore::Utils::SpinWaitLock::UniqueSpinMutex<uint32_t>> Lock {};
|
||||
|
||||
if ((Addr & 63) > 56) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock, 1);
|
||||
FEXCORE_TELEMETRY_SET(TYPE_HAS_SPLIT_LOCKS, 1);
|
||||
if (StrictSplitLockMutex && !Lock.has_value()) {
|
||||
Lock.emplace(StrictSplitLockMutex);
|
||||
}
|
||||
@@ -1184,7 +1175,7 @@ static uint64_t DoCAS64(uint64_t DesiredSrc, uint64_t ExpectedSrc, uint64_t Addr
|
||||
// 64bit
|
||||
uint64_t AlignmentMask = 0b1111;
|
||||
if ((Addr & AlignmentMask) > 8) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock16B, 1);
|
||||
FEXCORE_TELEMETRY_SET(TYPE_16BYTE_SPLIT, 1);
|
||||
if (StrictSplitLockMutex && !Lock.has_value()) {
|
||||
Lock.emplace(StrictSplitLockMutex);
|
||||
}
|
||||
@@ -1235,7 +1226,7 @@ static uint64_t DoCAS64(uint64_t DesiredSrc, uint64_t ExpectedSrc, uint64_t Addr
|
||||
// CAS managed to tear, we can't really solve this
|
||||
// Continue down the path to let the guest know values weren't expected
|
||||
Tear = true;
|
||||
FEXCORE_TELEMETRY_SET(Cas64Tear, 1);
|
||||
FEXCORE_TELEMETRY_SET(TYPE_CAS_64BIT_TEAR, 1);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <atomic>
|
||||
#include <chrono>
|
||||
#include <mutex>
|
||||
|
||||
@@ -73,7 +73,7 @@ void Shutdown(const fextl::string& ApplicationName) {
|
||||
for (size_t i = 0; i < TelemetryType::TYPE_LAST; ++i) {
|
||||
auto& Name = TelemetryNames.at(i);
|
||||
auto& Data = TelemetryValues.at(i);
|
||||
fextl::fmt::print(File, "{}: {}\n", Name, *Data);
|
||||
fextl::fmt::print(File, "{}: {}\n", Name, Data.load());
|
||||
}
|
||||
File.Flush();
|
||||
}
|
||||
|
||||
@@ -15,6 +15,7 @@
|
||||
#include <charconv>
|
||||
#include <optional>
|
||||
#include <stdint.h>
|
||||
#include <variant>
|
||||
|
||||
namespace FEXCore::Config {
|
||||
namespace Handler {
|
||||
@@ -114,9 +115,10 @@ namespace DefaultValues {
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
|
||||
namespace Type {
|
||||
using StringArrayType = fextl::list<fextl::string>;
|
||||
#define OPT_BASE(type, group, enum, json, default) using P(enum) = P(type);
|
||||
#define OPT_STR(group, enum, json, default) using P(enum) = fextl::string;
|
||||
#define OPT_STRARRAY(group, enum, json, default) OPT_STR(group, enum, json, default)
|
||||
#define OPT_STRARRAY(group, enum, json, default) using P(enum) = StringArrayType;
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
} // namespace Type
|
||||
#define FEX_CONFIG_OPT(name, enum) \
|
||||
@@ -137,7 +139,9 @@ FEX_DEFAULT_VISIBILITY const fextl::string& GetConfigDirectory(bool Global);
|
||||
FEX_DEFAULT_VISIBILITY const fextl::string& GetConfigFileLocation(bool Global = false);
|
||||
FEX_DEFAULT_VISIBILITY fextl::string GetApplicationConfig(const std::string_view Program, bool Global);
|
||||
|
||||
using LayerValue = fextl::list<fextl::string>;
|
||||
using LayerValue =
|
||||
std::variant< fextl::string, DefaultValues::Type::StringArrayType, uint8_t, int8_t, uint16_t, int16_t, uint32_t, int32_t, uint64_t, int64_t, bool >;
|
||||
|
||||
using LayerOptions = fextl::unordered_map<ConfigOption, LayerValue>;
|
||||
|
||||
class FEX_DEFAULT_VISIBILITY Layer {
|
||||
@@ -151,13 +155,16 @@ public:
|
||||
return OptionMap.find(Option) != OptionMap.end();
|
||||
}
|
||||
|
||||
std::optional<LayerValue*> All(ConfigOption Option) {
|
||||
std::optional<DefaultValues::Type::StringArrayType*> All(ConfigOption Option) {
|
||||
const auto it = OptionMap.find(Option);
|
||||
if (it == OptionMap.end()) {
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
return &it->second;
|
||||
auto& Value = it->second;
|
||||
LOGMAN_THROW_A_FMT(std::holds_alternative<DefaultValues::Type::StringArrayType>(Value), "Tried to get config of invalid type!");
|
||||
|
||||
return &std::get<DefaultValues::Type::StringArrayType>(Value);
|
||||
}
|
||||
|
||||
std::optional<fextl::string*> Get(ConfigOption Option) {
|
||||
@@ -166,31 +173,44 @@ public:
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
return &it->second.front();
|
||||
auto& Value = it->second;
|
||||
LOGMAN_THROW_A_FMT(std::holds_alternative<fextl::string>(Value), "Tried to get config of invalid type!");
|
||||
|
||||
return &std::get<fextl::string>(Value);
|
||||
}
|
||||
|
||||
// Set will overwrite the object with a fextl::string without tests.
|
||||
void Set(ConfigOption Option, const char* Data) {
|
||||
LOGMAN_THROW_A_FMT(Data != nullptr, "Data can't be null");
|
||||
OptionMap[Option].emplace_back(fextl::string(Data));
|
||||
OptionMap[Option].emplace<fextl::string>(fextl::string(Data));
|
||||
}
|
||||
|
||||
void Set(ConfigOption Option, std::string_view Data) {
|
||||
OptionMap[Option].emplace_back(fextl::string(Data));
|
||||
OptionMap[Option].emplace<fextl::string>(fextl::string(Data));
|
||||
}
|
||||
|
||||
void Set(ConfigOption Option, fextl::string Data) {
|
||||
OptionMap[Option].emplace_back(std::move(Data));
|
||||
OptionMap[Option].emplace<fextl::string>(std::move(Data));
|
||||
}
|
||||
|
||||
void Set(ConfigOption Option, std::optional<fextl::string> Data) {
|
||||
if (Data) {
|
||||
OptionMap[Option].emplace_back(std::move(*Data));
|
||||
OptionMap[Option].emplace<fextl::string>(std::move(*Data));
|
||||
}
|
||||
}
|
||||
|
||||
void EraseSet(ConfigOption Option, std::string_view Data) {
|
||||
Erase(Option);
|
||||
Set(Option, Data);
|
||||
// AppendStrArrayValue will append strings to its StringArrayType.
|
||||
// If the value was previously a different type, then throw an assert.
|
||||
void AppendStrArrayValue(ConfigOption Option, std::string_view Data) {
|
||||
auto it = OptionMap.find(Option);
|
||||
if (it == OptionMap.end()) {
|
||||
// If the option didn't exist as a StringArrayType yet, emplace it.
|
||||
it = OptionMap.emplace(Option, DefaultValues::Type::StringArrayType {}).first;
|
||||
}
|
||||
|
||||
auto& Value = it->second;
|
||||
LOGMAN_THROW_A_FMT(std::holds_alternative<DefaultValues::Type::StringArrayType>(Value), "Tried to get config of invalid type!");
|
||||
std::get<DefaultValues::Type::StringArrayType>(Value).emplace_back(Data);
|
||||
}
|
||||
|
||||
void Erase(ConfigOption Option) {
|
||||
@@ -220,60 +240,25 @@ FEX_DEFAULT_VISIBILITY fextl::string FindContainerPrefix();
|
||||
FEX_DEFAULT_VISIBILITY void AddLayer(fextl::unique_ptr<FEXCore::Config::Layer> _Layer);
|
||||
|
||||
FEX_DEFAULT_VISIBILITY bool Exists(ConfigOption Option);
|
||||
FEX_DEFAULT_VISIBILITY std::optional<LayerValue*> All(ConfigOption Option);
|
||||
FEX_DEFAULT_VISIBILITY std::optional<DefaultValues::Type::StringArrayType*> All(ConfigOption Option);
|
||||
FEX_DEFAULT_VISIBILITY std::optional<fextl::string*> Get(ConfigOption Option);
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void Set(ConfigOption Option, std::string_view Data);
|
||||
FEX_DEFAULT_VISIBILITY void Erase(ConfigOption Option);
|
||||
FEX_DEFAULT_VISIBILITY void EraseSet(ConfigOption Option, std::string_view Data);
|
||||
|
||||
template<typename T>
|
||||
class FEX_DEFAULT_VISIBILITY Value {
|
||||
public:
|
||||
// Single value type.
|
||||
template<typename TT = T>
|
||||
requires (!std::is_same_v<TT, fextl::string>)
|
||||
Value(FEXCore::Config::ConfigOption _Option, TT Default)
|
||||
: Option {_Option} {
|
||||
requires (std::is_fundamental_v<TT> || std::is_same_v<TT, fextl::string>)
|
||||
Value(FEXCore::Config::ConfigOption Option, TT Default) {
|
||||
ValueData = GetIfExists(Option, Default);
|
||||
}
|
||||
|
||||
template<typename TT = T>
|
||||
requires (std::is_same_v<TT, fextl::string>)
|
||||
Value(FEXCore::Config::ConfigOption _Option, TT Default)
|
||||
: Option {_Option} {
|
||||
requires (std::is_fundamental_v<TT> || std::is_same_v<TT, fextl::string>)
|
||||
Value(FEXCore::Config::ConfigOption Option, std::string_view Default) {
|
||||
ValueData = GetIfExists(Option, Default);
|
||||
GetListIfExists(Option, &AppendList);
|
||||
}
|
||||
|
||||
template<typename TT = T>
|
||||
requires (std::is_same_v<TT, fextl::string>)
|
||||
Value(FEXCore::Config::ConfigOption _Option, std::string_view Default)
|
||||
: Option {_Option} {
|
||||
ValueData = GetIfExists(Option, Default);
|
||||
GetListIfExists(Option, &AppendList);
|
||||
}
|
||||
|
||||
template<typename TT = T>
|
||||
requires (!std::is_same_v<TT, fextl::string>)
|
||||
Value(FEXCore::Config::ConfigOption _Option)
|
||||
: Option {_Option} {
|
||||
if (!FEXCore::Config::Exists(Option)) {
|
||||
ERROR_AND_DIE_FMT("FEXCore::Config::Value has no value");
|
||||
}
|
||||
|
||||
ValueData = Get(Option);
|
||||
}
|
||||
|
||||
template<typename TT = T>
|
||||
requires (std::is_same_v<TT, fextl::string>)
|
||||
Value(FEXCore::Config::ConfigOption _Option)
|
||||
: Option {_Option} {
|
||||
if (!FEXCore::Config::Exists(Option)) {
|
||||
ERROR_AND_DIE_FMT("FEXCore::Config::Value has no value");
|
||||
}
|
||||
|
||||
ValueData = GetIfExists(Option);
|
||||
GetListIfExists(Option, &AppendList);
|
||||
}
|
||||
|
||||
operator T() const {
|
||||
@@ -281,7 +266,7 @@ public:
|
||||
}
|
||||
|
||||
template<typename TT = T>
|
||||
requires (!std::is_same_v<TT, fextl::string>)
|
||||
requires (std::is_fundamental_v<TT>)
|
||||
T operator()() const {
|
||||
return ValueData;
|
||||
}
|
||||
@@ -292,22 +277,31 @@ public:
|
||||
return ValueData;
|
||||
}
|
||||
|
||||
template<typename TT = T>
|
||||
requires (!std::is_same_v<TT, DefaultValues::Type::StringArrayType>)
|
||||
Value<T>(T Value) {
|
||||
ValueData = std::move(Value);
|
||||
}
|
||||
fextl::list<T>& All() {
|
||||
return AppendList;
|
||||
|
||||
// Array value types.
|
||||
template<typename TT = T>
|
||||
requires (std::is_same_v<TT, DefaultValues::Type::StringArrayType>)
|
||||
Value(FEXCore::Config::ConfigOption Option, std::string_view) {
|
||||
GetListIfExists(Option, &ValueData);
|
||||
}
|
||||
|
||||
template<typename TT = T>
|
||||
requires (std::is_same_v<TT, DefaultValues::Type::StringArrayType>)
|
||||
DefaultValues::Type::StringArrayType& All() {
|
||||
return ValueData;
|
||||
}
|
||||
|
||||
private:
|
||||
FEXCore::Config::ConfigOption Option;
|
||||
T ValueData;
|
||||
fextl::list<T> AppendList;
|
||||
T ValueData {};
|
||||
|
||||
static T Get(FEXCore::Config::ConfigOption Option);
|
||||
static T GetIfExists(FEXCore::Config::ConfigOption Option, T Default);
|
||||
static T GetIfExists(FEXCore::Config::ConfigOption Option, std::string_view Default);
|
||||
|
||||
static void GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string>* List);
|
||||
static void GetListIfExists(FEXCore::Config::ConfigOption Option, DefaultValues::Type::StringArrayType* List);
|
||||
};
|
||||
} // namespace FEXCore::Config
|
||||
@@ -7,6 +7,7 @@
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/IntervalList.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/set.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
@@ -216,6 +217,15 @@ public:
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual void AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t GuestThunkEntrypoint) = 0;
|
||||
|
||||
/**
|
||||
* @brief Adds additional per-instruction granularity TSO enable/disable information for the given range.
|
||||
*
|
||||
* @param ValidRanges The set of address ranges covered by this information
|
||||
* @param Instructions The set of instruction addresses within the given ranges for which TSO should be enabled
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual void AddForceTSOInformation(const IntervalList<uint64_t>& ValidRanges, fextl::set<uint64_t>&& Instructions) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void RemoveForceTSOInformation(uint64_t Address, uint64_t Size) = 0;
|
||||
private:
|
||||
};
|
||||
|
||||
|
||||
@@ -92,12 +92,18 @@ struct CPUState {
|
||||
// Counts the nesting depth of program sections that cause signals to be deferred.
|
||||
NonAtomicRefCounter<uint64_t> DeferredSignalRefCount;
|
||||
|
||||
// PF/AF raw values. Really only a byte of each matters, but this layout
|
||||
// (32-bits and in the first 256 bytes) is necessary to use ldp/stp to
|
||||
// spill/fill these togethers efficiently.
|
||||
uint32_t pf_raw {};
|
||||
uint32_t af_raw {};
|
||||
|
||||
uint64_t rip {}; ///< Current core's RIP. May not be entirely accurate while JIT is active
|
||||
|
||||
// The high 128-bits of AVX registers when not being emulated by SVE256.
|
||||
uint64_t avx_high[16][2];
|
||||
|
||||
uint64_t rip {}; ///< Current core's RIP. May not be entirely accurate while JIT is active
|
||||
uint64_t gregs[16] {};
|
||||
uint64_t _pad {};
|
||||
XMMRegs xmm {};
|
||||
|
||||
// Raw segment register indexes
|
||||
@@ -110,8 +116,8 @@ struct CPUState {
|
||||
uint64_t gs_cached {};
|
||||
uint64_t fs_cached {};
|
||||
uint8_t flags[48] {};
|
||||
uint64_t pf_raw {};
|
||||
uint64_t af_raw {};
|
||||
uint64_t _pad1 {};
|
||||
uint64_t _pad2 {};
|
||||
uint64_t mm[8][2] {};
|
||||
|
||||
// 32bit x86 state
|
||||
|
||||
@@ -36,6 +36,7 @@ struct HostFeatures {
|
||||
bool SupportsAES256 {};
|
||||
bool SupportsSVEBitPerm {};
|
||||
bool SupportsCPUIndexInTPIDRRO {};
|
||||
bool SupportsFRINTTS {};
|
||||
|
||||
// Float exception behaviour
|
||||
bool SupportsAFP {};
|
||||
|
||||
@@ -197,7 +197,7 @@ private:
|
||||
bool ShouldClose {};
|
||||
bool IsValidHandle {};
|
||||
|
||||
FileHandleType Handle;
|
||||
FileHandleType Handle {};
|
||||
#ifndef _WIN32
|
||||
static constexpr int DEFAULT_USER_PERMS = S_IRWXU | S_IRWXG | S_IRWXO;
|
||||
|
||||
|
||||
+19
-2
@@ -6,6 +6,7 @@
|
||||
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
namespace FEXCore {
|
||||
template<typename SizeType>
|
||||
class IntervalList {
|
||||
public:
|
||||
@@ -66,6 +67,12 @@ public:
|
||||
FirstIt->End = End;
|
||||
}
|
||||
|
||||
void Insert(const IntervalList<SizeType>& Other) {
|
||||
for (const auto& Interval : Other.Intervals) {
|
||||
Insert(Interval);
|
||||
}
|
||||
}
|
||||
|
||||
void Remove(Interval Entry) {
|
||||
if (Entry.Offset == Entry.End) {
|
||||
return;
|
||||
@@ -121,7 +128,7 @@ public:
|
||||
Intervals.erase(EraseStartIt, EraseEndIt);
|
||||
}
|
||||
|
||||
QueryResult Query(SizeType Offset) {
|
||||
QueryResult Query(SizeType Offset) const {
|
||||
const auto It = std::upper_bound(Intervals.begin(), Intervals.end(), Offset, [](const auto& LHS, const auto& RHS) {
|
||||
return LHS < RHS.End;
|
||||
}); // Lowest offset interval that (maybe) overlaps with the query offset
|
||||
@@ -135,11 +142,21 @@ public:
|
||||
}
|
||||
}
|
||||
|
||||
bool Intersect(Interval Entry) {
|
||||
bool Intersect(Interval Entry) const {
|
||||
const auto It = std::upper_bound(Intervals.begin(), Intervals.end(), Entry, [](const auto& LHS, const auto& RHS) {
|
||||
return LHS.Offset < RHS.End;
|
||||
}); // Lowest offset interval that (maybe) overlaps with the query offset
|
||||
|
||||
return It != Intervals.end() && It->Offset < Entry.End;
|
||||
}
|
||||
|
||||
bool Contains(Interval Entry) const {
|
||||
const auto It = std::upper_bound(Intervals.begin(), Intervals.end(), Entry, [](const auto& LHS, const auto& RHS) {
|
||||
return LHS.Offset < RHS.End;
|
||||
}); // Lowest offset interval that (maybe) overlaps with the query offset
|
||||
|
||||
return It != Intervals.end() && It->Offset <= Entry.Offset && It->End >= Entry.End;
|
||||
}
|
||||
};
|
||||
|
||||
} // namespace FEXCore
|
||||
@@ -33,36 +33,12 @@ enum TelemetryType {
|
||||
};
|
||||
|
||||
#ifndef FEX_DISABLE_TELEMETRY
|
||||
class Value;
|
||||
|
||||
class Value final {
|
||||
public:
|
||||
Value() = default;
|
||||
Value(uint64_t Default)
|
||||
: Data {Default} {}
|
||||
|
||||
uint64_t operator*() const {
|
||||
return Data;
|
||||
}
|
||||
void operator=(uint64_t Value) {
|
||||
Data = Value;
|
||||
}
|
||||
void operator|=(uint64_t Value) {
|
||||
Data |= Value;
|
||||
}
|
||||
void operator++(int) {
|
||||
Data++;
|
||||
}
|
||||
|
||||
std::atomic<uint64_t>* GetAddr() {
|
||||
return &Data;
|
||||
}
|
||||
|
||||
private:
|
||||
std::atomic<uint64_t> Data;
|
||||
};
|
||||
using Value = std::atomic<uint64_t>;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY extern std::array<Value, FEXCore::Telemetry::TelemetryType::TYPE_LAST> TelemetryValues;
|
||||
// This returns the internal structure to the telemetry data structures
|
||||
// One must be careful with placing these in the hot path of code execution
|
||||
// It can be fairly costly, especially in the static version where it puts barriers in the code
|
||||
inline Value& GetTelemetryValue(TelemetryType Type) {
|
||||
return FEXCore::Telemetry::TelemetryValues[Type];
|
||||
}
|
||||
@@ -71,26 +47,28 @@ FEX_DEFAULT_VISIBILITY void Initialize();
|
||||
FEX_DEFAULT_VISIBILITY void Shutdown(const fextl::string& ApplicationName);
|
||||
|
||||
// Telemetry object declaration
|
||||
// This returns the internal structure to the telemetry data structures
|
||||
// One must be careful with placing these in the hot path of code execution
|
||||
// It can be fairly costly, especially in the static version where it puts barriers in the code
|
||||
#define FEXCORE_TELEMETRY_STATIC_INIT(Name, Type) \
|
||||
static FEXCore::Telemetry::Value& Name = FEXCore::Telemetry::GetTelemetryValue(FEXCore::Telemetry::Type)
|
||||
#define FEXCORE_TELEMETRY_INIT(Name, Type) FEXCore::Telemetry::Value& Name = FEXCore::Telemetry::GetTelemetryValue(FEXCore::Telemetry::Type)
|
||||
// Telemetry ALU operations
|
||||
// These are typically 3-4 instructions depending on what you're doing
|
||||
#define FEXCORE_TELEMETRY_SET(Name, Value) Name = Value
|
||||
#define FEXCORE_TELEMETRY_OR(Name, Value) Name |= Value
|
||||
#define FEXCORE_TELEMETRY_INC(Name) Name++
|
||||
#define FEXCORE_TELEMETRY_SET(Type, Value) \
|
||||
do { \
|
||||
auto& Name = FEXCore::Telemetry::TelemetryValues[FEXCore::Telemetry::Type]; \
|
||||
Name = Value; \
|
||||
} while (0)
|
||||
#define FEXCORE_TELEMETRY_OR(Type, Value) \
|
||||
do { \
|
||||
auto& Name = FEXCore::Telemetry::TelemetryValues[FEXCore::Telemetry::Type]; \
|
||||
Name |= Value; \
|
||||
} while (0)
|
||||
#define FEXCORE_TELEMETRY_INC(Type, Value) \
|
||||
do { \
|
||||
auto& Name = FEXCore::Telemetry::TelemetryValues[FEXCore::Telemetry::Type]; \
|
||||
Name++; \
|
||||
} while (0)
|
||||
|
||||
// Returns a pointer to std::atomic<uint64_t>. Can be useful if you are attempting to JIT telemetry accesses for debug purposes
|
||||
// Not recommended to do telemetry inside JIT code in production code
|
||||
#define FEXCORE_TELEMETRY_Addr(Name) Name->GetAddr()
|
||||
#else
|
||||
static inline void Initialize() {}
|
||||
static inline void Shutdown(const fextl::string& ApplicationName) {}
|
||||
|
||||
#define FEXCORE_TELEMETRY_STATIC_INIT(Name, Type)
|
||||
#define FEXCORE_TELEMETRY_INIT(Name, Type)
|
||||
#define FEXCORE_TELEMETRY(Name, Value) \
|
||||
do { \
|
||||
@@ -104,6 +82,5 @@ static inline void Shutdown(const fextl::string& ApplicationName) {}
|
||||
#define FEXCORE_TELEMETRY_INC(Name) \
|
||||
do { \
|
||||
} while (0)
|
||||
#define FEXCORE_TELEMETRY_Addr(Name) reinterpret_cast<std::atomic<uint64_t>*>(nullptr)
|
||||
#endif
|
||||
} // namespace FEXCore::Telemetry
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
#include <catch2/generators/catch_generators_range.hpp>
|
||||
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Utils/SpinWaitLock.h"
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
#include <chrono>
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
#include <catch2/generators/catch_generators_range.hpp>
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
#include <catch2/generators/catch_generators_range.hpp>
|
||||
#include <catch2/generators/catch_generators_random.hpp>
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "TestDisassembler.h"
|
||||
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "TestDisassembler.h"
|
||||
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
@@ -6,33 +7,24 @@
|
||||
using namespace ARMEmitter;
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ASIMD: Cryptographic AES") {
|
||||
if (false) {
|
||||
// vixl doesn't support these instructions.
|
||||
TEST_SINGLE(aese(VReg::v30, VReg::v29), "aese v30, v29");
|
||||
TEST_SINGLE(aesd(VReg::v30, VReg::v29), "aesd v30, v29");
|
||||
TEST_SINGLE(aesmc(VReg::v30, VReg::v29), "aesmc v30, v29");
|
||||
TEST_SINGLE(aesimc(VReg::v30, VReg::v29), "aesimc v30, v29");
|
||||
}
|
||||
TEST_SINGLE(aese(VReg::v30, VReg::v29), "aese v30.16b, v29.16b");
|
||||
TEST_SINGLE(aesd(VReg::v30, VReg::v29), "aesd v30.16b, v29.16b");
|
||||
TEST_SINGLE(aesmc(VReg::v30, VReg::v29), "aesmc v30.16b, v29.16b");
|
||||
TEST_SINGLE(aesimc(VReg::v30, VReg::v29), "aesimc v30.16b, v29.16b");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ASIMD: Cryptographic three-register SHA") {
|
||||
if (false) {
|
||||
// vixl doesn't support these instructions.
|
||||
TEST_SINGLE(sha1c(VReg::v30, SReg::s29, VReg::v28), "sha1c v30, s29, v28");
|
||||
TEST_SINGLE(sha1p(VReg::v30, SReg::s29, VReg::v28), "sha1p v30, s29, v28");
|
||||
TEST_SINGLE(sha1m(VReg::v30, SReg::s29, VReg::v28), "sha1m v30, s29, v28");
|
||||
TEST_SINGLE(sha1su0(VReg::v30, VReg::v29, VReg::v28), "sha1su0 v30, v29, v28");
|
||||
TEST_SINGLE(sha256h(VReg::v30, VReg::v29, VReg::v28), "sha256h v30, v29, v28");
|
||||
TEST_SINGLE(sha256h2(VReg::v30, VReg::v29, VReg::v28), "sha256h2 v30, v29, v28");
|
||||
TEST_SINGLE(sha256su1(VReg::v30, VReg::v29, VReg::v28), "sha256su1 v30, v29, v28");
|
||||
}
|
||||
TEST_SINGLE(sha1c(VReg::v30, SReg::s29, VReg::v28), "sha1c q30, s29, v28.4s");
|
||||
TEST_SINGLE(sha1p(VReg::v30, SReg::s29, VReg::v28), "sha1p q30, s29, v28.4s");
|
||||
TEST_SINGLE(sha1m(VReg::v30, SReg::s29, VReg::v28), "sha1m q30, s29, v28.4s");
|
||||
TEST_SINGLE(sha1su0(VReg::v30, VReg::v29, VReg::v28), "sha1su0 v30.4s, v29.4s, v28.4s");
|
||||
TEST_SINGLE(sha256h(VReg::v30, VReg::v29, VReg::v28), "sha256h q30, q29, v28.4s");
|
||||
TEST_SINGLE(sha256h2(VReg::v30, VReg::v29, VReg::v28), "sha256h2 q30, q29, v28.4s");
|
||||
TEST_SINGLE(sha256su1(VReg::v30, VReg::v29, VReg::v28), "sha256su1 v30.4s, v29.4s, v28.4s");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ASIMD: Cryptographic two-register SHA") {
|
||||
if (false) {
|
||||
// vixl doesn't support these instructions.
|
||||
TEST_SINGLE(sha1h(SReg::s30, SReg::s29), "sha1h s30, s29");
|
||||
TEST_SINGLE(sha1su1(VReg::v30, VReg::v29), "sha1su1 v30, v29");
|
||||
TEST_SINGLE(sha256su0(VReg::v30, VReg::v29), "sha256su0 v30, v29");
|
||||
}
|
||||
TEST_SINGLE(sha1h(SReg::s30, SReg::s29), "sha1h s30, s29");
|
||||
TEST_SINGLE(sha1su1(VReg::v30, VReg::v29), "sha1su1 v30.4s, v29.4s");
|
||||
TEST_SINGLE(sha256su0(VReg::v30, VReg::v29), "sha256su0 v30.4s, v29.4s");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ASIMD: Advanced SIMD table lookup") {
|
||||
TEST_SINGLE(tbl(QReg::q30, QReg::q26, QReg::q25), "tbl v30.16b, {v26.16b}, v25.16b");
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "TestDisassembler.h"
|
||||
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "TestDisassembler.h"
|
||||
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "TestDisassembler.h"
|
||||
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
@@ -3376,13 +3377,28 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE2 Histogram Computation - S
|
||||
TEST_SINGLE(histseg(ZReg::z30, ZReg::z29, ZReg::z28), "histseg z30.b, z29.b, z28.b");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE2 crypto unary operations") {
|
||||
// TODO: Implement in emitter.
|
||||
TEST_SINGLE(aesimc(ZReg::z7, ZReg::z7), "aesimc z7.b, z7.b");
|
||||
TEST_SINGLE(aesimc(ZReg::z31, ZReg::z31), "aesimc z31.b, z31.b");
|
||||
|
||||
TEST_SINGLE(aesmc(ZReg::z7, ZReg::z7), "aesmc z7.b, z7.b");
|
||||
TEST_SINGLE(aesmc(ZReg::z31, ZReg::z31), "aesmc z31.b, z31.b");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE2 crypto destructive binary operations") {
|
||||
// TODO: Implement in emitter.
|
||||
TEST_SINGLE(aesd(ZReg::z7, ZReg::z7, ZReg::z8), "aesd z7.b, z7.b, z8.b");
|
||||
TEST_SINGLE(aesd(ZReg::z30, ZReg::z30, ZReg::z31), "aesd z30.b, z30.b, z31.b");
|
||||
|
||||
TEST_SINGLE(aese(ZReg::z7, ZReg::z7, ZReg::z8), "aese z7.b, z7.b, z8.b");
|
||||
TEST_SINGLE(aese(ZReg::z30, ZReg::z30, ZReg::z31), "aese z30.b, z30.b, z31.b");
|
||||
|
||||
TEST_SINGLE(sm4e(ZReg::z7, ZReg::z7, ZReg::z8), "sm4e z7.s, z7.s, z8.s");
|
||||
TEST_SINGLE(sm4e(ZReg::z30, ZReg::z30, ZReg::z31), "sm4e z30.s, z30.s, z31.s");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE2 crypto constructive binary operations") {
|
||||
// TODO: Implement in emitter.
|
||||
TEST_SINGLE(sm4ekey(ZReg::z0, ZReg::z1, ZReg::z2), "sm4ekey z0.s, z1.s, z2.s");
|
||||
TEST_SINGLE(sm4ekey(ZReg::z29, ZReg::z30, ZReg::z31), "sm4ekey z29.s, z30.s, z31.s");
|
||||
|
||||
TEST_SINGLE(rax1(ZReg::z0, ZReg::z1, ZReg::z2), "rax1 z0.d, z1.d, z2.d");
|
||||
TEST_SINGLE(rax1(ZReg::z29, ZReg::z30, ZReg::z31), "rax1 z29.d, z30.d, z31.d");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: SVE: SVE BFloat16 floating-point dot product (indexed)") {
|
||||
// TODO: Implement in emitter.
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "TestDisassembler.h"
|
||||
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "TestDisassembler.h"
|
||||
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
|
||||
@@ -168,7 +168,7 @@ def HandleFunctionDeclCursor(Arch, Cursor):
|
||||
|
||||
def PrintFunctionDecls():
|
||||
for Decl in FunctionDecls:
|
||||
print("template<> struct fex_gen_config<{}> {{}};".format(Decl.Name))
|
||||
print("template<>\nstruct fex_gen_config<{}> {{}};".format(Decl.Name))
|
||||
|
||||
def FindClangArguments(OriginalArguments):
|
||||
AddedArguments = ["clang"]
|
||||
@@ -529,9 +529,9 @@ def main():
|
||||
BaseArgs.append(sys.argv[ArgIndex])
|
||||
|
||||
args_x86_64 = [
|
||||
"-I/usr/include/x86_64-linux-gnu",
|
||||
"-I/usr/x86_64-linux-gnu/include/c++/10/x86_64-linux-gnu/",
|
||||
"-I/usr/x86_64-linux-gnu/include/",
|
||||
"-isystem", "/usr/include/x86_64-linux-gnu",
|
||||
"-isystem", "/usr/x86_64-linux-gnu/include/c++/10/x86_64-linux-gnu/",
|
||||
"-isystem", "/usr/x86_64-linux-gnu/include/",
|
||||
"-O2",
|
||||
"--target=x86_64-linux-unknown",
|
||||
"-D_M_X86_64",
|
||||
|
||||
@@ -58,6 +58,7 @@ class HostFeatures(Flag) :
|
||||
FEATURE_TSO = (1 << 13)
|
||||
FEATURE_LRCPC = (1 << 14)
|
||||
FEATURE_LRCPC2 = (1 << 15)
|
||||
FEATURE_FRINTTS = (1 << 16)
|
||||
|
||||
HostFeaturesLookup = {
|
||||
"SVE128" : HostFeatures.FEATURE_SVE128,
|
||||
@@ -76,6 +77,7 @@ HostFeaturesLookup = {
|
||||
"TSO" : HostFeatures.FEATURE_TSO,
|
||||
"LRCPC" : HostFeatures.FEATURE_LRCPC,
|
||||
"LRCPC2" : HostFeatures.FEATURE_LRCPC2,
|
||||
"FRINTTS" : HostFeatures.FEATURE_FRINTTS,
|
||||
}
|
||||
|
||||
def GetHostFeatures(data):
|
||||
|
||||
@@ -659,21 +659,21 @@ def main():
|
||||
BaseArgs.append(sys.argv[ArgIndex])
|
||||
|
||||
args_x86_32 = [
|
||||
"-I/usr/i686-linux-gnu/include",
|
||||
"-isystem", "/usr/i686-linux-gnu/include",
|
||||
"-O2",
|
||||
"-m32",
|
||||
"--target=i686-linux-unknown",
|
||||
]
|
||||
|
||||
args_x86_64 = [
|
||||
"-I/usr/x86_64-linux-gnu/include",
|
||||
"-isystem", "/usr/x86_64-linux-gnu/include",
|
||||
"-O2",
|
||||
"--target=x86_64-linux-unknown",
|
||||
"-D_M_X86_64",
|
||||
]
|
||||
|
||||
args_aarch64 = [
|
||||
"-I/usr/aarch64-linux-gnu/include",
|
||||
"-isystem", "/usr/aarch64-linux-gnu/include",
|
||||
"-O2",
|
||||
"--target=aarch64-linux-unknown",
|
||||
"-D_M_ARM_64",
|
||||
|
||||
@@ -18,11 +18,7 @@ BigCoreIDs = {
|
||||
tuple([0x41, 0xd0b]): "cortex-a76",
|
||||
tuple([0x41, 0xd0d]): "cortex-a77",
|
||||
tuple([0x41, 0xd41]): "cortex-a78",
|
||||
tuple([0x41, 0xd41]):
|
||||
[ ["cortex-a78", "0.0"],
|
||||
["cortex-x1c", "14.0"], # Claim to be x1c as an alternative if available.
|
||||
["cortex-a78c", "9999.0"], # Doesn't exist in clang as of version 15.0
|
||||
],
|
||||
tuple([0x41, 0xd4b]): "cortex-a78c",
|
||||
tuple([0x41, 0xd44]): "cortex-x1",
|
||||
tuple([0x41, 0xd4c]):
|
||||
[ ["cortex-x1", "0.0"],
|
||||
@@ -36,8 +32,44 @@ BigCoreIDs = {
|
||||
[ ["cortex-x1", "0.0"],
|
||||
["cortex-x2", "14.0"],
|
||||
],
|
||||
tuple([0x41, 0xd4d]):
|
||||
[ ["cortex-a78", "0.0"],
|
||||
["cortex-a715", "17.0"],
|
||||
],
|
||||
tuple([0x41, 0xd81]):
|
||||
[ ["cortex-a78", "0.0"],
|
||||
["cortex-a720", "18.0"],
|
||||
],
|
||||
tuple([0x41, 0xd87]):
|
||||
[ ["cortex-a78", "0.0"],
|
||||
["cortex-a725", "19.0"],
|
||||
],
|
||||
tuple([0x41, 0xd85]):
|
||||
[ ["cortex-a78", "0.0"],
|
||||
["cortex-x925", "19.0"],
|
||||
],
|
||||
# Neoverse-N class
|
||||
tuple([0x41, 0xd0c]): "neoverse-n1",
|
||||
tuple([0x41, 0xd49]): "neoverse-n2",
|
||||
tuple([0x41, 0xd8e]):
|
||||
[ ["cortex-a78", "0.0"],
|
||||
["neoverse-n3", "19.0"],
|
||||
],
|
||||
|
||||
# Neoverse-V class
|
||||
tuple([0x41, 0xd40]): "neoverse-v1",
|
||||
tuple([0x41, 0xd4f]):
|
||||
[ ["cortex-a78", "0.0"],
|
||||
["neoverse-v2", "16.0"],
|
||||
],
|
||||
tuple([0x41, 0xd83]):
|
||||
[ ["cortex-a78", "0.0"],
|
||||
["neoverse-v3ae", "19.0"],
|
||||
],
|
||||
tuple([0x41, 0xd84]):
|
||||
[ ["cortex-a78", "0.0"],
|
||||
["neoverse-v3", "19.0"],
|
||||
],
|
||||
## Nvidia
|
||||
tuple([0x4e, 0x004]): "carmel", # Carmel
|
||||
# Qualcomm
|
||||
@@ -60,7 +92,10 @@ LittleCoreIDs = {
|
||||
[ ["cortex-a55", "0.0"],
|
||||
["cortex-a510", "14.0"],
|
||||
],
|
||||
|
||||
tuple([0x41, 0xd80]):
|
||||
[ ["cortex-a55", "0.0"],
|
||||
["cortex-a520", "18.0"],
|
||||
],
|
||||
# Qualcomm
|
||||
tuple([0x51, 0x801]): "cortex-a53", # Kryo 2xx Silver
|
||||
tuple([0x51, 0x803]): "cortex-a55", # Kryo 3xx Silver
|
||||
|
||||
@@ -18,6 +18,4 @@ endif()
|
||||
|
||||
add_library(${NAME} STATIC ${SRCS})
|
||||
target_link_libraries(${NAME} FEXCore_Base cpp-optparse tiny-json FEXHeaderUtils)
|
||||
target_include_directories(${NAME} PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/External/cpp-optparse/)
|
||||
target_include_directories(${NAME} PRIVATE ${CMAKE_BINARY_DIR}/generated)
|
||||
target_include_directories(${NAME} PRIVATE ${CMAKE_SOURCE_DIR}/External/xbyak/)
|
||||
@@ -31,8 +31,7 @@ namespace JSON {
|
||||
const json_t* json = FEX::JSON::CreateJSON(Data, Pool);
|
||||
|
||||
if (!json) {
|
||||
LogMan::Msg::EFmt("Couldn't create json");
|
||||
return;
|
||||
ERROR_AND_DIE_FMT("Failed to parse JSON from file '{}' - invalid JSON format", Config);
|
||||
}
|
||||
|
||||
const json_t* ConfigList = json_getProperty(json, "Config");
|
||||
@@ -47,12 +46,12 @@ namespace JSON {
|
||||
const char* ConfigString = json_getValue(ConfigItem);
|
||||
|
||||
if (!ConfigName) {
|
||||
LogMan::Msg::EFmt("Couldn't get config name");
|
||||
LogMan::Msg::EFmt("JSON file '{}': Couldn't get config name for an item", Config);
|
||||
return;
|
||||
}
|
||||
|
||||
if (!ConfigString) {
|
||||
LogMan::Msg::EFmt("Couldn't get ConfigString for '{}'", ConfigName);
|
||||
LogMan::Msg::EFmt("JSON file '{}': Couldn't get value for config item '{}'", Config, ConfigName);
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -76,8 +75,14 @@ static char* SaveLayerToJSON(char* JsonBuffer, const FEXCore::Config::Layer* Lay
|
||||
break;
|
||||
}
|
||||
}
|
||||
for (auto& var : it.second) {
|
||||
JsonBuffer = json_str(JsonBuffer, Name.data(), var.c_str());
|
||||
if (std::holds_alternative<fextl::string>(it.second)) {
|
||||
JsonBuffer = json_str(JsonBuffer, Name.data(), std::get<fextl::string>(it.second).c_str());
|
||||
} else if (std::holds_alternative<FEXCore::Config::DefaultValues::Type::StringArrayType>(it.second)) {
|
||||
for (auto& var : std::get<FEXCore::Config::DefaultValues::Type::StringArrayType>(it.second)) {
|
||||
JsonBuffer = json_str(JsonBuffer, Name.data(), var.c_str());
|
||||
}
|
||||
} else {
|
||||
LogMan::Msg::AFmt("Trying to store config with pre-converted type");
|
||||
}
|
||||
}
|
||||
return json_objClose(JsonBuffer);
|
||||
@@ -289,6 +294,10 @@ void EnvLoader::Load() {
|
||||
#define OPT_BASE(type, group, enum, json, default) \
|
||||
Value = GetVar(EnvMap, "FEX_" #enum); \
|
||||
if (Value.has_value()) Set(FEXCore::Config::ConfigOption::CONFIG_##enum, *Value);
|
||||
#define OPT_STRARRAY(group, enum, json, default) \
|
||||
Value = GetVar(EnvMap, "FEX_" #enum); \
|
||||
if (Value.has_value()) AppendStrArrayValue(FEXCore::Config::ConfigOption::CONFIG_##enum, *Value);
|
||||
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}
|
||||
|
||||
@@ -420,7 +429,7 @@ ApplicationNames GetApplicationNames(const fextl::vector<fextl::string>& Args, b
|
||||
}
|
||||
}
|
||||
|
||||
ProgramName = CurrentProgramName;
|
||||
ProgramName = std::move(CurrentProgramName);
|
||||
|
||||
// Past any wine program names
|
||||
break;
|
||||
|
||||
@@ -204,7 +204,7 @@ bool SetupClient(std::string_view InterpreterPath) {
|
||||
fextl::string RootFSPath = FEXServerClient::RequestRootFSPath(ServerFD);
|
||||
|
||||
//// If everything has passed then we can now update the rootfs path
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, RootFSPath);
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_ROOTFS, RootFSPath);
|
||||
}
|
||||
|
||||
return true;
|
||||
|
||||
@@ -8,14 +8,7 @@
|
||||
#include <FEXCore/Utils/StringUtils.h>
|
||||
|
||||
#ifdef _M_X86_64
|
||||
#define XBYAK64
|
||||
#define XBYAK_NO_EXCEPTION
|
||||
#include <FEXCore/fextl/list.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
#include <FEXCore/fextl/unordered_set.h>
|
||||
|
||||
#include <xbyak/xbyak.h>
|
||||
#include <xbyak/xbyak_util.h>
|
||||
#include "Common/X86Features.h"
|
||||
#endif
|
||||
|
||||
namespace FEX {
|
||||
@@ -437,6 +430,7 @@ static void OverrideFeatures(FEXCore::HostFeatures* Features, uint64_t ForceSVEW
|
||||
ENABLE_DISABLE_OPTION(SupportsFCMA, FCMA, FCMA);
|
||||
ENABLE_DISABLE_OPTION(SupportsFlagM, FlagM, FLAGM);
|
||||
ENABLE_DISABLE_OPTION(SupportsFlagM2, FlagM2, FLAGM2);
|
||||
ENABLE_DISABLE_OPTION(SupportsFRINTTS, FRINTTS, FRINTTS);
|
||||
ENABLE_DISABLE_OPTION(SupportsRPRES, RPRES, RPRES);
|
||||
ENABLE_DISABLE_OPTION(SupportsSVEBitPerm, SVEBITPERM, SVEBITPERM);
|
||||
ENABLE_DISABLE_OPTION(SupportsPreserveAllABI, PRESERVEALLABI, PRESERVEALLABI);
|
||||
@@ -486,6 +480,7 @@ FEXCore::HostFeatures FetchHostFeatures(FEX::CPUFeatures& Features, bool Support
|
||||
HostFeatures.SupportsFCMA = Features.Supports(CPUFeatures::Feature::FCMA);
|
||||
HostFeatures.SupportsFlagM = Features.Supports(CPUFeatures::Feature::FlagM);
|
||||
HostFeatures.SupportsFlagM2 = Features.Supports(CPUFeatures::Feature::FlagM2);
|
||||
HostFeatures.SupportsFRINTTS = Features.Supports(CPUFeatures::Feature::FRINTTS);
|
||||
HostFeatures.SupportsRPRES = Features.Supports(CPUFeatures::Feature::RPRES);
|
||||
HostFeatures.SupportsSVEBitPerm = Features.Supports(CPUFeatures::Feature::SVE_BitPerm);
|
||||
|
||||
@@ -572,27 +567,17 @@ FEXCore::HostFeatures FetchHostFeatures(FEX::CPUFeatures& Features, bool Support
|
||||
}
|
||||
|
||||
#if defined(_M_X86_64) && !defined(VIXL_SIMULATOR)
|
||||
Xbyak::util::Cpu X86Features {};
|
||||
HostFeatures.SupportsAES = X86Features.has(Xbyak::util::Cpu::tAESNI);
|
||||
HostFeatures.SupportsCRC = X86Features.has(Xbyak::util::Cpu::tSSE42);
|
||||
HostFeatures.SupportsRAND = X86Features.has(Xbyak::util::Cpu::tRDRAND) && X86Features.has(Xbyak::util::Cpu::tRDSEED);
|
||||
FEX::X86::Features Feature {};
|
||||
HostFeatures.SupportsAES = Feature.Feat_aes;
|
||||
HostFeatures.SupportsCRC = Feature.Feat_crc;
|
||||
HostFeatures.SupportsRAND = Feature.Feat_rand;
|
||||
HostFeatures.SupportsRCPC = true;
|
||||
HostFeatures.SupportsTSOImm9 = true;
|
||||
HostFeatures.SupportsAVX = true;
|
||||
HostFeatures.SupportsSHA = X86Features.has(Xbyak::util::Cpu::tSHA);
|
||||
HostFeatures.SupportsPMULL_128Bit = X86Features.has(Xbyak::util::Cpu::tPCLMULQDQ);
|
||||
HostFeatures.SupportsAES256 = HostFeatures.SupportsAES && X86Features.has(Xbyak::util::Cpu::tVAES);
|
||||
|
||||
// xbyak doesn't know how to check for CLZero
|
||||
// First ensure we support a new enough extended CPUID function range
|
||||
|
||||
uint32_t data[4];
|
||||
Xbyak::util::Cpu::getCpuid(0x8000'0000, data);
|
||||
if (data[0] >= 0x8000'0008U) {
|
||||
// CLZero defined in 8000_00008_EBX[bit 0]
|
||||
Xbyak::util::Cpu::getCpuid(0x8000'0008, data);
|
||||
HostFeatures.SupportsCLZERO = data[1] & 1;
|
||||
}
|
||||
HostFeatures.SupportsSHA = Feature.Feat_rand;
|
||||
HostFeatures.SupportsPMULL_128Bit = Feature.Feat_pclmulqdq;
|
||||
HostFeatures.SupportsAES256 = Feature.Feat_aes;
|
||||
HostFeatures.SupportsCLZERO = Feature.Feat_clzero;
|
||||
|
||||
HostFeatures.SupportsAFP = true;
|
||||
HostFeatures.SupportsFloatExceptions = true;
|
||||
|
||||
@@ -0,0 +1,74 @@
|
||||
#pragma once
|
||||
#include <cstdint>
|
||||
|
||||
#ifdef _M_X86_64
|
||||
#include <cpuid.h>
|
||||
|
||||
namespace FEX::X86 {
|
||||
class Features final {
|
||||
public:
|
||||
Features() {
|
||||
cpuid_data data {};
|
||||
data = cpuid(0);
|
||||
|
||||
if (data.eax >= 1) {
|
||||
auto data_1 = cpuid(0x1);
|
||||
|
||||
Feat_aes = data_1.ecx & (1U << 25);
|
||||
Feat_crc = data_1.ecx & (1U << 20);
|
||||
Feat_rand = data_1.ecx & (1U << 30);
|
||||
Feat_pclmulqdq = data_1.ecx & (1U << 1);
|
||||
}
|
||||
|
||||
if (data.eax >= 7) {
|
||||
auto data_7 = cpuid(0x7);
|
||||
Feat_bmi1 = data_7.ebx & (1U << 3);
|
||||
Feat_bmi2 = data_7.ebx & (1U << 8);
|
||||
Feat_clwb = data_7.ebx & (1U << 24);
|
||||
Feat_rand &= data_7.ebx & (1U << 18);
|
||||
Feat_sha = data_7.ebx & (1U << 29);
|
||||
Feat_vaes = data_7.ecx & (1U << 9);
|
||||
Feat_pclmulqdq &= data_7.ecx & (1U << 10);
|
||||
}
|
||||
|
||||
data = cpuid(0x8000'0000U);
|
||||
if (data.eax >= 0x8000'0001U) {
|
||||
auto data_8000_0001 = cpuid(0x8000'0001U);
|
||||
Feat_3dnow = (data_8000_0001.edx >> 30) == 0b11;
|
||||
Feat_sse4a = data_8000_0001.ecx & (1U << 6);
|
||||
}
|
||||
|
||||
if (data.eax >= 0x8000'0008U) {
|
||||
auto data_8000_0008 = cpuid(0x8000'0008U);
|
||||
|
||||
Feat_clzero = data_8000_0008.ebx & 1;
|
||||
}
|
||||
}
|
||||
|
||||
// Features.
|
||||
bool Feat_3dnow {};
|
||||
bool Feat_sse4a {};
|
||||
bool Feat_bmi1 {};
|
||||
bool Feat_bmi2 {};
|
||||
bool Feat_clwb {};
|
||||
bool Feat_aes {};
|
||||
bool Feat_crc {};
|
||||
bool Feat_rand {};
|
||||
bool Feat_sha {};
|
||||
bool Feat_pclmulqdq {};
|
||||
bool Feat_vaes {};
|
||||
bool Feat_clzero {};
|
||||
|
||||
private:
|
||||
struct cpuid_data {
|
||||
uint32_t eax, ebx, ecx, edx;
|
||||
};
|
||||
|
||||
cpuid_data cpuid(uint32_t Function, uint32_t Leaf = 0) {
|
||||
cpuid_data data;
|
||||
__cpuid_count(Function, Leaf, data.eax, data.ebx, data.ecx, data.edx);
|
||||
return data;
|
||||
}
|
||||
};
|
||||
} // namespace FEX::X86
|
||||
#endif
|
||||
Submodule Source/Common/cpp-optparse updated: eab4212ae8...9f94388a33.
@@ -447,7 +447,18 @@ public:
|
||||
|
||||
for (auto& it : EnvConfigLookup) {
|
||||
if (auto Value = GetVar(it.first); Value) {
|
||||
Set(it.second, *Value);
|
||||
#define OPT_BASE(type, group, enum, json, default) // Nothing
|
||||
#define OPT_STRARRAY(group, enum, json, default) \
|
||||
else if (it.second == FEXCore::Config::ConfigOption::CONFIG_##enum) { \
|
||||
AppendStrArrayValue(it.second, *Value); \
|
||||
}
|
||||
|
||||
if (false) {
|
||||
}
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
else {
|
||||
Set(it.second, *Value);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -479,17 +490,17 @@ int main(int argc, char** argv, char** const envp) {
|
||||
|
||||
// Setup configurations that this tool needs
|
||||
// Maximum one instruction.
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_MAXINST, "1");
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_MAXINST, "1");
|
||||
// Enable block disassembly.
|
||||
FEXCore::Config::EraseSet(
|
||||
FEXCore::Config::Set(
|
||||
FEXCore::Config::CONFIG_DISASSEMBLE,
|
||||
fextl::fmt::format("{}", static_cast<uint64_t>(FEXCore::Config::Disassemble::BLOCKS | FEXCore::Config::Disassemble::STATS)));
|
||||
// Choose bitness.
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_IS64BIT_MODE, TestHeaderData->Bitness == 64 ? "1" : "0");
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_IS64BIT_MODE, TestHeaderData->Bitness == 64 ? "1" : "0");
|
||||
// Disable telemetry, it can affect instruction counts.
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_DISABLETELEMETRY, "1");
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_DISABLETELEMETRY, "1");
|
||||
// Disable vixl simulator indirect calls as it can affect instruction counts.
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_DISABLE_VIXL_INDIRECT_RUNTIME_CALLS, "1");
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_DISABLE_VIXL_INDIRECT_RUNTIME_CALLS, "1");
|
||||
|
||||
// Host feature override. Only supports overriding SVE width.
|
||||
enum HostFeatures {
|
||||
@@ -509,6 +520,7 @@ int main(int argc, char** argv, char** const envp) {
|
||||
FEATURE_TSO = (1U << 13),
|
||||
FEATURE_LRCPC = (1U << 14),
|
||||
FEATURE_LRCPC2 = (1U << 15),
|
||||
FEATURE_FRINTTS = (1U << 16),
|
||||
};
|
||||
|
||||
uint64_t SVEWidth = 0;
|
||||
@@ -556,13 +568,16 @@ int main(int argc, char** argv, char** const envp) {
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_LRCPC2) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLELRCPC2);
|
||||
}
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_FRINTTS) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLEFRINTTS);
|
||||
}
|
||||
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_TSO) {
|
||||
// Always disable auto migration.
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::ConfigOption::CONFIG_TSOAUTOMIGRATION, "0");
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::ConfigOption::CONFIG_TSOENABLED, "1");
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::ConfigOption::CONFIG_VECTORTSOENABLED, "1");
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::ConfigOption::CONFIG_MEMCPYSETTSOENABLED, "1");
|
||||
FEXCore::Config::Set(FEXCore::Config::ConfigOption::CONFIG_TSOAUTOMIGRATION, "0");
|
||||
FEXCore::Config::Set(FEXCore::Config::ConfigOption::CONFIG_TSOENABLED, "1");
|
||||
FEXCore::Config::Set(FEXCore::Config::ConfigOption::CONFIG_VECTORTSOENABLED, "1");
|
||||
FEXCore::Config::Set(FEXCore::Config::ConfigOption::CONFIG_MEMCPYSETTSOENABLED, "1");
|
||||
}
|
||||
|
||||
// Always enable ARMv8.1 LSE atomics.
|
||||
@@ -607,20 +622,23 @@ int main(int argc, char** argv, char** const envp) {
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_LRCPC2) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLELRCPC2);
|
||||
}
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_FRINTTS) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLEFRINTTS);
|
||||
}
|
||||
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_TSO) {
|
||||
// Always disable auto migration.
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::ConfigOption::CONFIG_TSOAUTOMIGRATION, "0");
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::ConfigOption::CONFIG_TSOENABLED, "0");
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::ConfigOption::CONFIG_VECTORTSOENABLED, "0");
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::ConfigOption::CONFIG_MEMCPYSETTSOENABLED, "0");
|
||||
FEXCore::Config::Set(FEXCore::Config::ConfigOption::CONFIG_TSOAUTOMIGRATION, "0");
|
||||
FEXCore::Config::Set(FEXCore::Config::ConfigOption::CONFIG_TSOENABLED, "0");
|
||||
FEXCore::Config::Set(FEXCore::Config::ConfigOption::CONFIG_VECTORTSOENABLED, "0");
|
||||
FEXCore::Config::Set(FEXCore::Config::ConfigOption::CONFIG_MEMCPYSETTSOENABLED, "0");
|
||||
}
|
||||
|
||||
// Always enable preserve_all abi.
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLEPRESERVEALLABI);
|
||||
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_HOSTFEATURES, fextl::fmt::format("{}", HostFeatureControl));
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_FORCESVEWIDTH, fextl::fmt::format("{}", SVEWidth));
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_HOSTFEATURES, fextl::fmt::format("{}", HostFeatureControl));
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_FORCESVEWIDTH, fextl::fmt::format("{}", SVEWidth));
|
||||
|
||||
// Initialize static tables.
|
||||
FEXCore::Context::InitializeStaticTables(TestHeaderData->Bitness == 64 ? FEXCore::Context::MODE_64BIT : FEXCore::Context::MODE_32BIT);
|
||||
|
||||
@@ -137,17 +137,6 @@ ELFContainer::ELFContainer(const fextl::string& Filename, const fextl::string& R
|
||||
|
||||
CalculateMemoryLayouts();
|
||||
CalculateSymbols();
|
||||
|
||||
// Print Information
|
||||
// PrintHeader();
|
||||
// PrintSectionHeaders();
|
||||
// PrintProgramHeaders();
|
||||
// PrintSymbolTable();
|
||||
// PrintRelocationTable();
|
||||
// PrintInitArray();
|
||||
// PrintDynamicTable();
|
||||
|
||||
// LOGMAN_THROW_A_FMT(InterpreterHeader == nullptr, "Can only handle static programs");
|
||||
}
|
||||
|
||||
ELFContainer::~ELFContainer() {
|
||||
@@ -674,251 +663,6 @@ void ELFContainer::AddUnwindEntries(UnwindAdder Adder) {
|
||||
}
|
||||
}
|
||||
|
||||
void ELFContainer::PrintHeader() const {
|
||||
if (Mode == MODE_32BIT) {
|
||||
LogMan::Msg::IFmt("Type: {}", Header._32.e_type);
|
||||
LogMan::Msg::IFmt("Machine: {}", Header._32.e_machine);
|
||||
LogMan::Msg::IFmt("Version: {}", Header._32.e_version);
|
||||
LogMan::Msg::IFmt("Entry point: 0x{:x}", Header._32.e_entry);
|
||||
LogMan::Msg::IFmt("PH Off: {}", Header._32.e_phoff);
|
||||
LogMan::Msg::IFmt("SH Off: {}", Header._32.e_shoff);
|
||||
LogMan::Msg::IFmt("Flags: {}", Header._32.e_flags);
|
||||
LogMan::Msg::IFmt("EH Size: {}", Header._32.e_ehsize);
|
||||
LogMan::Msg::IFmt("PH Num: {}", Header._32.e_phnum);
|
||||
LogMan::Msg::IFmt("SH Num: {}", Header._32.e_shnum);
|
||||
LogMan::Msg::IFmt("PH Entry Size: {}", Header._32.e_phentsize);
|
||||
LogMan::Msg::IFmt("SH Entry Size: {}", Header._32.e_shentsize);
|
||||
LogMan::Msg::IFmt("SH Str Index: {}", Header._32.e_shstrndx);
|
||||
} else {
|
||||
LogMan::Msg::IFmt("Type: {}", Header._64.e_type);
|
||||
LogMan::Msg::IFmt("Machine: {}", Header._64.e_machine);
|
||||
LogMan::Msg::IFmt("Version: {}", Header._64.e_version);
|
||||
LogMan::Msg::IFmt("Entry point: 0x{:x}", Header._64.e_entry);
|
||||
LogMan::Msg::IFmt("PH Off: {}", Header._64.e_phoff);
|
||||
LogMan::Msg::IFmt("SH Off: {}", Header._64.e_shoff);
|
||||
LogMan::Msg::IFmt("Flags: {}", Header._64.e_flags);
|
||||
LogMan::Msg::IFmt("EH Size: {}", Header._64.e_ehsize);
|
||||
LogMan::Msg::IFmt("PH Num: {}", Header._64.e_phnum);
|
||||
LogMan::Msg::IFmt("SH Num: {}", Header._64.e_shnum);
|
||||
LogMan::Msg::IFmt("PH Entry Size: {}", Header._64.e_phentsize);
|
||||
LogMan::Msg::IFmt("SH Entry Size: {}", Header._64.e_shentsize);
|
||||
LogMan::Msg::IFmt("SH Str Index: {}", Header._64.e_shstrndx);
|
||||
}
|
||||
}
|
||||
|
||||
void ELFContainer::PrintSectionHeaders() const {
|
||||
if (Mode == MODE_32BIT) {
|
||||
LOGMAN_THROW_A_FMT(Header._32.e_shstrndx < SectionHeaders.size(), "String index section is wrong index!");
|
||||
const Elf32_Shdr* StrHeader = SectionHeaders.at(Header._32.e_shstrndx)._32;
|
||||
const char* SHStrings = &RawFile.at(StrHeader->sh_offset);
|
||||
for (size_t i = 0; i < SectionHeaders.size(); ++i) {
|
||||
const Elf32_Shdr* hdr = SectionHeaders[i]._32;
|
||||
LogMan::Msg::IFmt("Index: {}", i);
|
||||
LogMan::Msg::IFmt("Name: {}", &SHStrings[hdr->sh_name]);
|
||||
LogMan::Msg::IFmt("Type: {}", hdr->sh_type);
|
||||
LogMan::Msg::IFmt("Flags: {}", hdr->sh_flags);
|
||||
LogMan::Msg::IFmt("Addr: 0x{:x}", hdr->sh_addr);
|
||||
LogMan::Msg::IFmt("Offset: 0x{:x}", hdr->sh_offset);
|
||||
LogMan::Msg::IFmt("Size: {}", hdr->sh_size);
|
||||
LogMan::Msg::IFmt("Link: {}", hdr->sh_link);
|
||||
LogMan::Msg::IFmt("Info: {}", hdr->sh_info);
|
||||
LogMan::Msg::IFmt("AddrAlign: {}", hdr->sh_addralign);
|
||||
LogMan::Msg::IFmt("Entry Size: {}", hdr->sh_entsize);
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(Header._64.e_shstrndx < SectionHeaders.size(), "String index section is wrong index!");
|
||||
const Elf64_Shdr* StrHeader = SectionHeaders.at(Header._64.e_shstrndx)._64;
|
||||
const char* SHStrings = &RawFile.at(StrHeader->sh_offset);
|
||||
for (size_t i = 0; i < SectionHeaders.size(); ++i) {
|
||||
const Elf64_Shdr* hdr = SectionHeaders[i]._64;
|
||||
LogMan::Msg::IFmt("Index: {}", i);
|
||||
LogMan::Msg::IFmt("Name: {}", &SHStrings[hdr->sh_name]);
|
||||
LogMan::Msg::IFmt("Type: {}", hdr->sh_type);
|
||||
LogMan::Msg::IFmt("Flags: {}", hdr->sh_flags);
|
||||
LogMan::Msg::IFmt("Addr: 0x{:x}", hdr->sh_addr);
|
||||
LogMan::Msg::IFmt("Offset: 0x{:x}", hdr->sh_offset);
|
||||
LogMan::Msg::IFmt("Size: {}", hdr->sh_size);
|
||||
LogMan::Msg::IFmt("Link: {}", hdr->sh_link);
|
||||
LogMan::Msg::IFmt("Info: {}", hdr->sh_info);
|
||||
LogMan::Msg::IFmt("AddrAlign: {}", hdr->sh_addralign);
|
||||
LogMan::Msg::IFmt("Entry Size: {}", hdr->sh_entsize);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void ELFContainer::PrintProgramHeaders() const {
|
||||
if (Mode == MODE_32BIT) {
|
||||
LOGMAN_THROW_A_FMT(Header._32.e_shstrndx < SectionHeaders.size(), "String index section is wrong index!");
|
||||
for (size_t i = 0; i < ProgramHeaders.size(); ++i) {
|
||||
const Elf32_Phdr* hdr = ProgramHeaders[i]._32;
|
||||
LogMan::Msg::IFmt("Type: {}", hdr->p_type);
|
||||
LogMan::Msg::IFmt("Flags: {}", hdr->p_flags);
|
||||
LogMan::Msg::IFmt("Offset: {}", hdr->p_offset);
|
||||
LogMan::Msg::IFmt("VAddr: 0x{:x}", hdr->p_vaddr);
|
||||
LogMan::Msg::IFmt("PAddr: 0x{:x}", hdr->p_paddr);
|
||||
LogMan::Msg::IFmt("FSize: {}", hdr->p_filesz);
|
||||
LogMan::Msg::IFmt("MemSize: {}", hdr->p_memsz);
|
||||
LogMan::Msg::IFmt("Align: {}", hdr->p_align);
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(Header._64.e_shstrndx < SectionHeaders.size(), "String index section is wrong index!");
|
||||
for (size_t i = 0; i < ProgramHeaders.size(); ++i) {
|
||||
const Elf64_Phdr* hdr = ProgramHeaders[i]._64;
|
||||
LogMan::Msg::IFmt("Type: {}", hdr->p_type);
|
||||
LogMan::Msg::IFmt("Flags: {}", hdr->p_flags);
|
||||
LogMan::Msg::IFmt("Offset: {}", hdr->p_offset);
|
||||
LogMan::Msg::IFmt("VAddr: 0x{:x}", hdr->p_vaddr);
|
||||
LogMan::Msg::IFmt("PAddr: 0x{:x}", hdr->p_paddr);
|
||||
LogMan::Msg::IFmt("FSize: {}", hdr->p_filesz);
|
||||
LogMan::Msg::IFmt("MemSize: {}", hdr->p_memsz);
|
||||
LogMan::Msg::IFmt("Align: {}", hdr->p_align);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void ELFContainer::PrintSymbolTable() const {
|
||||
if (Mode == MODE_32BIT) {
|
||||
// Find the symbol table
|
||||
const Elf32_Shdr* SymTabHeader {nullptr};
|
||||
const Elf32_Shdr* StringTableHeader {nullptr};
|
||||
const char* StrTab {nullptr};
|
||||
for (uint32_t i = 0; i < SectionHeaders.size(); ++i) {
|
||||
const Elf32_Shdr* hdr = SectionHeaders.at(i)._32;
|
||||
if (hdr->sh_type == SHT_SYMTAB) {
|
||||
SymTabHeader = hdr;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!SymTabHeader) {
|
||||
LogMan::Msg::IFmt("No Symbol table");
|
||||
return;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(SymTabHeader->sh_link < SectionHeaders.size(), "Symbol table string table section is wrong");
|
||||
LOGMAN_THROW_A_FMT(SymTabHeader->sh_entsize == sizeof(Elf32_Sym), "Entry size doesn't match symbol entry");
|
||||
|
||||
StringTableHeader = SectionHeaders.at(SymTabHeader->sh_link)._32;
|
||||
StrTab = &RawFile.at(StringTableHeader->sh_offset);
|
||||
|
||||
const uint64_t NumSymbols = SymTabHeader->sh_size / SymTabHeader->sh_entsize;
|
||||
for (uint64_t i = 0; i < NumSymbols; ++i) {
|
||||
const uint64_t offset = SymTabHeader->sh_offset + i * SymTabHeader->sh_entsize;
|
||||
const auto* Symbol = reinterpret_cast<const Elf32_Sym*>(&RawFile.at(offset));
|
||||
|
||||
LogMan::Msg::IFmt("{} : {:x} {} {} {} {} {}", i, Symbol->st_value, Symbol->st_size, uint32_t(Symbol->st_info),
|
||||
uint32_t(Symbol->st_other), Symbol->st_shndx, &StrTab[Symbol->st_name]);
|
||||
}
|
||||
} else {
|
||||
// Find the symbol table
|
||||
const Elf64_Shdr* SymTabHeader {nullptr};
|
||||
const Elf64_Shdr* StringTableHeader {nullptr};
|
||||
const char* StrTab {nullptr};
|
||||
for (uint32_t i = 0; i < SectionHeaders.size(); ++i) {
|
||||
const Elf64_Shdr* hdr = SectionHeaders.at(i)._64;
|
||||
if (hdr->sh_type == SHT_SYMTAB) {
|
||||
SymTabHeader = hdr;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!SymTabHeader) {
|
||||
LogMan::Msg::IFmt("No Symbol table");
|
||||
return;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(SymTabHeader->sh_link < SectionHeaders.size(), "Symbol table string table section is wrong");
|
||||
LOGMAN_THROW_A_FMT(SymTabHeader->sh_entsize == sizeof(Elf64_Sym), "Entry size doesn't match symbol entry");
|
||||
|
||||
StringTableHeader = SectionHeaders.at(SymTabHeader->sh_link)._64;
|
||||
StrTab = &RawFile.at(StringTableHeader->sh_offset);
|
||||
|
||||
const uint64_t NumSymbols = SymTabHeader->sh_size / SymTabHeader->sh_entsize;
|
||||
for (uint64_t i = 0; i < NumSymbols; ++i) {
|
||||
const uint64_t offset = SymTabHeader->sh_offset + i * SymTabHeader->sh_entsize;
|
||||
const auto* Symbol = reinterpret_cast<const Elf64_Sym*>(&RawFile.at(offset));
|
||||
|
||||
LogMan::Msg::IFmt("{} : {:x} {} {} {} {} {}", i, Symbol->st_value, Symbol->st_size, uint32_t(Symbol->st_info),
|
||||
uint32_t(Symbol->st_other), Symbol->st_shndx, &StrTab[Symbol->st_name]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void ELFContainer::PrintRelocationTable() const {
|
||||
if (Mode == MODE_32BIT) {
|
||||
} else {
|
||||
const Elf64_Shdr* RelaHeader {nullptr};
|
||||
const Elf64_Shdr* DynSymHeader {nullptr};
|
||||
|
||||
const Elf64_Shdr* StrHeader = SectionHeaders.at(Header._64.e_shstrndx)._64;
|
||||
const char* SHStrings = &RawFile.at(StrHeader->sh_offset);
|
||||
|
||||
const Elf64_Shdr* StringTableHeader {nullptr};
|
||||
const char* StrTab {nullptr};
|
||||
|
||||
for (uint32_t i = 0; i < SectionHeaders.size(); ++i) {
|
||||
const Elf64_Shdr* hdr = SectionHeaders.at(i)._64;
|
||||
if (hdr->sh_type == SHT_REL) {
|
||||
LogMan::Msg::DFmt("Unhandled REL section");
|
||||
} else if (hdr->sh_type == SHT_RELA) {
|
||||
RelaHeader = hdr;
|
||||
LogMan::Msg::DFmt("Relocation Section: '{}'", &SHStrings[RelaHeader->sh_name]);
|
||||
|
||||
if (RelaHeader->sh_info != 0) {
|
||||
LOGMAN_THROW_A_FMT(RelaHeader->sh_info < SectionHeaders.size(), "Rela header pointers to invalid GOT header");
|
||||
}
|
||||
|
||||
if (RelaHeader->sh_link != 0) {
|
||||
LOGMAN_THROW_A_FMT(RelaHeader->sh_link < SectionHeaders.size(), "Rela header pointers to invalid dyndym header");
|
||||
DynSymHeader = SectionHeaders.at(RelaHeader->sh_link)._64;
|
||||
|
||||
StringTableHeader = SectionHeaders.at(DynSymHeader->sh_link)._64;
|
||||
StrTab = &RawFile.at(StringTableHeader->sh_offset);
|
||||
}
|
||||
|
||||
const size_t EntryCount = RelaHeader->sh_size / RelaHeader->sh_entsize;
|
||||
const auto* Entries = reinterpret_cast<const Elf64_Rela*>(&RawFile.at(RelaHeader->sh_offset));
|
||||
|
||||
for (size_t j = 0; j < EntryCount; ++j) {
|
||||
const auto* Entry = &Entries[j];
|
||||
const uint32_t Sym = Entry->r_info >> 32;
|
||||
const uint32_t Type = Entry->r_info & ~0U;
|
||||
LogMan::Msg::DFmt("RELA Entry {}", j);
|
||||
LogMan::Msg::DFmt("\toffset: 0x{:x}", Entry->r_offset);
|
||||
LogMan::Msg::DFmt("\tSym: 0x{:x}", Sym);
|
||||
if (DynSymHeader && Sym != 0) {
|
||||
LOGMAN_THROW_A_FMT(DynSymHeader->sh_entsize == sizeof(Elf64_Sym), "Oops, entry size doesn't match");
|
||||
|
||||
const uint64_t offset = DynSymHeader->sh_offset + Sym * DynSymHeader->sh_entsize;
|
||||
const auto* Symbol = reinterpret_cast<const Elf64_Sym*>(&RawFile.at(offset));
|
||||
LogMan::Msg::DFmt("\tSym Name: '{}'", &StrTab[Symbol->st_name]);
|
||||
}
|
||||
|
||||
LogMan::Msg::DFmt("\tType: 0x{:x}", Type);
|
||||
LogMan::Msg::DFmt("\tadded: 0x{:x}", Entry->r_addend);
|
||||
if (Type == R_X86_64_IRELATIVE) { // 37/0x25
|
||||
LogMan::Msg::DFmt("\tR_x86_64_IRELATIVE");
|
||||
} else if (Type == R_X86_64_64) {
|
||||
LogMan::Msg::DFmt("\tR_X86_64_64");
|
||||
} else if (Type == R_X86_64_RELATIVE) {
|
||||
LogMan::Msg::DFmt("\tR_X86_64_RELATIVE");
|
||||
} else if (Type == R_X86_64_GLOB_DAT) {
|
||||
LogMan::Msg::DFmt("\tR_X86_64_GLOB_DAT");
|
||||
} else if (Type == R_X86_64_JUMP_SLOT) {
|
||||
LogMan::Msg::DFmt("\tR_X86_64_JUMP_SLOT");
|
||||
} else if (Type == R_X86_64_DTPMOD64) {
|
||||
LogMan::Msg::DFmt("\tR_X86_64_DTPMOD64");
|
||||
} else if (Type == R_X86_64_DTPOFF64) {
|
||||
LogMan::Msg::DFmt("\tR_X86_64_DTPOFF64");
|
||||
} else if (Type == R_X86_64_TPOFF64) {
|
||||
LogMan::Msg::DFmt("\tR_X86_64_TPOFF64");
|
||||
} else {
|
||||
LogMan::Msg::DFmt("Unknown relocation type: {}(0x{:x})", Type, Type);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void ELFContainer::FixupRelocations(void* ELFBase, uint64_t GuestELFBase, SymbolGetter Getter) {
|
||||
if (Mode == MODE_32BIT) {
|
||||
} else {
|
||||
@@ -1065,150 +809,6 @@ void ELFContainer::FixupRelocations(void* ELFBase, uint64_t GuestELFBase, Symbol
|
||||
}
|
||||
}
|
||||
|
||||
void ELFContainer::PrintInitArray() const {
|
||||
if (Mode == MODE_32BIT) {
|
||||
for (size_t i = 0; i < SectionHeaders.size(); ++i) {
|
||||
const auto* hdr = SectionHeaders[i]._32;
|
||||
if (hdr->sh_type == SHT_INIT_ARRAY) {
|
||||
const size_t Entries = hdr->sh_size / hdr->sh_entsize;
|
||||
for (size_t j = 0; j < Entries; ++j) {
|
||||
LogMan::Msg::DFmt("init_array[{}]", j);
|
||||
LogMan::Msg::DFmt("\t{}", *reinterpret_cast<const uint64_t*>(&RawFile.at(hdr->sh_offset + j * hdr->sh_entsize)));
|
||||
}
|
||||
}
|
||||
}
|
||||
} else {
|
||||
for (size_t i = 0; i < SectionHeaders.size(); ++i) {
|
||||
const auto* hdr = SectionHeaders[i]._64;
|
||||
if (hdr->sh_type == SHT_INIT_ARRAY) {
|
||||
const size_t Entries = hdr->sh_size / hdr->sh_entsize;
|
||||
for (size_t j = 0; j < Entries; ++j) {
|
||||
LogMan::Msg::DFmt("init_array[{}]", j);
|
||||
LogMan::Msg::DFmt("\t{}", *reinterpret_cast<const uint64_t*>(&RawFile.at(hdr->sh_offset + j * hdr->sh_entsize)));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void ELFContainer::PrintDynamicTable() const {
|
||||
if (Mode == MODE_32BIT) {
|
||||
for (size_t i = 0; i < SectionHeaders.size(); ++i) {
|
||||
const auto* hdr = SectionHeaders[i]._32;
|
||||
if (hdr->sh_type == SHT_DYNAMIC) {
|
||||
const auto* StrHeader = SectionHeaders.at(hdr->sh_link)._32;
|
||||
const char* SHStrings = &RawFile.at(StrHeader->sh_offset);
|
||||
|
||||
const size_t Entries = hdr->sh_size / hdr->sh_entsize;
|
||||
for (size_t j = 0; i < Entries; ++j) {
|
||||
const auto* Dynamic = reinterpret_cast<const Elf32_Dyn*>(&RawFile.at(hdr->sh_offset + j * hdr->sh_entsize));
|
||||
#define PRINT(x, y, z) x(Dynamic->d_tag == DT_##y) LogMan::Msg::DFmt("Dyn {}: (" #y ") 0x{:x}", j, Dynamic->d_un.z);
|
||||
if (Dynamic->d_tag == DT_NULL) {
|
||||
break;
|
||||
} else if (Dynamic->d_tag == DT_NEEDED) {
|
||||
LogMan::Msg::DFmt("Dyn {}: (NEEDED) '{}'", j, &SHStrings[Dynamic->d_un.d_val]);
|
||||
} else if (Dynamic->d_tag == DT_SONAME) {
|
||||
LogMan::Msg::DFmt("Dyn {}: (SONAME) '{}'", j, &SHStrings[Dynamic->d_un.d_val]);
|
||||
}
|
||||
PRINT(else if, HASH, d_val)
|
||||
PRINT(else if, INIT, d_val)
|
||||
PRINT(else if, FINI, d_val)
|
||||
PRINT(else if, INIT_ARRAY, d_val)
|
||||
PRINT(else if, INIT_ARRAYSZ, d_val)
|
||||
PRINT(else if, FINI_ARRAY, d_val)
|
||||
PRINT(else if, FINI_ARRAYSZ, d_val)
|
||||
PRINT(else if, GNU_HASH, d_val)
|
||||
PRINT(else if, STRTAB, d_val)
|
||||
PRINT(else if, SYMTAB, d_val)
|
||||
PRINT(else if, STRSZ, d_val)
|
||||
PRINT(else if, SYMENT, d_val)
|
||||
PRINT(else if, DEBUG, d_val)
|
||||
PRINT(else if, PLTGOT, d_val)
|
||||
PRINT(else if, PLTRELSZ, d_val)
|
||||
PRINT(else if, PLTREL, d_val)
|
||||
PRINT(else if, JMPREL, d_val)
|
||||
PRINT(else if, RELA, d_val)
|
||||
PRINT(else if, RELASZ, d_val)
|
||||
PRINT(else if, RELAENT, d_val)
|
||||
PRINT(else if, VERNEED, d_val)
|
||||
PRINT(else if, VERNEEDNUM, d_val)
|
||||
PRINT(else if, VERSYM, d_val)
|
||||
else if (Dynamic->d_tag >= DT_LOOS && Dynamic->d_tag <= DT_HIOS) {
|
||||
LogMan::Msg::DFmt("Dyn {}: (OSSpecific) 0x{:x}", j, Dynamic->d_tag);
|
||||
}
|
||||
else if (Dynamic->d_tag >= DT_LOPROC && Dynamic->d_tag <= DT_HIPROC) {
|
||||
LogMan::Msg::DFmt("Dyn {}: (Proc-Specific) 0x{:x}", j, Dynamic->d_tag);
|
||||
}
|
||||
PRINT(else if, RELACOUNT, d_val)
|
||||
PRINT(else if, RELCOUNT, d_val)
|
||||
PRINT(else if, VERDEF, d_val)
|
||||
PRINT(else if, VERDEFNUM, d_val)
|
||||
PRINT(else if, FLAGS, d_val)
|
||||
else LogMan::Msg::DFmt("Unknown dynamic section: {}(0x{:x})", Dynamic->d_tag, Dynamic->d_tag);
|
||||
#undef PRINT
|
||||
}
|
||||
}
|
||||
}
|
||||
} else {
|
||||
for (size_t i = 0; i < SectionHeaders.size(); ++i) {
|
||||
const auto* hdr = SectionHeaders[i]._64;
|
||||
if (hdr->sh_type == SHT_DYNAMIC) {
|
||||
const auto* StrHeader = SectionHeaders.at(hdr->sh_link)._64;
|
||||
const char* SHStrings = &RawFile.at(StrHeader->sh_offset);
|
||||
|
||||
const size_t Entries = hdr->sh_size / hdr->sh_entsize;
|
||||
for (size_t j = 0; i < Entries; ++j) {
|
||||
const auto* Dynamic = reinterpret_cast<const Elf64_Dyn*>(&RawFile.at(hdr->sh_offset + j * hdr->sh_entsize));
|
||||
#define PRINT(x, y, z) x(Dynamic->d_tag == DT_##y) LogMan::Msg::DFmt("Dyn {}: (" #y ") 0x{:x}", j, Dynamic->d_un.z);
|
||||
if (Dynamic->d_tag == DT_NULL) {
|
||||
break;
|
||||
} else if (Dynamic->d_tag == DT_NEEDED) {
|
||||
LogMan::Msg::DFmt("Dyn {}: (NEEDED) '{}'", j, &SHStrings[Dynamic->d_un.d_val]);
|
||||
} else if (Dynamic->d_tag == DT_SONAME) {
|
||||
LogMan::Msg::DFmt("Dyn {}: (SONAME) '{}'", j, &SHStrings[Dynamic->d_un.d_val]);
|
||||
}
|
||||
PRINT(else if, HASH, d_val)
|
||||
PRINT(else if, INIT, d_val)
|
||||
PRINT(else if, FINI, d_val)
|
||||
PRINT(else if, INIT_ARRAY, d_val)
|
||||
PRINT(else if, INIT_ARRAYSZ, d_val)
|
||||
PRINT(else if, FINI_ARRAY, d_val)
|
||||
PRINT(else if, FINI_ARRAYSZ, d_val)
|
||||
PRINT(else if, GNU_HASH, d_val)
|
||||
PRINT(else if, STRTAB, d_val)
|
||||
PRINT(else if, SYMTAB, d_val)
|
||||
PRINT(else if, STRSZ, d_val)
|
||||
PRINT(else if, SYMENT, d_val)
|
||||
PRINT(else if, DEBUG, d_val)
|
||||
PRINT(else if, PLTGOT, d_val)
|
||||
PRINT(else if, PLTRELSZ, d_val)
|
||||
PRINT(else if, PLTREL, d_val)
|
||||
PRINT(else if, JMPREL, d_val)
|
||||
PRINT(else if, RELA, d_val)
|
||||
PRINT(else if, RELASZ, d_val)
|
||||
PRINT(else if, RELAENT, d_val)
|
||||
PRINT(else if, VERNEED, d_val)
|
||||
PRINT(else if, VERNEEDNUM, d_val)
|
||||
PRINT(else if, VERSYM, d_val)
|
||||
else if (Dynamic->d_tag >= DT_LOOS && Dynamic->d_tag <= DT_HIOS) {
|
||||
LogMan::Msg::DFmt("Dyn {}: (OSSpecific) 0x{:x}", j, Dynamic->d_tag);
|
||||
}
|
||||
else if (Dynamic->d_tag >= DT_LOPROC && Dynamic->d_tag <= DT_HIPROC) {
|
||||
LogMan::Msg::DFmt("Dyn {}: (Proc-Specific) 0x{:x}", j, Dynamic->d_tag);
|
||||
}
|
||||
PRINT(else if, RELACOUNT, d_val)
|
||||
PRINT(else if, RELCOUNT, d_val)
|
||||
PRINT(else if, VERDEF, d_val)
|
||||
PRINT(else if, VERDEFNUM, d_val)
|
||||
PRINT(else if, FLAGS, d_val)
|
||||
else LogMan::Msg::DFmt("Unknown dynamic section: {}(0x{:x})", Dynamic->d_tag, Dynamic->d_tag);
|
||||
#undef PRINT
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void ELFContainer::GetInitLocations(uint64_t GuestELFBase, fextl::vector<uint64_t>* Locations) {
|
||||
if (Mode == MODE_32BIT) {
|
||||
// If INIT exists then add that first
|
||||
|
||||
@@ -88,8 +88,6 @@ public:
|
||||
return &NecessaryLibs;
|
||||
}
|
||||
|
||||
void PrintRelocationTable() const;
|
||||
|
||||
using SymbolGetter = std::function<ELFSymbol*(const char*, uint8_t)>;
|
||||
void FixupRelocations(void* ELFBase, uint64_t GuestELFBase, SymbolGetter Getter);
|
||||
|
||||
@@ -146,14 +144,6 @@ private:
|
||||
void CalculateSymbols();
|
||||
void GetDynamicLibs();
|
||||
|
||||
// Information functions
|
||||
void PrintHeader() const;
|
||||
void PrintSectionHeaders() const;
|
||||
void PrintProgramHeaders() const;
|
||||
void PrintSymbolTable() const;
|
||||
void PrintInitArray() const;
|
||||
void PrintDynamicTable() const;
|
||||
|
||||
fextl::vector<char> RawFile;
|
||||
union {
|
||||
Elf32_Ehdr _32;
|
||||
|
||||
@@ -45,7 +45,9 @@ struct ELFParser {
|
||||
}
|
||||
|
||||
// Reset to beginning
|
||||
lseek(fd, 0, SEEK_SET);
|
||||
if (lseek(fd, 0, SEEK_SET) == -1) {
|
||||
return false;
|
||||
}
|
||||
|
||||
uint8_t header[5];
|
||||
if (pread(fd, header, sizeof(header), 0) == -1) {
|
||||
@@ -146,6 +148,12 @@ struct ELFParser {
|
||||
return false;
|
||||
}
|
||||
|
||||
// sanity check program header offset size.
|
||||
if (ehdr.e_phoff > Size || (ehdr.e_phentsize * ehdr.e_phnum) > (Size - ehdr.e_phoff)) {
|
||||
LogMan::Msg::EFmt("Program headers exceeds size of program");
|
||||
return false;
|
||||
}
|
||||
|
||||
if (type == ::ELFLoader::ELFContainer::TYPE_X86_32) {
|
||||
fextl::vector<Elf32_Phdr> phdrs32(ehdr.e_phnum);
|
||||
|
||||
|
||||
Loaded 100 of 289 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user