mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-07 09:00:17 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
a8c1a36c12 | ||
|
|
acfb24f871 | ||
|
|
ecff5aea71 | ||
|
|
74cb225ccb | ||
|
|
d5db2ccf18 | ||
|
|
cebcf50c65 | ||
|
|
3a3c9101c7 | ||
|
|
ec1c7797f4 | ||
|
|
313528c34b | ||
|
|
907fd6b04b | ||
|
|
aa0fc9071e | ||
|
|
0bd924eb7e | ||
|
|
852109c142 | ||
|
|
9b0bb29d78 | ||
|
|
c3e71de1d7 | ||
|
|
0256d6820c | ||
|
|
1b18bfaff5 | ||
|
|
6065e7a62b | ||
|
|
8aecdc536c | ||
|
|
86a2e9e655 | ||
|
|
8f50106187 | ||
|
|
02d3a319f9 | ||
|
|
d18d0435ae | ||
|
|
cdaf1c5262 | ||
|
|
4b0e3bff54 | ||
|
|
c4f7b27459 | ||
|
|
3eb8be9953 | ||
|
|
1f08f8df0d | ||
|
|
0038a0b19c | ||
|
|
166a7c7e53 | ||
|
|
8ac296bd6f | ||
|
|
e092a38e0f | ||
|
|
b145e894e4 | ||
|
|
63a8b66b28 | ||
|
|
a16cc87852 | ||
|
|
1050b60057 | ||
|
|
9188e85164 | ||
|
|
f9b369c550 | ||
|
|
949b205f42 | ||
|
|
31ee8d8178 | ||
|
|
3155590e87 | ||
|
|
feab0bce4b | ||
|
|
0599d80b13 | ||
|
|
d58e12c5e4 | ||
|
|
68939c5a5c | ||
|
|
25c4fb8508 | ||
|
|
88e6c48db7 | ||
|
|
218b0d491a | ||
|
|
bfba74dab9 | ||
|
|
5708846a07 | ||
|
|
fd28783f85 | ||
|
|
0ca34d11ad | ||
|
|
bf3275ba4a | ||
|
|
d8e4e00b2b | ||
|
|
8b38a6dd08 | ||
|
|
63f72621fc | ||
|
|
580a8c9c61 | ||
|
|
249351cef4 | ||
|
|
46690ae352 | ||
|
|
68b5a90518 | ||
|
|
cb54823622 | ||
|
|
3a2ca41724 | ||
|
|
dd4e6d29ca | ||
|
|
497ee32f59 | ||
|
|
d34c287f69 | ||
|
|
b744ca16e1 | ||
|
|
573c262bb2 | ||
|
|
ff2c2f1e1f | ||
|
|
8058391955 | ||
|
|
54dbb9f248 | ||
|
|
7d1351f402 | ||
|
|
21233430aa | ||
|
|
bd1d6820b7 | ||
|
|
18e84ae2e8 | ||
|
|
f9c29056e6 | ||
|
|
40b1c32008 | ||
|
|
b5ed804578 | ||
|
|
fbe3a86c4e | ||
|
|
498c86b47d | ||
|
|
b8b6f81c44 | ||
|
|
aab9e1b751 | ||
|
|
ec976f3f75 | ||
|
|
f169fa5da2 | ||
|
|
35ee12e7e9 | ||
|
|
4497ab8844 | ||
|
|
9f399f3313 | ||
|
|
3939213336 | ||
|
|
cc27a0f666 | ||
|
|
1ff2216063 | ||
|
|
b8165813b4 | ||
|
|
2215b153db | ||
|
|
cb91d585c3 | ||
|
|
c971d4044f | ||
|
|
fdb4a078f2 | ||
|
|
42ea711850 | ||
|
|
96721978c0 | ||
|
|
d64698e4c9 | ||
|
|
4565f2b689 | ||
|
|
5fa7f1d50d | ||
|
|
2cfc42bd6a | ||
|
|
003de2b659 | ||
|
|
4ab6c2252d | ||
|
|
8f9d818368 | ||
|
|
4ebc307744 | ||
|
|
0b519b29d9 | ||
|
|
8881e8d96e | ||
|
|
9a99608f68 | ||
|
|
26c26308db | ||
|
|
123c5d809e | ||
|
|
7c42c7798c | ||
|
|
0f98daf1d9 | ||
|
|
c3de7c63b4 | ||
|
|
ab9123a427 | ||
|
|
e504a8c979 | ||
|
|
f6dd87a3a1 | ||
|
|
89c530054d | ||
|
|
f87edbe1cb | ||
|
|
4aa477de3e | ||
|
|
694e674fe6 | ||
|
|
7ebc0f32b8 | ||
|
|
68cacc2fc3 | ||
|
|
43eb597044 | ||
|
|
7efd827e78 | ||
|
|
2f6f8b93e9 | ||
|
|
d48413cbec | ||
|
|
6bafca688b | ||
|
|
53356f1aa7 | ||
|
|
51281f6a3a | ||
|
|
7a5e08c5ab | ||
|
|
642903a7bf | ||
|
|
f51fd6c78d | ||
|
|
03cf15a9e1 | ||
|
|
fe19c04f6a | ||
|
|
b86cbda03d | ||
|
|
b1af6e23cf | ||
|
|
e26b9b12fa | ||
|
|
9bf47b3f23 | ||
|
|
2887416b2e | ||
|
|
9a92f6f743 | ||
|
|
4929480719 | ||
|
|
c8437d2303 | ||
|
|
ef25ae4d3b | ||
|
|
9399790c11 | ||
|
|
ab7f6484cf | ||
|
|
2d8c5da379 | ||
|
|
b18f148575 | ||
|
|
6426428718 | ||
|
|
af2ee426f5 | ||
|
|
6c4e9ff42d | ||
|
|
62a37a7d70 | ||
|
|
ef83addc74 | ||
|
|
fb308f5947 | ||
|
|
cedb93c11c | ||
|
|
61d2a09827 | ||
|
|
9c6423f37a | ||
|
|
e18a661b50 | ||
|
|
16e5777816 | ||
|
|
64020e8828 | ||
|
|
30798556fc | ||
|
|
cd3518d99d | ||
|
|
bc87c3d494 | ||
|
|
726228c418 | ||
|
|
929b111648 | ||
|
|
1700a73382 | ||
|
|
0f9d791911 | ||
|
|
61acc76be7 | ||
|
|
0f8ba5bf32 | ||
|
|
2cb8a96f0e | ||
|
|
1102122639 | ||
|
|
66e026a9e8 | ||
|
|
b901b42417 | ||
|
|
9f2f10a65f | ||
|
|
b0b41d00ee | ||
|
|
2d56f5eba0 | ||
|
|
8ffc6fbc6b | ||
|
|
0ffd94dadb | ||
|
|
585320093a | ||
|
|
bdd351a42c | ||
|
|
feaee702e9 | ||
|
|
d89fc84fcf | ||
|
|
688cd1a4bb | ||
|
|
c056875a00 | ||
|
|
054118139f | ||
|
|
447148d95a | ||
|
|
f6b1e42a0b | ||
|
|
90e74e5572 | ||
|
|
e4fc5fe5be | ||
|
|
809b2c6115 | ||
|
|
659538ef4b | ||
|
|
4b677f4b42 | ||
|
|
87699ea5a0 | ||
|
|
a2ae113ee9 | ||
|
|
901e2c75d4 | ||
|
|
871d140b7c | ||
|
|
2f6ae1ad02 | ||
|
|
806e98925c | ||
|
|
c7dbd2fac2 | ||
|
|
1b7729efed | ||
|
|
6d1d5aeffb | ||
|
|
9ae04b5771 | ||
|
|
fc61e3b1b5 | ||
|
|
fa1d9910e4 | ||
|
|
d425873eed | ||
|
|
d89c54dd95 | ||
|
|
5e023c55bd | ||
|
|
2487172df3 | ||
|
|
bdd078df17 | ||
|
|
db75335ad0 | ||
|
|
531ab5b5b1 | ||
|
|
0601a863f9 | ||
|
|
10d34c9564 | ||
|
|
a37d6a3841 | ||
|
|
d5234a43da | ||
|
|
b7733540c1 | ||
|
|
2ae02ded74 | ||
|
|
2e57ad644d | ||
|
|
031afbfb18 | ||
|
|
377ce2e2f6 | ||
|
|
a2fd3c077d | ||
|
|
0f5aff73ab | ||
|
|
f0b208e692 | ||
|
|
7fcb5fc590 | ||
|
|
b76f819759 | ||
|
|
4b36d4f1ea | ||
|
|
f4d0c6c807 | ||
|
|
79ab76b42e | ||
|
|
8c3ac61f8d | ||
|
|
a8bc20f76b | ||
|
|
53a55baf23 | ||
|
|
63d6800cf3 | ||
|
|
44492b4828 | ||
|
|
8ed9f8aef5 | ||
|
|
711021e24e | ||
|
|
55dea335fe | ||
|
|
7097532ddf | ||
|
|
66841ce35c | ||
|
|
3d814cb7c1 | ||
|
|
07afdca58b | ||
|
|
93ed346fd2 | ||
|
|
c9eee9bf7f | ||
|
|
d1a4029bc5 | ||
|
|
97070aad25 | ||
|
|
39640185a3 | ||
|
|
4b17506ffe | ||
|
|
2435ebecbe | ||
|
|
410a35b968 | ||
|
|
c8d234a767 | ||
|
|
2bf87ff40f | ||
|
|
ba347c49c9 | ||
|
|
81434cd233 | ||
|
|
46d0df9cba | ||
|
|
b7f58e68c5 | ||
|
|
6ba2accbdc | ||
|
|
29ee94d331 | ||
|
|
cdae654fe4 | ||
|
|
a506c84bc7 | ||
|
|
8e4a47181b | ||
|
|
b5241e0f60 | ||
|
|
57ed466a7f | ||
|
|
4f46f55f2d | ||
|
|
69cfc78ee1 | ||
|
|
04f1ab8571 | ||
|
|
95694b2017 | ||
|
|
00bed2f0c0 | ||
|
|
e718fc35f8 | ||
|
|
717015bae8 | ||
|
|
530d3d809b | ||
|
|
596b32d15c | ||
|
|
8d3918b4f0 | ||
|
|
4c7e31513b | ||
|
|
765509d7f5 | ||
|
|
dbb58d10a6 | ||
|
|
d448976b3c | ||
|
|
de6931b1f5 | ||
|
|
35268d185e | ||
|
|
beef9eee0a | ||
|
|
34a274d4e6 | ||
|
|
ef28a6c19a | ||
|
|
50b5971ee5 | ||
|
|
9a70ae18ea | ||
|
|
d10853b775 | ||
|
|
cdf6a16efc | ||
|
|
02d7261f51 | ||
|
|
9c9ddeffbe | ||
|
|
afa8b3a5c9 | ||
|
|
bbcd4c168c | ||
|
|
d22bd9cac7 | ||
|
|
5ccf25196e | ||
|
|
caf15a2dac | ||
|
|
982a05450c | ||
|
|
3e381b742c | ||
|
|
6b82664166 | ||
|
|
bb30a2eb1e | ||
|
|
b3fdf5c48f | ||
|
|
17d5ed847f | ||
|
|
4335d17fc0 | ||
|
|
ae07958577 | ||
|
|
c51b9ba3d6 | ||
|
|
cc6ff5e9e6 | ||
|
|
b968ea7e7e | ||
|
|
d14b6e160e | ||
|
|
b09b9488ef | ||
|
|
a8120ee7ef | ||
|
|
fb82059750 | ||
|
|
0fc6240d72 | ||
|
|
a7c6fdb1fc | ||
|
|
b76a2963cf | ||
|
|
42b0fbd34c | ||
|
|
f69ef8606f | ||
|
|
afa5ad5f9f | ||
|
|
1fc82708e9 | ||
|
|
917cbbadde | ||
|
|
c37dc81839 | ||
|
|
df718d55ef | ||
|
|
54412f1d5e | ||
|
|
da76023bea | ||
|
|
41e9309a36 | ||
|
|
116268b275 | ||
|
|
73802492b8 | ||
|
|
7cd4d53fa9 | ||
|
|
c44757975e | ||
|
|
f25cdcdf63 | ||
|
|
5d37253e85 | ||
|
|
cd5f42ec79 | ||
|
|
6651f9e94b | ||
|
|
e3ee579f92 | ||
|
|
73e7240574 | ||
|
|
391f9aa97d | ||
|
|
8ad54e7bd5 | ||
|
|
9eccc01dd3 | ||
|
|
39c1f816fc | ||
|
|
a32b892787 | ||
|
|
1b144ba3f0 | ||
|
|
6a39a8db72 | ||
|
|
602c530615 | ||
|
|
c8c27f26f7 | ||
|
|
549cdc4c2c | ||
|
|
3160e0a430 | ||
|
|
dcebe85f3a | ||
|
|
2ba0b66426 | ||
|
|
5c9543f159 | ||
|
|
f6e3689f30 | ||
|
|
a761343717 | ||
|
|
b4c47a3d24 | ||
|
|
906988c49b | ||
|
|
4186b2ad82 | ||
|
|
2943cff73f | ||
|
|
53ac5579fb | ||
|
|
0d7a9f911a | ||
|
|
dce9de222d | ||
|
|
d46722a95c | ||
|
|
3ba4da7736 | ||
|
|
6abf5b90b7 | ||
|
|
00aa4ddea0 | ||
|
|
d0c6f9de22 | ||
|
|
a85cc85081 | ||
|
|
75793300f2 | ||
|
|
672805584e | ||
|
|
5b4fd590d1 | ||
|
|
0ccd38f593 | ||
|
|
b46e5d4488 | ||
|
|
d7223d598f | ||
|
|
7a0368132d | ||
|
|
aff3914a66 | ||
|
|
8876047875 | ||
|
|
78e2aa16f0 | ||
|
|
3ef695cf70 | ||
|
|
c4d8dd6413 | ||
|
|
a49d30f6e2 | ||
|
|
d80daf2692 | ||
|
|
0e5c9e8b06 | ||
|
|
6bc4aed82c | ||
|
|
e69e1200f5 | ||
|
|
81e253b06b | ||
|
|
43dcc84c07 | ||
|
|
6e9d5f00de | ||
|
|
a65ca9663f | ||
|
|
02b767c0ea | ||
|
|
4b1c1d266d | ||
|
|
d9bf140971 | ||
|
|
1402776ba6 | ||
|
|
e36fb47d98 | ||
|
|
2bb37357c0 | ||
|
|
7494ac7615 | ||
|
|
44bc3fb90b | ||
|
|
423e29ba42 |
No files matched your search
@@ -28,7 +28,7 @@ jobs:
|
||||
|
||||
- name: Get changed files
|
||||
id: changed-files
|
||||
uses: tj-actions/changed-files@v39
|
||||
uses: step-security/changed-files@3dbe17c78367e7d60f00d78ae6781a35be47b4a1 # v45.0.1
|
||||
with:
|
||||
separator: ","
|
||||
skip_initial_fetch: true
|
||||
|
||||
+3
-4
@@ -5,10 +5,6 @@
|
||||
[submodule "External/cpp-optparse"]
|
||||
path = Source/Common/cpp-optparse
|
||||
url = https://github.com/Sonicadvance1/cpp-optparse
|
||||
[submodule "External/xbyak"]
|
||||
shallow = true
|
||||
path = External/xbyak
|
||||
url = https://github.com/herumi/xbyak.git
|
||||
[submodule "External/fex-posixtest-bins"]
|
||||
shallow = true
|
||||
path = External/fex-posixtest-bins
|
||||
@@ -47,3 +43,6 @@
|
||||
[submodule "External/jemalloc_glibc"]
|
||||
path = External/jemalloc_glibc
|
||||
url = https://github.com/FEX-Emu/jemalloc.git
|
||||
[submodule "External/tracy"]
|
||||
path = External/tracy
|
||||
url = https://github.com/wolfpld/tracy
|
||||
+21
-5
@@ -30,7 +30,7 @@ option(ENABLE_VIXL_SIMULATOR "Enable use of VIXL simulator for emulation (only u
|
||||
option(ENABLE_VIXL_DISASSEMBLER "Enables debug disassembler output with VIXL" FALSE)
|
||||
option(USE_LEGACY_BINFMTMISC "Uses legacy method of setting up binfmt_misc" FALSE)
|
||||
option(ENABLE_FEXCORE_PROFILER "Enables use of the FEXCore timeline profiling capabilities" FALSE)
|
||||
set (FEXCORE_PROFILER_BACKEND "gpuvis" CACHE STRING "Set which backend you want to use for the FEXCore profiler")
|
||||
set (FEXCORE_PROFILER_BACKEND "gpuvis" CACHE STRING "Set which backend to use for the FEXCore profiler (gpuvis, tracy)")
|
||||
option(ENABLE_GLIBC_ALLOCATOR_HOOK_FAULT "Enables glibc memory allocation hooking with fault for CI testing")
|
||||
option(USE_PDB_DEBUGINFO "Builds debug info in PDB format" FALSE)
|
||||
|
||||
@@ -61,6 +61,22 @@ if (ENABLE_FEXCORE_PROFILER)
|
||||
|
||||
if (FEXCORE_PROFILER_BACKEND STREQUAL "GPUVIS")
|
||||
add_definitions(-DFEXCORE_PROFILER_BACKEND=1)
|
||||
elseif (FEXCORE_PROFILER_BACKEND STREQUAL "TRACY")
|
||||
add_definitions(-DFEXCORE_PROFILER_BACKEND=2)
|
||||
add_definitions(-DTRACY_ENABLE=1)
|
||||
# Required so that Tracy will only start in the selected guest application
|
||||
add_definitions(-DTRACY_MANUAL_LIFETIME=1)
|
||||
add_definitions(-DTRACY_DELAYED_INIT=1)
|
||||
# This interferes with FEX's signal handling
|
||||
add_definitions(-DTRACY_NO_CRASH_HANDLER=1)
|
||||
# Tracy can gather call stack samples in regular intervals, but this
|
||||
# isn't useful for us since it would usually sample opaque JIT code
|
||||
add_definitions(-DTRACY_NO_SAMPLING=1)
|
||||
# This pulls in libbacktrace which allocators in global constructors (before FEX can set up its allocator hooks)
|
||||
add_definitions(-DTRACY_NO_CALLSTACK=1)
|
||||
if (MINGW_BUILD)
|
||||
message(FATAL_ERROR "Tracy profiler not supported")
|
||||
endif()
|
||||
else()
|
||||
message(FATAL_ERROR "Unknown FEXCore profiler backend ${FEXCORE_PROFILER_BACKEND}")
|
||||
endif()
|
||||
@@ -270,6 +286,10 @@ if (BUILD_TESTS OR ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
|
||||
include_directories(SYSTEM External/vixl/src/)
|
||||
endif()
|
||||
|
||||
if (ENABLE_FEXCORE_PROFILER AND FEXCORE_PROFILER_BACKEND STREQUAL "TRACY")
|
||||
add_subdirectory(External/tracy)
|
||||
endif()
|
||||
|
||||
if (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
|
||||
# This means we were attempted to get compiled with GCC
|
||||
message(FATAL_ERROR "FEX doesn't support getting compiled with GCC!")
|
||||
@@ -364,10 +384,6 @@ if (TUNE_CPU STREQUAL "native")
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-mcpu=native")
|
||||
endif()
|
||||
else()
|
||||
# Due to an oversight in llvm, it declares any reasonably new Kryo CPU to only be ARMv8.0
|
||||
# Manually detect newer CPU revisions until clang and llvm fixes their bug
|
||||
# This script will either provide a supported CPU or 'native'
|
||||
# Additionally -march doesn't work under AArch64+Clang, so you have to use -mcpu or -mtune
|
||||
execute_process(COMMAND python3 "${PROJECT_SOURCE_DIR}/Scripts/aarch64_fit_native.py" "/proc/cpuinfo" "${CMAKE_CXX_COMPILER_VERSION}"
|
||||
OUTPUT_VARIABLE AARCH64_CPU)
|
||||
|
||||
|
||||
@@ -64,7 +64,7 @@ public:
|
||||
}
|
||||
void sha256h2(ARMEmitter::VRegister rd, ARMEmitter::VRegister rn, ARMEmitter::VRegister rm) {
|
||||
constexpr uint32_t Op = 0b0101'1110'0000'0000'0000'00 << 10;
|
||||
Crypto3RegSHA(Op, 0b100, rd, rn, rm);
|
||||
Crypto3RegSHA(Op, 0b101, rd, rn, rm);
|
||||
}
|
||||
void sha256su1(ARMEmitter::VRegister rd, ARMEmitter::VRegister rn, ARMEmitter::VRegister rm) {
|
||||
constexpr uint32_t Op = 0b0101'1110'0000'0000'0000'00 << 10;
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <cstring>
|
||||
#include <type_traits>
|
||||
|
||||
namespace ARMEmitter {
|
||||
class Buffer {
|
||||
@@ -21,29 +22,25 @@ public:
|
||||
Size = BaseSize;
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
requires (std::is_trivially_copyable_v<T>)
|
||||
void dcn(const T& Data) {
|
||||
std::memcpy(CurrentOffset, &Data, sizeof(Data));
|
||||
CurrentOffset += sizeof(Data);
|
||||
}
|
||||
void dc8(uint8_t Data) {
|
||||
decltype(Data)* Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
|
||||
*Memory = Data;
|
||||
CurrentOffset += sizeof(Data);
|
||||
dcn(Data);
|
||||
}
|
||||
|
||||
void dc16(uint16_t Data) {
|
||||
decltype(Data)* Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
|
||||
*Memory = Data;
|
||||
CurrentOffset += sizeof(Data);
|
||||
dcn(Data);
|
||||
}
|
||||
|
||||
void dc32(uint32_t Data) {
|
||||
decltype(Data)* Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
|
||||
*Memory = Data;
|
||||
CurrentOffset += sizeof(Data);
|
||||
dcn(Data);
|
||||
}
|
||||
void dc64(uint64_t Data) {
|
||||
dcn(Data);
|
||||
}
|
||||
|
||||
void dc64(uint64_t Data) {
|
||||
decltype(Data)* Memory = reinterpret_cast<decltype(Data)*>(CurrentOffset);
|
||||
*Memory = Data;
|
||||
CurrentOffset += sizeof(Data);
|
||||
}
|
||||
void EmitString(const char* String) {
|
||||
const auto StringLength = strlen(String);
|
||||
memcpy(CurrentOffset, String, StringLength);
|
||||
@@ -95,10 +92,6 @@ public:
|
||||
|
||||
protected:
|
||||
|
||||
void ResetBuffer() {
|
||||
CurrentOffset = BufferBase;
|
||||
}
|
||||
|
||||
uint8_t* BufferBase;
|
||||
uint8_t* CurrentOffset;
|
||||
uint64_t Size;
|
||||
|
||||
@@ -767,6 +767,7 @@ public:
|
||||
void st1(ARMEmitter::SubRegSize size, T rt, uint32_t Index, ARMEmitter::Register rn, uint32_t PostOffset) {
|
||||
LOGMAN_THROW_A_FMT(size == SubRegSize::i8Bit || size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit,
|
||||
"Incorrect size");
|
||||
LOGMAN_THROW_A_FMT((PostOffset * 8) == SubRegSizeInBits(size), "Post-Index size must match element size");
|
||||
constexpr uint32_t Op = 0b0000'1101'1 << 23;
|
||||
uint32_t Q;
|
||||
uint32_t R = 0;
|
||||
@@ -808,6 +809,7 @@ public:
|
||||
void ld1(ARMEmitter::SubRegSize size, T rt, uint32_t Index, ARMEmitter::Register rn, uint32_t PostOffset) {
|
||||
LOGMAN_THROW_A_FMT(size == SubRegSize::i8Bit || size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit,
|
||||
"Incorrect size");
|
||||
LOGMAN_THROW_A_FMT((PostOffset * 8) == SubRegSizeInBits(size), "Post-Index size must match element size");
|
||||
constexpr uint32_t Op = 0b0000'1101'1 << 23;
|
||||
uint32_t Q;
|
||||
uint32_t R = 0;
|
||||
@@ -899,6 +901,7 @@ public:
|
||||
void st2(SubRegSize size, T rt, T rt2, uint32_t Index, Register rn, uint32_t PostOffset) {
|
||||
LOGMAN_THROW_A_FMT(size == SubRegSize::i8Bit || size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit,
|
||||
"Incorrect size");
|
||||
LOGMAN_THROW_A_FMT((PostOffset * 8) == (SubRegSizeInBits(size) * 2), "Post-Index size must match element size");
|
||||
LOGMAN_THROW_A_FMT(AreVectorsSequential(rt, rt2), "rt and rt2 must be sequential");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1101'1 << 23;
|
||||
@@ -942,6 +945,7 @@ public:
|
||||
void ld2(SubRegSize size, T rt, T rt2, uint32_t Index, Register rn, uint32_t PostOffset) {
|
||||
LOGMAN_THROW_A_FMT(size == SubRegSize::i8Bit || size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit,
|
||||
"Incorrect size");
|
||||
LOGMAN_THROW_A_FMT((PostOffset * 8) == (SubRegSizeInBits(size) * 2), "Post-Index size must match element size");
|
||||
LOGMAN_THROW_A_FMT(AreVectorsSequential(rt, rt2), "rt and rt2 must be sequential");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1101'1 << 23;
|
||||
@@ -985,6 +989,7 @@ public:
|
||||
void st3(SubRegSize size, T rt, T rt2, T rt3, uint32_t Index, Register rn, uint32_t PostOffset) {
|
||||
LOGMAN_THROW_A_FMT(size == SubRegSize::i8Bit || size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit,
|
||||
"Incorrect size");
|
||||
LOGMAN_THROW_A_FMT((PostOffset * 8) == (SubRegSizeInBits(size) * 3), "Post-Index size must match element size");
|
||||
LOGMAN_THROW_A_FMT(AreVectorsSequential(rt, rt2, rt3), "rt, rt2, and rt3 must be sequential");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1101'1 << 23;
|
||||
@@ -1028,6 +1033,7 @@ public:
|
||||
void ld3(SubRegSize size, T rt, T rt2, T rt3, uint32_t Index, Register rn, uint32_t PostOffset) {
|
||||
LOGMAN_THROW_A_FMT(size == SubRegSize::i8Bit || size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit,
|
||||
"Incorrect size");
|
||||
LOGMAN_THROW_A_FMT((PostOffset * 8) == (SubRegSizeInBits(size) * 3), "Post-Index size must match element size");
|
||||
LOGMAN_THROW_A_FMT(AreVectorsSequential(rt, rt2, rt3), "rt, rt2, and rt3 must be sequential");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1101'1 << 23;
|
||||
@@ -1071,6 +1077,7 @@ public:
|
||||
void st4(SubRegSize size, T rt, T rt2, T rt3, T rt4, uint32_t Index, Register rn, uint32_t PostOffset) {
|
||||
LOGMAN_THROW_A_FMT(size == SubRegSize::i8Bit || size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit,
|
||||
"Incorrect size");
|
||||
LOGMAN_THROW_A_FMT((PostOffset * 8) == (SubRegSizeInBits(size) * 4), "Post-Index size must match element size");
|
||||
LOGMAN_THROW_A_FMT(AreVectorsSequential(rt, rt2, rt3, rt4), "rt, rt2, rt3, and rt4 must be sequential");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1101'1 << 23;
|
||||
@@ -1114,6 +1121,7 @@ public:
|
||||
void ld4(SubRegSize size, T rt, T rt2, T rt3, T rt4, uint32_t Index, Register rn, uint32_t PostOffset) {
|
||||
LOGMAN_THROW_A_FMT(size == SubRegSize::i8Bit || size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit,
|
||||
"Incorrect size");
|
||||
LOGMAN_THROW_A_FMT((PostOffset * 8) == (SubRegSizeInBits(size) * 4), "Post-Index size must match element size");
|
||||
LOGMAN_THROW_A_FMT(AreVectorsSequential(rt, rt2, rt3, rt4), "rt, rt2, rt3, and rt4 must be sequential");
|
||||
|
||||
constexpr uint32_t Op = 0b0000'1101'1 << 23;
|
||||
|
||||
@@ -30,9 +30,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(Register) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<Register>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<Register>, "Needs to be standard");
|
||||
static_assert(sizeof(Register) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<Register>);
|
||||
static_assert(std::is_standard_layout_v<Register>);
|
||||
|
||||
/* 32-bit GPR register class.
|
||||
* This class will imply a 32-bit register size being used.
|
||||
@@ -58,9 +58,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(Register) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<Register>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<Register>, "Needs to be standard");
|
||||
static_assert(sizeof(WRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<WRegister>);
|
||||
static_assert(std::is_standard_layout_v<WRegister>);
|
||||
|
||||
/* 64-bit GPR register class.
|
||||
* This class will imply a 64-bit register size being used.
|
||||
@@ -86,9 +86,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(Register) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<Register>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<Register>, "Needs to be standard");
|
||||
static_assert(sizeof(XRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<XRegister>);
|
||||
static_assert(std::is_standard_layout_v<XRegister>);
|
||||
|
||||
inline constexpr WRegister Register::W() const {
|
||||
return WRegister {Index};
|
||||
@@ -283,9 +283,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(VRegister) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<VRegister>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<VRegister>, "Needs to be standard");
|
||||
static_assert(sizeof(VRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<VRegister>);
|
||||
static_assert(std::is_standard_layout_v<VRegister>);
|
||||
|
||||
/* 8-bit ASIMD register class
|
||||
* This class implies 8-bit scalar register.
|
||||
@@ -315,9 +315,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(BRegister) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<BRegister>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<BRegister>, "Needs to be standard");
|
||||
static_assert(sizeof(BRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<BRegister>);
|
||||
static_assert(std::is_standard_layout_v<BRegister>);
|
||||
|
||||
/* 16-bit ASIMD register class
|
||||
* This class implies 16-bit scalar register.
|
||||
@@ -347,9 +347,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(HRegister) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<HRegister>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<HRegister>, "Needs to be standard");
|
||||
static_assert(sizeof(HRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<HRegister>);
|
||||
static_assert(std::is_standard_layout_v<HRegister>);
|
||||
|
||||
/* 32-bit ASIMD register class
|
||||
* This class implies 32-bit scalar register.
|
||||
@@ -379,9 +379,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(SRegister) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<SRegister>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<SRegister>, "Needs to be standard");
|
||||
static_assert(sizeof(SRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<SRegister>);
|
||||
static_assert(std::is_standard_layout_v<SRegister>);
|
||||
|
||||
/* 64-bit ASIMD register class
|
||||
* This class doesn't imply Vector or Scalar.
|
||||
@@ -412,9 +412,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(DRegister) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<DRegister>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<DRegister>, "Needs to be standard");
|
||||
static_assert(sizeof(DRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<DRegister>);
|
||||
static_assert(std::is_standard_layout_v<DRegister>);
|
||||
|
||||
/* 128-bit ASIMD register class
|
||||
* This class doesn't imply Vector or Scalar.
|
||||
@@ -445,9 +445,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(QRegister) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<QRegister>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<QRegister>, "Needs to be standard");
|
||||
static_assert(sizeof(QRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<QRegister>);
|
||||
static_assert(std::is_standard_layout_v<QRegister>);
|
||||
|
||||
/* Unsized SVE register class.
|
||||
* This class explicitly implies the instruction will operate using SVE.
|
||||
@@ -474,9 +474,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(ZRegister) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<ZRegister>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<ZRegister>, "Needs to be standard");
|
||||
static_assert(sizeof(ZRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<ZRegister>);
|
||||
static_assert(std::is_standard_layout_v<ZRegister>);
|
||||
|
||||
// VRegister
|
||||
inline constexpr BRegister VRegister::B() const {
|
||||
@@ -919,9 +919,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(PRegister) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<PRegister>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<PRegister>, "Needs to be standard");
|
||||
static_assert(sizeof(PRegister) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<PRegister>);
|
||||
static_assert(std::is_standard_layout_v<PRegister>);
|
||||
|
||||
// Unsized predicate register for SVE with zeroing semantics.
|
||||
class PRegisterZero {
|
||||
@@ -947,9 +947,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(PRegisterZero) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<PRegisterZero>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<PRegisterZero>, "Needs to be standard");
|
||||
static_assert(sizeof(PRegisterZero) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<PRegisterZero>);
|
||||
static_assert(std::is_standard_layout_v<PRegisterZero>);
|
||||
|
||||
// Unsized predicate register for SVE with merging semantics.
|
||||
class PRegisterMerge {
|
||||
@@ -975,9 +975,9 @@ public:
|
||||
private:
|
||||
uint32_t Index;
|
||||
};
|
||||
static_assert(sizeof(PRegisterZero) == sizeof(uint32_t), "Needs to be uint32_t");
|
||||
static_assert(std::is_trivial_v<PRegisterZero>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<PRegisterZero>, "Needs to be standard");
|
||||
static_assert(sizeof(PRegisterMerge) == sizeof(uint32_t));
|
||||
static_assert(std::is_trivially_copyable_v<PRegisterMerge>);
|
||||
static_assert(std::is_standard_layout_v<PRegisterMerge>);
|
||||
|
||||
// PRegister
|
||||
inline constexpr PRegisterZero PRegister::Zeroing() const {
|
||||
|
||||
@@ -2134,11 +2134,31 @@ public:
|
||||
|
||||
// SVE2 Crypto Extensions
|
||||
// SVE2 crypto unary operations
|
||||
// XXX:
|
||||
void aesimc(ZRegister zdn, ZRegister zn) {
|
||||
SVE2CryptoUnaryOperation(1, zdn, zn);
|
||||
}
|
||||
void aesmc(ZRegister zdn, ZRegister zn) {
|
||||
SVE2CryptoUnaryOperation(0, zdn, zn);
|
||||
}
|
||||
|
||||
// SVE2 crypto destructive binary operations
|
||||
// XXX:
|
||||
void aese(ZRegister zdn, ZRegister zn, ZRegister zm) {
|
||||
SVE2CryptoDestructiveBinaryOperation(0, 0, zdn, zn, zm);
|
||||
}
|
||||
void aesd(ZRegister zdn, ZRegister zn, ZRegister zm) {
|
||||
SVE2CryptoDestructiveBinaryOperation(0, 1, zdn, zn, zm);
|
||||
}
|
||||
void sm4e(ZRegister zdn, ZRegister zn, ZRegister zm) {
|
||||
SVE2CryptoDestructiveBinaryOperation(1, 0, zdn, zn, zm);
|
||||
}
|
||||
|
||||
// SVE2 crypto constructive binary operations
|
||||
// XXX:
|
||||
void sm4ekey(ZRegister zd, ZRegister zn, ZRegister zm) {
|
||||
SVE2CryptoConstructiveBinaryOperation(0, zd, zn, zm);
|
||||
}
|
||||
void rax1(ZRegister zd, ZRegister zn, ZRegister zm) {
|
||||
SVE2CryptoConstructiveBinaryOperation(1, zd, zn, zm);
|
||||
}
|
||||
|
||||
// SVE Floating Point Widening Multiply-Add - Indexed
|
||||
// SVE BFloat16 floating-point dot product (indexed)
|
||||
@@ -2379,13 +2399,13 @@ public:
|
||||
} else if (srcsize == SubRegSize::i32Bit) {
|
||||
// Srcsize = fp32, opc1 encodes dst size
|
||||
LOGMAN_THROW_A_FMT(dstsize != SubRegSize::i16Bit, "Unsupported size in {}", __func__);
|
||||
opc1 = dstsize == SubRegSize::i64Bit ? 0b11 : 0b10;
|
||||
opc2 = 0b10;
|
||||
opc1 = dstsize == SubRegSize::i64Bit ? 0b11 : dstsize == SubRegSize::i32Bit ? 0b10 : 0b00;
|
||||
} else if (srcsize == SubRegSize::i64Bit) {
|
||||
LOGMAN_THROW_A_FMT(dstsize != SubRegSize::i16Bit, "Unsupported size in {}", __func__);
|
||||
// SrcSize = fp64, opc2 encodes dst size
|
||||
opc1 = 0b11;
|
||||
opc2 = dstsize == SubRegSize::i64Bit ? 0b11 : dstsize == SubRegSize::i32Bit ? 0b00 : 0b00;
|
||||
opc2 = dstsize == SubRegSize::i64Bit ? 0b11 : 0b00;
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
@@ -2400,13 +2420,13 @@ public:
|
||||
} else if (srcsize == SubRegSize::i32Bit) {
|
||||
// Srcsize = fp32, opc1 encodes dst size
|
||||
LOGMAN_THROW_A_FMT(dstsize != SubRegSize::i16Bit, "Unsupported size in {}", __func__);
|
||||
opc1 = dstsize == SubRegSize::i64Bit ? 0b11 : 0b10;
|
||||
opc2 = 0b10;
|
||||
opc1 = dstsize == SubRegSize::i64Bit ? 0b11 : dstsize == SubRegSize::i32Bit ? 0b10 : 0b00;
|
||||
} else if (srcsize == SubRegSize::i64Bit) {
|
||||
LOGMAN_THROW_A_FMT(dstsize != SubRegSize::i16Bit, "Unsupported size in {}", __func__);
|
||||
// SrcSize = fp64, opc2 encodes dst size
|
||||
opc1 = 0b11;
|
||||
opc2 = dstsize == SubRegSize::i64Bit ? 0b11 : dstsize == SubRegSize::i32Bit ? 0b00 : 0b00;
|
||||
opc2 = dstsize == SubRegSize::i64Bit ? 0b11 : 0b00;
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
@@ -3892,6 +3912,35 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
void SVE2CryptoUnaryOperation(uint32_t op, ZRegister zdn, ZRegister zn) {
|
||||
LOGMAN_THROW_A_FMT(zdn == zn, "zdn and zn must be the same register");
|
||||
|
||||
uint32_t Instr = 0b0100'0101'0010'0000'1110'0000'0000'0000;
|
||||
Instr |= op << 10;
|
||||
Instr |= zdn.Idx();
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
void SVE2CryptoDestructiveBinaryOperation(uint32_t op, uint32_t o2, ZRegister zdn, ZRegister zn, ZRegister zm) {
|
||||
LOGMAN_THROW_A_FMT(zdn == zn, "zdn and zn must be the same register");
|
||||
|
||||
uint32_t Instr = 0b0100'0101'0010'0010'1110'0000'0000'0000;
|
||||
Instr |= op << 16;
|
||||
Instr |= o2 << 10;
|
||||
Instr |= zm.Idx() << 5;
|
||||
Instr |= zdn.Idx();
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
void SVE2CryptoConstructiveBinaryOperation(uint32_t op, ZRegister zd, ZRegister zn, ZRegister zm) {
|
||||
uint32_t Instr = 0b0100'0101'0010'0000'1111'0000'0000'0000;
|
||||
Instr |= zm.Idx() << 16;
|
||||
Instr |= op << 10;
|
||||
Instr |= zn.Idx() << 5;
|
||||
Instr |= zd.Idx();
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
void SVE2BitwisePermute(SubRegSize size, uint32_t opc, ZRegister zd, ZRegister zn, ZRegister zm) {
|
||||
LOGMAN_THROW_A_FMT(size != SubRegSize::i128Bit, "Can't use 128-bit element size");
|
||||
|
||||
@@ -5029,7 +5078,7 @@ private:
|
||||
void SVE2IntegerMultiplyLong(uint32_t SUT, SubRegSize size, ZRegister zd, ZRegister zn, ZRegister zm) {
|
||||
// PMULLB and PMULLT support the use of 128-bit element sizes (with the SVE2PMULL128 extension)
|
||||
if (SUT == 0b010 || SUT == 0b011) {
|
||||
LOGMAN_THROW_A_FMT(size != SubRegSize::i8Bit, "Can't use 8-bit element size");
|
||||
LOGMAN_THROW_A_FMT(size != SubRegSize::i8Bit && size != SubRegSize::i32Bit, "Can't use 8-bit or 32-bit element size");
|
||||
|
||||
// 128-bit variant is encoded as if it were 8-bit (0b00)
|
||||
if (size == SubRegSize::i128Bit) {
|
||||
@@ -5051,7 +5100,7 @@ private:
|
||||
const uint32_t element_size = SubRegSizeInBits(size);
|
||||
|
||||
if (is_left_shift) {
|
||||
LOGMAN_THROW_A_FMT(shift >= 0 && shift < element_size, "Invalid left shift value ({}). Must be within [0, {}]", shift, element_size - 1);
|
||||
LOGMAN_THROW_A_FMT(shift < element_size, "Invalid left shift value ({}). Must be within [0, {}]", shift, element_size - 1);
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(shift > 0 && shift <= element_size, "Invalid right shift value ({}). Must be within [1, {}]", shift, element_size);
|
||||
}
|
||||
|
||||
+1
-136
@@ -9,25 +9,6 @@
|
||||
"@PREFIX_LIB@/libGL.so.1.7.0"
|
||||
]
|
||||
},
|
||||
"GLESv2": {
|
||||
"Library": "libGLESv2-guest.so",
|
||||
"Depends": [
|
||||
"X11"
|
||||
],
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libGLESv2.so",
|
||||
"@PREFIX_LIB@/libGLESv2.so.2",
|
||||
"@PREFIX_LIB@/libGLESv2.so.2.0.0"
|
||||
]
|
||||
},
|
||||
"X11": {
|
||||
"Library": "libX11-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libX11.so",
|
||||
"@PREFIX_LIB@/libX11.so.6",
|
||||
"@PREFIX_LIB@/libX11.so.6.4.0"
|
||||
]
|
||||
},
|
||||
"Vulkan": {
|
||||
"Library": "libvulkan-guest.so",
|
||||
"Overlay": [
|
||||
@@ -36,89 +17,6 @@
|
||||
"@HOME@/.local/share/Steam/ubuntu12_32/steam-runtime/pinned_libs_64/libvulkan.so.1"
|
||||
]
|
||||
},
|
||||
"xcb": {
|
||||
"Depends": [
|
||||
"X11"
|
||||
],
|
||||
"Library": "libxcb-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxcb.so",
|
||||
"@PREFIX_LIB@/libxcb.so.1",
|
||||
"@PREFIX_LIB@/libxcb.so.1.1.0"
|
||||
]
|
||||
},
|
||||
"xcb-dri2": {
|
||||
"Library": "libxcb-dri2-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxcb-dri2.so",
|
||||
"@PREFIX_LIB@/libxcb-dri2.so.0",
|
||||
"@PREFIX_LIB@/libxcb-dri2.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-dri3": {
|
||||
"Library": "libxcb-dri3-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxcb-dri3.so",
|
||||
"@PREFIX_LIB@/libxcb-dri3.so.0",
|
||||
"@PREFIX_LIB@/libxcb-dri3.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-xfixes": {
|
||||
"Library": "libxcb-xfixes-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxcb-xfixes.so",
|
||||
"@PREFIX_LIB@/libxcb-xfixes.so.0",
|
||||
"@PREFIX_LIB@/libxcb-xfixes.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-shm": {
|
||||
"Library": "libxcb-shm-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxcb-shm.so",
|
||||
"@PREFIX_LIB@/libxcb-shm.so.0",
|
||||
"@PREFIX_LIB@/libxcb-shm.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-sync": {
|
||||
"Library": "libxcb-sync-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxcb-sync.so",
|
||||
"@PREFIX_LIB@/libxcb-sync.so.1",
|
||||
"@PREFIX_LIB@/libxcb-sync.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-randr": {
|
||||
"Library": "libxcb-randr-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxcb-randr.so",
|
||||
"@PREFIX_LIB@/libxcb-randr.so.0",
|
||||
"@PREFIX_LIB@/libxcb-randr.so.0.1.0"
|
||||
]
|
||||
},
|
||||
"xcb-present": {
|
||||
"Library": "libxcb-present-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxcb-present.so",
|
||||
"@PREFIX_LIB@/libxcb-present.so.0",
|
||||
"@PREFIX_LIB@/libxcb-present.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-glx": {
|
||||
"Library": "libxcb-glx-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxcb-glx.so",
|
||||
"@PREFIX_LIB@/libxcb-glx.so.0",
|
||||
"@PREFIX_LIB@/libxcb-glx.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xshmfence": {
|
||||
"Library": "libxshmfence-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libxshmfence.so",
|
||||
"@PREFIX_LIB@/libxshmfence.so.1",
|
||||
"@PREFIX_LIB@/libxshmfence.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"drm": {
|
||||
"Library": "libdrm-guest.so",
|
||||
"Overlay": [
|
||||
@@ -141,38 +39,6 @@
|
||||
"@PREFIX_LIB@/libfex_thunk_test.so"
|
||||
]
|
||||
},
|
||||
"Xrender": {
|
||||
"Library": "libXrender-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libXrender.so",
|
||||
"@PREFIX_LIB@/libXrender.so.1",
|
||||
"@PREFIX_LIB@/libXrender.so.1.3.0"
|
||||
]
|
||||
},
|
||||
"Xext": {
|
||||
"Library": "libXext-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libXext.so",
|
||||
"@PREFIX_LIB@/libXext.so.6",
|
||||
"@PREFIX_LIB@/libXext.so.6.4.0"
|
||||
]
|
||||
},
|
||||
"Xfixes": {
|
||||
"Library": "libXfixes-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libXfixes.so",
|
||||
"@PREFIX_LIB@/libXfixes.so.3",
|
||||
"@PREFIX_LIB@/libXfixes.so.3.1.0"
|
||||
]
|
||||
},
|
||||
"OpenCL": {
|
||||
"Library" : "libOpenCL-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libOpenCL.so",
|
||||
"@PREFIX_LIB@/libOpenCL.so.1",
|
||||
"@PREFIX_LIB@/libOpenCL.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"WaylandClient": {
|
||||
"Library" : "libwayland-client-guest.so",
|
||||
"Overlay": [
|
||||
@@ -180,7 +46,6 @@
|
||||
"@PREFIX_LIB@/libwayland-client.so.0",
|
||||
"@PREFIX_LIB@/libwayland-client.so.0.20.0"
|
||||
]
|
||||
},
|
||||
"":{}
|
||||
}
|
||||
}
|
||||
}
|
||||
Vendored
+1
-1
Submodule External/Vulkan-Headers updated: 29f979ee5a...cacef3039d.
Vendored
+1
-1
Submodule External/fmt updated: 873670ba3f...123913715a.
+1
Submodule External/tracy added at 5d542dc09f.
Vendored
+1
-1
Submodule External/vixl updated: 3180ab603b...84bc10c107.
Vendored
-1
Submodule External/xbyak deleted from c68cc53d18.
@@ -188,27 +188,33 @@ def print_man_environment_tail():
|
||||
|
||||
# Additional environment variables that live outside of the normal loop
|
||||
print_man_env_option(
|
||||
"FEX_APP_CONFIG_LOCATION",
|
||||
"APP_CONFIG_LOCATION",
|
||||
[
|
||||
"Allows the user to override where FEX looks for configuration files",
|
||||
"By default FEX will look in {$HOME, $XDG_CONFIG_HOME}/.fex-emu/",
|
||||
"This will override the full path",
|
||||
"If FEX_PORTABLE is declared then relative paths are also supported",
|
||||
"For FEXInterpreter: Relative to the FEXInterpreter binary",
|
||||
"For WINE: Relative to %LOCALAPPDATA%"
|
||||
],
|
||||
"''", True)
|
||||
|
||||
print_man_env_option(
|
||||
"FEX_APP_CONFIG",
|
||||
"APP_CONFIG",
|
||||
[
|
||||
"Allows the user to override where FEX looks for only the application config file",
|
||||
"By default FEX will look in {$HOME, $XDG_CONFIG_HOME}/.fex-emu/Config.json",
|
||||
"This will override this file location",
|
||||
"One must be careful with this option as it will override any applications that load with execve as well"
|
||||
"If you need to support applications that execve then use FEX_APP_CONFIG_LOCATION instead"
|
||||
"If FEX_PORTABLE is declared then relative paths are also supported",
|
||||
"For FEXInterpreter: Relative to the FEXInterpreter binary",
|
||||
"For WINE: Relative to %LOCALAPPDATA%"
|
||||
],
|
||||
"''", True)
|
||||
|
||||
print_man_env_option(
|
||||
"FEX_APP_DATA_LOCATION",
|
||||
"APP_DATA_LOCATION",
|
||||
[
|
||||
"Allows the user to override where FEX looks for data files",
|
||||
"By default FEX will look in {$HOME, $XDG_DATA_HOME}/.fex-emu/",
|
||||
@@ -218,7 +224,7 @@ def print_man_environment_tail():
|
||||
"''", True)
|
||||
|
||||
print_man_env_option(
|
||||
"FEX_PORTABLE",
|
||||
"PORTABLE",
|
||||
[
|
||||
"Allows FEX to run without installation. Global locations for configuration and binfmt_misc are ignored.",
|
||||
"For FEXInterpreter on Linux:",
|
||||
@@ -417,7 +423,7 @@ def print_parse_argloader_options(options):
|
||||
# these need a bit more help
|
||||
output_argloader.write("\tauto Array = Options.all(\"{0}\");\n".format(op_key))
|
||||
output_argloader.write("\tfor (auto iter = Array.begin(); iter != Array.end(); ++iter) {\n")
|
||||
output_argloader.write("\t\tSet(FEXCore::Config::ConfigOption::CONFIG_{0}, *iter);\n".format(op_key.upper()))
|
||||
output_argloader.write("\t\tAppendStrArrayValue(FEXCore::Config::ConfigOption::CONFIG_{0}, *iter);\n".format(op_key.upper()))
|
||||
output_argloader.write("\t}\n")
|
||||
else:
|
||||
if (NeedsString):
|
||||
@@ -455,14 +461,19 @@ def print_parse_jsonloader_options(options):
|
||||
output_argloader.write("#ifdef JSONLOADER\n")
|
||||
output_argloader.write("#undef JSONLOADER\n")
|
||||
output_argloader.write("if (false) {}\n")
|
||||
op_key = None
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
value_type = op_vals["Type"]
|
||||
if (value_type == "strenum"):
|
||||
output_argloader.write("else if (KeyName == \"{0}\") {{\n".format(op_key))
|
||||
output_argloader.write("Set(KeyOption, FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, Value_View));\n".format(op_key, op_key, op_key))
|
||||
output_argloader.write("\tSet(KeyOption, FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, Value_View));\n".format(op_key, op_key, op_key))
|
||||
output_argloader.write("}\n")
|
||||
|
||||
elif (value_type == "strarray"):
|
||||
output_argloader.write("else if (KeyName == \"{0}\") {{\n".format(op_key))
|
||||
output_argloader.write("\tAppendStrArrayValue(KeyOption, ConfigString);\n")
|
||||
output_argloader.write("}\n")
|
||||
assert op_key is not None, "No options found in JSONLOADER"
|
||||
output_argloader.write("else {{\n".format(op_key))
|
||||
output_argloader.write("Set(KeyOption, ConfigString);\n")
|
||||
output_argloader.write("}\n")
|
||||
|
||||
@@ -374,7 +374,7 @@ def print_ir_structs(defines):
|
||||
output_file.write("};\n")
|
||||
|
||||
# Add a static assert that the IR ops must be pod
|
||||
output_file.write("static_assert(std::is_trivial_v<IROp_{}>);\n".format(op.Name))
|
||||
output_file.write("static_assert(std::is_trivially_copyable_v<IROp_{}>);\n".format(op.Name))
|
||||
output_file.write("static_assert(std::is_standard_layout_v<IROp_{}>);\n\n".format(op.Name))
|
||||
|
||||
output_file.write("#undef IROP_STRUCTS\n")
|
||||
|
||||
@@ -24,9 +24,6 @@ set (SRCS
|
||||
Common/SoftFloat-3e/extF80_mul.c
|
||||
Common/SoftFloat-3e/extF80_rem.c
|
||||
Common/SoftFloat-3e/extF80_sqrt.c
|
||||
Common/SoftFloat-3e/s_add128.c
|
||||
Common/SoftFloat-3e/s_sub128.c
|
||||
Common/SoftFloat-3e/s_le128.c
|
||||
Common/SoftFloat-3e/extF80_to_i32.c
|
||||
Common/SoftFloat-3e/extF80_to_i64.c
|
||||
Common/SoftFloat-3e/extF80_to_ui64.c
|
||||
@@ -39,7 +36,6 @@ set (SRCS
|
||||
Common/SoftFloat-3e/s_roundToUI64.c
|
||||
Common/SoftFloat-3e/s_f128UIToCommonNaN.c
|
||||
Common/SoftFloat-3e/s_commonNaNToF128UI.c
|
||||
Common/SoftFloat-3e/s_shortShiftRight128.c
|
||||
Common/SoftFloat-3e/s_normSubnormalF128Sig.c
|
||||
Common/SoftFloat-3e/s_roundToI32.c
|
||||
Common/SoftFloat-3e/s_roundToI64.c
|
||||
@@ -48,22 +44,14 @@ set (SRCS
|
||||
Common/SoftFloat-3e/s_extF80UIToCommonNaN.c
|
||||
Common/SoftFloat-3e/s_commonNaNToF32UI.c
|
||||
Common/SoftFloat-3e/s_commonNaNToF64UI.c
|
||||
Common/SoftFloat-3e/s_shortShiftRightJam64.c
|
||||
Common/SoftFloat-3e/s_shortShiftRightJam64Extra.c
|
||||
Common/SoftFloat-3e/s_shiftRightJam64Extra.c
|
||||
Common/SoftFloat-3e/s_shortShiftRightJam64Extra.c
|
||||
Common/SoftFloat-3e/s_roundPackToF64.c
|
||||
Common/SoftFloat-3e/s_propagateNaNExtF80UI.c
|
||||
Common/SoftFloat-3e/s_roundPackToExtF80.c
|
||||
Common/SoftFloat-3e/s_normSubnormalExtF80Sig.c
|
||||
Common/SoftFloat-3e/s_shiftRightJam64.c
|
||||
Common/SoftFloat-3e/s_subMagsExtF80.c
|
||||
Common/SoftFloat-3e/s_shiftRightJam32.c
|
||||
Common/SoftFloat-3e/s_shiftRightJam128.c
|
||||
Common/SoftFloat-3e/s_shiftRightJam128Extra.c
|
||||
Common/SoftFloat-3e/s_normRoundPackToExtF80.c
|
||||
Common/SoftFloat-3e/s_shortShiftLeft128.c
|
||||
Common/SoftFloat-3e/s_approxRecip32_1.c
|
||||
Common/SoftFloat-3e/s_approxRecip_1Ks.c
|
||||
Common/SoftFloat-3e/s_approxRecipSqrt32_1.c
|
||||
Common/SoftFloat-3e/s_approxRecipSqrt_1Ks.c
|
||||
@@ -75,12 +63,6 @@ set (SRCS
|
||||
Common/SoftFloat-3e/extF80_roundToInt.c
|
||||
Common/SoftFloat-3e/extF80_eq.c
|
||||
Common/SoftFloat-3e/extF80_lt.c
|
||||
Common/SoftFloat-3e/s_lt128.c
|
||||
Common/SoftFloat-3e/s_mul64ByShifted32To128.c
|
||||
Common/SoftFloat-3e/s_mul64To128.c
|
||||
Common/SoftFloat-3e/s_countLeadingZeros8.c
|
||||
Common/SoftFloat-3e/s_countLeadingZeros32.c
|
||||
Common/SoftFloat-3e/s_countLeadingZeros64.c
|
||||
Common/SoftFloat-3e/f32_to_extF80.c
|
||||
Common/SoftFloat-3e/s_normSubnormalF32Sig.c
|
||||
Common/SoftFloat-3e/s_f32UIToCommonNaN.c
|
||||
@@ -88,6 +70,7 @@ set (SRCS
|
||||
Interface/Core/LookupCache.cpp
|
||||
Interface/Core/Core.cpp
|
||||
Interface/Core/CPUBackend.cpp
|
||||
Interface/Core/Addressing.cpp
|
||||
Interface/Core/CPUID.cpp
|
||||
Interface/Core/Frontend.cpp
|
||||
Interface/Core/ObjectCache/JobHandling.cpp
|
||||
@@ -105,6 +88,7 @@ set (SRCS
|
||||
Interface/Core/ArchHelpers/Arm64Emitter.cpp
|
||||
Interface/Core/Dispatcher/Dispatcher.cpp
|
||||
Interface/Core/Interpreter/Fallbacks/InterpreterFallbacks.cpp
|
||||
Interface/Core/Interpreter/Fallbacks/StringCompareFallbacks.cpp
|
||||
Interface/Core/JIT/JIT.cpp
|
||||
Interface/Core/JIT/ALUOps.cpp
|
||||
Interface/Core/JIT/AtomicOps.cpp
|
||||
@@ -175,7 +159,7 @@ else()
|
||||
endif()
|
||||
|
||||
# Some defines for the softfloat library
|
||||
list(APPEND DEFINES "-DSOFTFLOAT_BUILTIN_CLZ")
|
||||
list(APPEND DEFINES "-DSOFTFLOAT_BUILTIN_CLZ=1;-DINLINE=static inline;-DINLINE_LEVEL=4;-DSOFTFLOAT_FAST_INT64=1;-DSOFTFLOAT_FAST_DIV32TO16=1;-DSOFTFLOAT_FAST_DIV64TO32=1")
|
||||
|
||||
set (LIBS fmt::fmt xxHash::xxhash FEXHeaderUtils CodeEmitter)
|
||||
|
||||
@@ -337,6 +321,10 @@ add_library(FEXCore_Base STATIC ${FEXCORE_BASE_SRCS})
|
||||
target_link_libraries(FEXCore_Base ${LIBS})
|
||||
AddDefaultOptionsToTarget(FEXCore_Base)
|
||||
|
||||
if (ENABLE_FEXCORE_PROFILER AND FEXCORE_PROFILER_BACKEND STREQUAL "TRACY")
|
||||
target_link_libraries(FEXCore_Base TracyClient)
|
||||
endif()
|
||||
|
||||
function(AddObject Name Type)
|
||||
add_library(${Name} ${Type} ${SRCS})
|
||||
|
||||
|
||||
@@ -10,6 +10,8 @@
|
||||
#include <cstring>
|
||||
#include <stdint.h>
|
||||
|
||||
#include "Common/VectorRegType.h"
|
||||
|
||||
extern "C" {
|
||||
#include "SoftFloat-3e/platform.h"
|
||||
#include "SoftFloat-3e/softfloat.h"
|
||||
@@ -476,6 +478,12 @@ struct FEX_PACKED X80SoftFloat {
|
||||
return FEXCore::BitCast<double>(Result);
|
||||
}
|
||||
|
||||
FEXCore::VectorRegType ToVector() const {
|
||||
FEXCore::VectorRegType Ret {};
|
||||
memcpy(&Ret, this, sizeof(*this));
|
||||
return Ret;
|
||||
}
|
||||
|
||||
LIBRARY_PRECISION ToFMax(softfloat_state* state) const {
|
||||
#ifdef _WIN32
|
||||
return ToF64(state);
|
||||
@@ -567,12 +575,20 @@ struct FEX_PACKED X80SoftFloat {
|
||||
*this = i32_to_extF80(rhs);
|
||||
}
|
||||
|
||||
X80SoftFloat(const FEXCore::VectorRegType rhs) {
|
||||
memcpy(this, &rhs, sizeof(*this));
|
||||
}
|
||||
|
||||
void operator=(extFloat80_t rhs) {
|
||||
Significand = rhs.signif;
|
||||
Exponent = rhs.signExp & 0x7FFF;
|
||||
Sign = rhs.signExp >> 15;
|
||||
}
|
||||
|
||||
operator FEXCore::VectorRegType() const {
|
||||
return ToVector();
|
||||
}
|
||||
|
||||
operator extFloat80_t() const {
|
||||
extFloat80_t Result {};
|
||||
Result.signif = Significand;
|
||||
|
||||
@@ -19,12 +19,24 @@ static bool Conv(std::string_view Value, uint8_t* Result) {
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, int8_t* Result) {
|
||||
*Result = std::strtol(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, uint16_t* Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, int16_t* Result) {
|
||||
*Result = std::strtol(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, uint32_t* Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
@@ -42,6 +54,13 @@ static bool Conv(std::string_view Value, uint64_t* Result) {
|
||||
*Result = std::strtoull(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, int64_t* Result) {
|
||||
*Result = std::strtoll(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
template<typename T, typename = std::enable_if<std::is_enum<T>::value, T>>
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, T* Result) {
|
||||
|
||||
@@ -0,0 +1,16 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#ifdef _M_X86_64
|
||||
#include <xmmintrin.h>
|
||||
#endif
|
||||
|
||||
namespace FEXCore {
|
||||
#ifdef _M_ARM_64
|
||||
// Can't use uint8x16_t directly from arm_neon.h here.
|
||||
// Overrides softfloat-3e's defines which causes problems.
|
||||
using VectorRegType = __attribute__((neon_vector_type(16))) uint8_t;
|
||||
#elif defined(_M_X86_64)
|
||||
using VectorRegType = __m128i;
|
||||
#endif
|
||||
} // namespace FEXCore
|
||||
@@ -19,12 +19,10 @@
|
||||
|
||||
#include <array>
|
||||
#include <cstdlib>
|
||||
#include <functional>
|
||||
#include <optional>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <string_view>
|
||||
#include <system_error>
|
||||
#include <type_traits>
|
||||
#include <utility>
|
||||
|
||||
@@ -113,16 +111,9 @@ fextl::string GetApplicationConfig(const std::string_view Program, bool Global)
|
||||
return fextl::fmt::format("{}{}.json", ConfigFile, Program);
|
||||
}
|
||||
|
||||
void SetConfig(FEXCore::Context::Context* CTX, ConfigOption Option, uint64_t Config) {}
|
||||
|
||||
void SetConfig(FEXCore::Context::Context* CTX, ConfigOption Option, const fextl::string& Config) {}
|
||||
|
||||
uint64_t GetConfig(FEXCore::Context::Context* CTX, ConfigOption Option) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
static fextl::map<FEXCore::Config::LayerType, fextl::unique_ptr<FEXCore::Config::Layer>> ConfigLayers;
|
||||
static FEXCore::Config::Layer* Meta {};
|
||||
class MetaLayer;
|
||||
static FEXCore::Config::MetaLayer* Meta {};
|
||||
|
||||
constexpr std::array<FEXCore::Config::LayerType, 10> LoadOrder = {
|
||||
FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN, FEXCore::Config::LayerType::LAYER_MAIN,
|
||||
@@ -143,9 +134,39 @@ public:
|
||||
~MetaLayer() {}
|
||||
void Load();
|
||||
|
||||
template<typename T>
|
||||
requires (!std::is_same_v<fextl::string, T> && !std::is_same_v<DefaultValues::Type::StringArrayType, T>)
|
||||
std::optional<T> GetConv(ConfigOption Option) {
|
||||
const auto it = OptionMap.find(Option);
|
||||
if (it == OptionMap.end()) {
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
const auto& Value = it->second;
|
||||
LOGMAN_THROW_A_FMT(!std::holds_alternative<DefaultValues::Type::StringArrayType>(Value), "Tried to get config of invalid type!");
|
||||
|
||||
if (std::holds_alternative<T>(Value)) [[likely]] {
|
||||
return std::get<T>(Value);
|
||||
}
|
||||
|
||||
T ConvertedValue;
|
||||
if (std::holds_alternative<fextl::string>(Value)) {
|
||||
const auto& StrVal = std::get<fextl::string>(Value);
|
||||
if (FEXCore::StrConv::Conv(StrVal, &ConvertedValue)) {
|
||||
// Convert the value.
|
||||
OptionMap[Option].emplace<T>(ConvertedValue);
|
||||
return ConvertedValue;
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Couldn't Convert {} to specified type!", StrVal);
|
||||
}
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
private:
|
||||
void MergeConfigMap(const LayerOptions& Options);
|
||||
void MergeEnvironmentVariables(const ConfigOption& Option, const LayerValue& Value);
|
||||
void MergeEnvironmentVariables(const ConfigOption& Option, const DefaultValues::Type::StringArrayType& Value);
|
||||
};
|
||||
|
||||
void MetaLayer::Load() {
|
||||
@@ -161,7 +182,7 @@ void MetaLayer::Load() {
|
||||
}
|
||||
|
||||
|
||||
void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const LayerValue& Value) {
|
||||
void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const DefaultValues::Type::StringArrayType& Value) {
|
||||
// Environment variables need a bit of additional work
|
||||
// We want to merge the arrays rather than overwrite entirely
|
||||
auto MetaEnvironment = OptionMap.find(Option);
|
||||
@@ -173,7 +194,7 @@ void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const Laye
|
||||
|
||||
// If an environment variable exists in both current meta and in the incoming layer then the meta layer value is overwritten
|
||||
fextl::unordered_map<fextl::string, fextl::string> LookupMap;
|
||||
const auto AddToMap = [&LookupMap](const FEXCore::Config::LayerValue& Value) {
|
||||
const auto AddToMap = [&LookupMap](const DefaultValues::Type::StringArrayType& Value) {
|
||||
for (const auto& EnvVar : Value) {
|
||||
const auto ItEq = EnvVar.find_first_of('=');
|
||||
if (ItEq == fextl::string::npos) {
|
||||
@@ -189,7 +210,7 @@ void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const Laye
|
||||
}
|
||||
};
|
||||
|
||||
AddToMap(MetaEnvironment->second);
|
||||
AddToMap(std::get<DefaultValues::Type::StringArrayType>(MetaEnvironment->second));
|
||||
AddToMap(Value);
|
||||
|
||||
// Now with the two layers merged in the map
|
||||
@@ -197,7 +218,7 @@ void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const Laye
|
||||
Erase(Option);
|
||||
for (auto& Val : LookupMap) {
|
||||
// Set will emplace multiple options in to its list
|
||||
Set(Option, Val.first + "=" + Val.second);
|
||||
AppendStrArrayValue(Option, Val.first + "=" + Val.second);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -205,7 +226,8 @@ void MetaLayer::MergeConfigMap(const LayerOptions& Options) {
|
||||
// Insert this layer's options, overlaying previous options that exist here
|
||||
for (auto& it : Options) {
|
||||
if (it.first == FEXCore::Config::ConfigOption::CONFIG_ENV || it.first == FEXCore::Config::ConfigOption::CONFIG_HOSTENV) {
|
||||
MergeEnvironmentVariables(it.first, it.second);
|
||||
LOGMAN_THROW_A_FMT(std::holds_alternative<DefaultValues::Type::StringArrayType>(it.second), "Tried to get config of invalid type!");
|
||||
MergeEnvironmentVariables(it.first, std::get<DefaultValues::Type::StringArrayType>(it.second));
|
||||
} else {
|
||||
OptionMap.insert_or_assign(it.first, it.second);
|
||||
}
|
||||
@@ -214,7 +236,7 @@ void MetaLayer::MergeConfigMap(const LayerOptions& Options) {
|
||||
|
||||
void Initialize() {
|
||||
AddLayer(fextl::make_unique<MetaLayer>(FEXCore::Config::LayerType::LAYER_TOP));
|
||||
Meta = ConfigLayers.begin()->second.get();
|
||||
Meta = dynamic_cast<MetaLayer*>(ConfigLayers.begin()->second.get());
|
||||
}
|
||||
|
||||
void Shutdown() {
|
||||
@@ -322,7 +344,7 @@ void ReloadMetaLayer() {
|
||||
auto ExpandPathIfExists = [&ContainerPrefix](FEXCore::Config::ConfigOption Config, const fextl::string& PathName) {
|
||||
const auto NewPath = ExpandPath(ContainerPrefix, PathName);
|
||||
if (!NewPath.empty()) {
|
||||
FEXCore::Config::EraseSet(Config, NewPath);
|
||||
FEXCore::Config::Set(Config, NewPath);
|
||||
}
|
||||
};
|
||||
|
||||
@@ -331,7 +353,7 @@ void ReloadMetaLayer() {
|
||||
const auto ExpandedString = ExpandPath(ContainerPrefix, *PathName);
|
||||
if (!ExpandedString.empty()) {
|
||||
// Adjust the path if it ended up being relative
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, ExpandedString);
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_ROOTFS, ExpandedString);
|
||||
} else if (!PathName->empty()) {
|
||||
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
|
||||
const auto PathNameCopy = *PathName;
|
||||
@@ -339,7 +361,7 @@ void ReloadMetaLayer() {
|
||||
for (auto DirectoryFetchers : {GetDataDirectory, GetConfigDirectory}) {
|
||||
fextl::string NamedRootFS = DirectoryFetchers(Global) + "RootFS/" + PathNameCopy;
|
||||
if (FHU::Filesystem::Exists(NamedRootFS)) {
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_ROOTFS, NamedRootFS);
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_ROOTFS, NamedRootFS);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -358,7 +380,7 @@ void ReloadMetaLayer() {
|
||||
const auto ExpandedString = ExpandPath(ContainerPrefix, *PathName);
|
||||
if (!ExpandedString.empty()) {
|
||||
// Adjust the path if it ended up being relative
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, ExpandedString);
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_THUNKCONFIG, ExpandedString);
|
||||
} else if (!PathName->empty()) {
|
||||
// If the filesystem doesn't exist then let's see if it exists in the fex-emu folder
|
||||
const auto PathNameCopy = *PathName;
|
||||
@@ -366,7 +388,7 @@ void ReloadMetaLayer() {
|
||||
for (auto DirectoryFetchers : {GetDataDirectory, GetConfigDirectory}) {
|
||||
fextl::string NamedConfig = DirectoryFetchers(Global) + "ThunkConfigs/" + PathNameCopy;
|
||||
if (FHU::Filesystem::Exists(NamedConfig)) {
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_THUNKCONFIG, NamedConfig);
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_THUNKCONFIG, NamedConfig);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -383,12 +405,12 @@ void ReloadMetaLayer() {
|
||||
// If DumpIR is set but no PassManagerDumpIR configuration is set, then default to `afteropt`
|
||||
const auto PathName = *Meta->Get(FEXCore::Config::CONFIG_DUMPIR);
|
||||
if (*PathName != "no") {
|
||||
EraseSet(FEXCore::Config::ConfigOption::CONFIG_PASSMANAGERDUMPIR,
|
||||
fextl::fmt::format("{}", static_cast<uint64_t>(FEXCore::Config::PassManagerDumpIR::AFTEROPT)));
|
||||
Set(FEXCore::Config::ConfigOption::CONFIG_PASSMANAGERDUMPIR,
|
||||
fextl::fmt::format("{}", static_cast<uint64_t>(FEXCore::Config::PassManagerDumpIR::AFTEROPT)));
|
||||
}
|
||||
}
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_SINGLESTEP)) {
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_SINGLESTEP) && Meta->GetConv<bool>(FEXCore::Config::CONFIG_SINGLESTEP).value_or(false)) {
|
||||
// Single stepping also enforces single instruction size blocks
|
||||
Set(FEXCore::Config::ConfigOption::CONFIG_MAXINST, "1");
|
||||
}
|
||||
@@ -402,7 +424,7 @@ bool Exists(ConfigOption Option) {
|
||||
return Meta->OptionExists(Option);
|
||||
}
|
||||
|
||||
std::optional<LayerValue*> All(ConfigOption Option) {
|
||||
std::optional<DefaultValues::Type::StringArrayType*> All(ConfigOption Option) {
|
||||
return Meta->All(Option);
|
||||
}
|
||||
|
||||
@@ -410,6 +432,11 @@ std::optional<fextl::string*> Get(ConfigOption Option) {
|
||||
return Meta->Get(Option);
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
std::optional<T> GetConv(ConfigOption Option) {
|
||||
return Meta->GetConv<T>(Option);
|
||||
}
|
||||
|
||||
void Set(ConfigOption Option, std::string_view Data) {
|
||||
Meta->Set(Option, Data);
|
||||
}
|
||||
@@ -418,31 +445,14 @@ void Erase(ConfigOption Option) {
|
||||
Meta->Erase(Option);
|
||||
}
|
||||
|
||||
void EraseSet(ConfigOption Option, std::string_view Data) {
|
||||
Meta->EraseSet(Option, Data);
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
T Value<T>::Get(FEXCore::Config::ConfigOption Option) {
|
||||
T Result;
|
||||
auto Value = FEXCore::Config::Get(Option);
|
||||
|
||||
if (!FEXCore::StrConv::Conv(**Value, &Result)) {
|
||||
LOGMAN_MSG_A_FMT("Attempted to convert invalid value");
|
||||
}
|
||||
return Result;
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
T Value<T>::GetIfExists(FEXCore::Config::ConfigOption Option, T Default) {
|
||||
T Result;
|
||||
auto Value = FEXCore::Config::Get(Option);
|
||||
|
||||
if (Value && FEXCore::StrConv::Conv(**Value, &Result)) {
|
||||
return Result;
|
||||
} else {
|
||||
return Default;
|
||||
auto Value = FEXCore::Config::GetConv<T>(Option);
|
||||
if (Value) {
|
||||
return *Value;
|
||||
}
|
||||
|
||||
return Default;
|
||||
}
|
||||
|
||||
template<>
|
||||
@@ -482,12 +492,13 @@ template Value<uint8_t>::Value(FEXCore::Config::ConfigOption _Option, uint8_t De
|
||||
template Value<uint64_t>::Value(FEXCore::Config::ConfigOption _Option, uint64_t Default);
|
||||
|
||||
template<typename T>
|
||||
void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string>* List) {
|
||||
void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, DefaultValues::Type::StringArrayType* List) {
|
||||
auto Value = FEXCore::Config::All(Option);
|
||||
List->clear();
|
||||
if (Value) {
|
||||
*List = **Value;
|
||||
}
|
||||
}
|
||||
template void Value<fextl::string>::GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string>* List);
|
||||
template void Value<DefaultValues::Type::StringArrayType>::GetListIfExists(FEXCore::Config::ConfigOption Option,
|
||||
DefaultValues::Type::StringArrayType* List);
|
||||
} // namespace FEXCore::Config
|
||||
@@ -3,7 +3,7 @@
|
||||
"CPU": {
|
||||
"Multiblock": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Default": "true",
|
||||
"ShortArg": "m",
|
||||
"Desc": [
|
||||
"Controls multiblock code compilation",
|
||||
@@ -59,6 +59,8 @@
|
||||
"DISABLEFLAGM": "disableflagm",
|
||||
"ENABLEFLAGM2": "enableflagm2",
|
||||
"DISABLEFLAGM2": "disableflagm2",
|
||||
"ENABLEFRINTTS": "enablefrintts",
|
||||
"DISABLEFRINTTS": "disablefrintts",
|
||||
"ENABLECRYPTO": "enablecrypto",
|
||||
"DISABLECRYPTO": "disablecrypto",
|
||||
"ENABLERPRES": "enablerpres",
|
||||
@@ -90,19 +92,6 @@
|
||||
"\t{enable,disable}preserveallabi: Will force enable or disable preserve_all abi even if the host doesn't support it"
|
||||
]
|
||||
},
|
||||
"CPUID": {
|
||||
"Type": "strenum",
|
||||
"Default": "FEXCore::Config::CPUID::OFF",
|
||||
"Enums": {
|
||||
"ENABLESHA": "enablesha",
|
||||
"DISABLESHA": "disablesha"
|
||||
},
|
||||
"Desc": [
|
||||
"Allows controlling of the CPU features are exposed in CPUID.",
|
||||
"\toff: Default CPU features queried from CPU features",
|
||||
"\t{enable,disable}sha: Will force enable or disable sha even if the host doesn't support it"
|
||||
]
|
||||
},
|
||||
"SmallTSCScale": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
@@ -363,6 +352,14 @@
|
||||
"Redirects the telemetry folder that FEX usually writes to.",
|
||||
"By default telemetry data is stored in {$FEX_APP_DATA_LOCATION,{$XDG_DATA_HOME,$HOME}/.fex-emu/Telemetry/}"
|
||||
]
|
||||
},
|
||||
"ProfileStats": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Enables FEX's low-overhead sampling profile statistics.",
|
||||
"Requires a supported version of Mangohud to see the results"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Hacks": {
|
||||
@@ -425,6 +422,14 @@
|
||||
"Should work without issues in most cases."
|
||||
]
|
||||
},
|
||||
"VolatileMetadata": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"Use volatile metadata in PE files to inform TSO instructions when available.",
|
||||
"When metadata is unavailable falls back to the currently enabled TSO options."
|
||||
]
|
||||
},
|
||||
"X87ReducedPrecision": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
|
||||
@@ -186,6 +186,10 @@ public:
|
||||
|
||||
void AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t GuestThunkEntrypoint) override;
|
||||
|
||||
void AddForceTSOInformation(const IntervalList<uint64_t>& ValidRanges, fextl::set<uint64_t>&& Instructions) override;
|
||||
|
||||
void RemoveForceTSOInformation(uint64_t Address, uint64_t Size) override;
|
||||
|
||||
public:
|
||||
friend class FEXCore::HLE::SyscallHandler;
|
||||
#ifdef JIT_ARM64
|
||||
@@ -248,8 +252,6 @@ public:
|
||||
~ContextImpl();
|
||||
|
||||
static void ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP);
|
||||
static void ThreadAddBlockLink(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestDestination,
|
||||
FEXCore::Context::ExitFunctionLinkData* HostLink, const BlockDelinkerFunc& delinker);
|
||||
|
||||
template<auto Fn>
|
||||
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, ExitFunctionLinkData* Record) {
|
||||
@@ -360,8 +362,6 @@ private:
|
||||
*/
|
||||
void InitializeCompiler(FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
void AddBlockMapping(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, void* Ptr);
|
||||
|
||||
IR::AOTIRCaptureCache IRCaptureCache;
|
||||
fextl::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
|
||||
|
||||
@@ -377,5 +377,7 @@ private:
|
||||
std::shared_mutex CustomIRMutex;
|
||||
std::atomic<bool> HasCustomIRHandlers {};
|
||||
fextl::unordered_map<uint64_t, std::tuple<CustomIREntrypointHandler, void*, void*>> CustomIRHandlers;
|
||||
IntervalList<uint64_t> ForceTSOValidRanges; // The ranges for which ForceTSOInstructions has populated data
|
||||
fextl::set<uint64_t> ForceTSOInstructions;
|
||||
};
|
||||
} // namespace FEXCore::Context
|
||||
@@ -0,0 +1,158 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Core/Addressing.h"
|
||||
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "FEXCore/Utils/MathUtils.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage) {
|
||||
Ref Tmp = A.Base;
|
||||
|
||||
if (A.Offset) {
|
||||
Ref Offset = IREmit->_Constant(A.Offset);
|
||||
Tmp = Tmp ? IREmit->_Add(GPRSize, Tmp, Offset) : Offset;
|
||||
}
|
||||
|
||||
if (A.Index) {
|
||||
if (A.IndexScale != 1) {
|
||||
uint32_t Log2 = FEXCore::ilog2(A.IndexScale);
|
||||
|
||||
if (Tmp) {
|
||||
Tmp = IREmit->_AddShift(GPRSize, Tmp, A.Index, ShiftType::LSL, Log2);
|
||||
} else {
|
||||
Tmp = IREmit->_Lshl(GPRSize, A.Index, IREmit->_Constant(Log2));
|
||||
}
|
||||
} else {
|
||||
Tmp = Tmp ? IREmit->_Add(GPRSize, Tmp, A.Index) : A.Index;
|
||||
}
|
||||
}
|
||||
|
||||
// For 64-bit AddrSize can be 32-bit or 64-bit
|
||||
// For 32-bit AddrSize can be 32-bit or 16-bit
|
||||
//
|
||||
// If the AddrSize is not the GPRSize then we need to clear the upper bits.
|
||||
if ((A.AddrSize < GPRSize) && !AllowUpperGarbage && Tmp) {
|
||||
Tmp = IREmit->_Bfe(GPRSize, IR::OpSizeAsBits(A.AddrSize), 0, Tmp);
|
||||
}
|
||||
|
||||
if (A.Segment && AddSegmentBase) {
|
||||
Tmp = Tmp ? IREmit->_Add(GPRSize, Tmp, A.Segment) : A.Segment;
|
||||
}
|
||||
|
||||
return Tmp ?: IREmit->_Constant(0);
|
||||
}
|
||||
|
||||
AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO, bool Vector,
|
||||
IR::OpSize AccessSize) {
|
||||
auto SoftwareAddressCalculation = [IREmit, &A, GPRSize]() -> AddressMode {
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(IREmit, A, GPRSize, true),
|
||||
.Index = IREmit->Invalid(),
|
||||
};
|
||||
};
|
||||
|
||||
const auto Is32Bit = GPRSize == OpSize::i32Bit;
|
||||
const auto GPRSizeMatchesAddrSize = A.AddrSize == GPRSize;
|
||||
const auto OffsetIndexToLargeFor32Bit = Is32Bit && (A.Offset <= -16384 || A.Offset >= 16384);
|
||||
if (!GPRSizeMatchesAddrSize || OffsetIndexToLargeFor32Bit) {
|
||||
// If address size doesn't match GPR size then no optimizations can occur.
|
||||
return SoftwareAddressCalculation();
|
||||
}
|
||||
|
||||
// Loadstore rules:
|
||||
// Non-TSO GPR:
|
||||
// * LDR/STR: [Reg]
|
||||
// * LDR/STR: [Reg + Reg, {Shift <AccessSize>}]
|
||||
// * Can't use with 32-bit
|
||||
// * LDR/STR: [Reg + [0,4095] * <AccessSize>]
|
||||
// * Imm must be smaller than 16k with 32-bit
|
||||
// * LDUR/STUR: [Reg + [-256, 255]]
|
||||
//
|
||||
// TSO GPR:
|
||||
// * ARMv8.0:
|
||||
// LDAR/STLR: [Reg]
|
||||
// * FEAT_LRCPC:
|
||||
// LDAPR: [Reg]
|
||||
// * FEAT_LRCPC2:
|
||||
// LDAPUR/STLUR: [Reg + [-256, 255]]
|
||||
//
|
||||
// Non-TSO Vector:
|
||||
// * LDR/STR: [Reg + [0,4095] * <AccessSize>]
|
||||
// * LDUR/STUR: [Reg + [-256,255]]
|
||||
//
|
||||
// TSO Vector:
|
||||
// * ARMv8.0:
|
||||
// Just DMB + previous
|
||||
// * FEAT_LRCPC3 (Unsupported by FEXCore currently):
|
||||
// LDAPUR/STLUR: [Reg + [-256,255]]
|
||||
|
||||
const auto AccessSizeAsImm = OpSizeToSize(AccessSize);
|
||||
const bool OffsetIsSIMM9 = A.Offset && A.Offset >= -256 && A.Offset <= 255;
|
||||
const bool OffsetIsUnsignedScaled = A.Offset > 0 && (A.Offset & (AccessSizeAsImm - 1)) == 0 && (A.Offset / AccessSizeAsImm) <= 4095;
|
||||
|
||||
auto InlineImmOffsetLoadstore = [IREmit, &GPRSize](AddressMode A) -> AddressMode {
|
||||
// Peel off the offset
|
||||
AddressMode B = A;
|
||||
B.Offset = 0;
|
||||
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(IREmit, B, GPRSize, true /* AddSegmentBase */, false),
|
||||
.Index = IREmit->_Constant(A.Offset),
|
||||
.IndexType = MEM_OFFSET_SXTX,
|
||||
.IndexScale = 1,
|
||||
};
|
||||
};
|
||||
|
||||
auto ScaledRegisterLoadstore = [IREmit, GPRSize](AddressMode A) -> AddressMode {
|
||||
if (A.Index && A.Segment) {
|
||||
A.Base = IREmit->_Add(GPRSize, A.Base, A.Segment);
|
||||
} else if (A.Segment) {
|
||||
A.Index = A.Segment;
|
||||
A.IndexScale = 1;
|
||||
}
|
||||
return A;
|
||||
};
|
||||
|
||||
if (AtomicTSO) {
|
||||
if (!Vector) {
|
||||
if (HostSupportsTSOImm9 && OffsetIsSIMM9) {
|
||||
return InlineImmOffsetLoadstore(A);
|
||||
}
|
||||
} else {
|
||||
// TODO: LRCPC3 support for vector Imm9.
|
||||
}
|
||||
} else {
|
||||
if (OffsetIsSIMM9 || OffsetIsUnsignedScaled) {
|
||||
return InlineImmOffsetLoadstore(A);
|
||||
} else if (!Is32Bit && A.Base && (A.Index || A.Segment) && !A.Offset && (A.IndexScale == 1 || A.IndexScale == AccessSizeAsImm)) {
|
||||
return ScaledRegisterLoadstore(A);
|
||||
}
|
||||
}
|
||||
|
||||
if (Vector || !AtomicTSO) {
|
||||
if ((A.Base || A.Segment) && A.Offset) {
|
||||
const bool Const_16K = A.Offset > -16384 && A.Offset < 16384 && GPRSizeMatchesAddrSize && Is32Bit;
|
||||
|
||||
if (!Is32Bit || Const_16K) {
|
||||
// Peel off the offset
|
||||
AddressMode B = A;
|
||||
B.Offset = 0;
|
||||
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(IREmit, B, GPRSize, true /* AddSegmentBase */, false),
|
||||
.Index = IREmit->_Constant(A.Offset),
|
||||
.IndexType = MEM_OFFSET_SXTX,
|
||||
.IndexScale = 1,
|
||||
};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Fallback on software address calculation
|
||||
return SoftwareAddressCalculation();
|
||||
}
|
||||
|
||||
|
||||
}; // namespace FEXCore::IR
|
||||
@@ -0,0 +1,27 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include "Interface/IR/IR.h"
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class IREmitter;
|
||||
|
||||
struct AddressMode {
|
||||
Ref Segment {nullptr};
|
||||
Ref Base {nullptr};
|
||||
Ref Index {nullptr};
|
||||
MemOffsetType IndexType = MEM_OFFSET_SXTX;
|
||||
uint8_t IndexScale = 1;
|
||||
int64_t Offset = 0;
|
||||
|
||||
// Size in bytes for the address calculation. 8 for an arm64 hardware mode.
|
||||
IR::OpSize AddrSize;
|
||||
bool NonTSO;
|
||||
};
|
||||
|
||||
Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage = false);
|
||||
AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO, bool Vector,
|
||||
IR::OpSize AccessSize);
|
||||
|
||||
}; // namespace FEXCore::IR
|
||||
@@ -24,6 +24,22 @@
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
// LLVM's preserve_all doc, this is used throughout this file and reproduced
|
||||
// here for reference:
|
||||
//
|
||||
// the callee preserve all general purpose registers,
|
||||
// except X0-X8 and X16-X18. Furthermore it also preserves lower 128 bits of
|
||||
// V8-V31 SIMD - floating point registers.
|
||||
//
|
||||
// Note that the call necessarily also clobbers x30, the link register (LR)
|
||||
// which is not considered general purpose.
|
||||
//
|
||||
// Meanwhile, for non-preserve_all, the AAPCS64 ABI says:
|
||||
//
|
||||
// A subroutine invocation must preserve the contents of the registers
|
||||
// r19-r29 and SP.
|
||||
|
||||
namespace x64 {
|
||||
#ifndef _M_ARM_64EC
|
||||
// All but x19 and x29 are caller saved
|
||||
@@ -50,6 +66,13 @@ namespace x64 {
|
||||
REG_AF,
|
||||
};
|
||||
|
||||
// I wish this could get constexpr generated from SRA's definition but impossible until libstdc++12, libc++15.
|
||||
// SRA GPRs that need to be spilled when calling a function with `preserve_all` ABI.
|
||||
constexpr std::array<ARMEmitter::Register, 7> PreserveAll_SRA = {
|
||||
ARMEmitter::Reg::r4, ARMEmitter::Reg::r5, ARMEmitter::Reg::r6, ARMEmitter::Reg::r7,
|
||||
ARMEmitter::Reg::r8, ARMEmitter::Reg::r16, ARMEmitter::Reg::r17,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::Register, 8> RA = {
|
||||
// All these callee saved
|
||||
ARMEmitter::Reg::r20, ARMEmitter::Reg::r21, ARMEmitter::Reg::r22, ARMEmitter::Reg::r23,
|
||||
@@ -58,6 +81,14 @@ namespace x64 {
|
||||
|
||||
constexpr unsigned RAPairs = 6;
|
||||
|
||||
// Dynamic GPRs
|
||||
constexpr std::array<ARMEmitter::Register, 2> PreserveAll_Dynamic = {
|
||||
ARMEmitter::Reg::r18,
|
||||
ARMEmitter::Reg::r30,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::Register, 2> NotPreserved_Dynamic = PreserveAll_Dynamic;
|
||||
|
||||
// All are caller saved
|
||||
constexpr std::array<ARMEmitter::VRegister, 16> SRAFPR = {
|
||||
ARMEmitter::VReg::v16, ARMEmitter::VReg::v17, ARMEmitter::VReg::v18, ARMEmitter::VReg::v19,
|
||||
@@ -65,6 +96,11 @@ namespace x64 {
|
||||
ARMEmitter::VReg::v24, ARMEmitter::VReg::v25, ARMEmitter::VReg::v26, ARMEmitter::VReg::v27,
|
||||
ARMEmitter::VReg::v28, ARMEmitter::VReg::v29, ARMEmitter::VReg::v30, ARMEmitter::VReg::v31};
|
||||
|
||||
// SRA FPRs that need to be spilled when calling a function with `preserve_all` ABI.
|
||||
constexpr std::array<ARMEmitter::Register, 0> PreserveAll_SRAFPR = {
|
||||
// None.
|
||||
};
|
||||
|
||||
// v8..v15 = (lower 64bits) Callee saved
|
||||
constexpr std::array<ARMEmitter::VRegister, 14> RAFPR = {
|
||||
// v0 ~ v1 are used as temps.
|
||||
@@ -74,6 +110,10 @@ namespace x64 {
|
||||
ARMEmitter::VReg::v7, ARMEmitter::VReg::v8, ARMEmitter::VReg::v9, ARMEmitter::VReg::v10, ARMEmitter::VReg::v11,
|
||||
ARMEmitter::VReg::v12, ARMEmitter::VReg::v13, ARMEmitter::VReg::v14, ARMEmitter::VReg::v15,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::VRegister, 6> PreserveAll_DynamicFPR = {
|
||||
ARMEmitter::VReg::v2, ARMEmitter::VReg::v3, ARMEmitter::VReg::v4, ARMEmitter::VReg::v5, ARMEmitter::VReg::v6, ARMEmitter::VReg::v7,
|
||||
};
|
||||
#else
|
||||
constexpr std::array<ARMEmitter::Register, 18> SRA = {
|
||||
ARMEmitter::Reg::r8,
|
||||
@@ -98,11 +138,22 @@ namespace x64 {
|
||||
REG_AF,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::Register, 7> PreserveAll_SRA = {
|
||||
ARMEmitter::Reg::r0, ARMEmitter::Reg::r1, ARMEmitter::Reg::r2, ARMEmitter::Reg::r3,
|
||||
ARMEmitter::Reg::r4, ARMEmitter::Reg::r5, ARMEmitter::Reg::r8,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::Register, 7> RA = {
|
||||
ARMEmitter::Reg::r6, ARMEmitter::Reg::r7, ARMEmitter::Reg::r14, ARMEmitter::Reg::r15,
|
||||
ARMEmitter::Reg::r16, ARMEmitter::Reg::r17, ARMEmitter::Reg::r30,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::Register, 5> PreserveAll_Dynamic = {
|
||||
ARMEmitter::Reg::r6, ARMEmitter::Reg::r7, ARMEmitter::Reg::r16, ARMEmitter::Reg::r17, ARMEmitter::Reg::r30,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::Register, 7> NotPreserved_Dynamic = RA;
|
||||
|
||||
constexpr unsigned RAPairs = 6;
|
||||
|
||||
constexpr std::array<ARMEmitter::VRegister, 16> SRAFPR = {
|
||||
@@ -112,18 +163,20 @@ namespace x64 {
|
||||
ARMEmitter::VReg::v12, ARMEmitter::VReg::v13, ARMEmitter::VReg::v14, ARMEmitter::VReg::v15,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::VRegister, 8> PreserveAll_SRAFPR = {
|
||||
ARMEmitter::VReg::v0, ARMEmitter::VReg::v1, ARMEmitter::VReg::v2, ARMEmitter::VReg::v3,
|
||||
ARMEmitter::VReg::v4, ARMEmitter::VReg::v5, ARMEmitter::VReg::v6, ARMEmitter::VReg::v7,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::VRegister, 14> RAFPR = {
|
||||
ARMEmitter::VReg::v18, ARMEmitter::VReg::v19, ARMEmitter::VReg::v20, ARMEmitter::VReg::v21, ARMEmitter::VReg::v22,
|
||||
ARMEmitter::VReg::v23, ARMEmitter::VReg::v24, ARMEmitter::VReg::v25, ARMEmitter::VReg::v26, ARMEmitter::VReg::v27,
|
||||
ARMEmitter::VReg::v28, ARMEmitter::VReg::v29, ARMEmitter::VReg::v30, ARMEmitter::VReg::v31};
|
||||
#endif
|
||||
|
||||
// I wish this could get constexpr generated from SRA's definition but impossible until libstdc++12, libc++15.
|
||||
// SRA GPRs that need to be spilled when calling a function with `preserve_all` ABI.
|
||||
constexpr std::array<ARMEmitter::Register, 7> PreserveAll_SRA = {
|
||||
ARMEmitter::Reg::r4, ARMEmitter::Reg::r5, ARMEmitter::Reg::r6, ARMEmitter::Reg::r7,
|
||||
ARMEmitter::Reg::r8, ARMEmitter::Reg::r16, ARMEmitter::Reg::r17,
|
||||
constexpr std::array<ARMEmitter::VRegister, 0> PreserveAll_DynamicFPR = {
|
||||
// None
|
||||
};
|
||||
#endif
|
||||
|
||||
constexpr uint32_t PreserveAll_SRAMask = {[]() -> uint32_t {
|
||||
uint32_t Mask {};
|
||||
@@ -147,16 +200,6 @@ namespace x64 {
|
||||
return Mask;
|
||||
}()};
|
||||
|
||||
// Dynamic GPRs
|
||||
constexpr std::array<ARMEmitter::Register, 1> PreserveAll_Dynamic = {
|
||||
// Only LR needs to get saved.
|
||||
ARMEmitter::Reg::r30};
|
||||
|
||||
// SRA FPRs that need to be spilled when calling a function with `preserve_all` ABI.
|
||||
constexpr std::array<ARMEmitter::Register, 0> PreserveAll_SRAFPR = {
|
||||
// None.
|
||||
};
|
||||
|
||||
constexpr uint32_t PreserveAll_SRAFPRMask = {[]() -> uint32_t {
|
||||
uint32_t Mask {};
|
||||
for (auto Reg : PreserveAll_SRAFPR) {
|
||||
@@ -165,13 +208,6 @@ namespace x64 {
|
||||
return Mask;
|
||||
}()};
|
||||
|
||||
// Dynamic FPRs
|
||||
// - v0-v7
|
||||
constexpr std::array<ARMEmitter::VRegister, 6> PreserveAll_DynamicFPR = {
|
||||
// v0 ~ v1 are temps
|
||||
ARMEmitter::VReg::v2, ARMEmitter::VReg::v3, ARMEmitter::VReg::v4, ARMEmitter::VReg::v5, ARMEmitter::VReg::v6, ARMEmitter::VReg::v7,
|
||||
};
|
||||
|
||||
// SRA FPRs that need to be spilled when the host supports SVE-256bit with `preserve_all` ABI.
|
||||
// This is /all/ of the SRA registers
|
||||
constexpr std::array<ARMEmitter::VRegister, 16> PreserveAll_SRAFPRSVE = SRAFPR;
|
||||
@@ -232,6 +268,11 @@ namespace x32 {
|
||||
ARMEmitter::Reg::r19,
|
||||
};
|
||||
|
||||
constexpr std::array<ARMEmitter::Register, 7> NotPreserved_Dynamic = {
|
||||
ARMEmitter::Reg::r12, ARMEmitter::Reg::r13, ARMEmitter::Reg::r14, ARMEmitter::Reg::r15,
|
||||
ARMEmitter::Reg::r16, ARMEmitter::Reg::r17, ARMEmitter::Reg::r30,
|
||||
};
|
||||
|
||||
constexpr unsigned RAPairs = 12;
|
||||
|
||||
// All are caller saved
|
||||
@@ -284,17 +325,7 @@ namespace x32 {
|
||||
constexpr std::array<ARMEmitter::Register, 3> PreserveAll_Dynamic = {ARMEmitter::Reg::r16, ARMEmitter::Reg::r17, ARMEmitter::Reg::r30};
|
||||
|
||||
// SRA FPRs that need to be spilled when calling a function with `preserve_all` ABI.
|
||||
constexpr std::array<ARMEmitter::Register, 0> PreserveAll_SRAFPR = {
|
||||
// None.
|
||||
};
|
||||
|
||||
constexpr uint32_t PreserveAll_SRAFPRMask = {[]() -> uint32_t {
|
||||
uint32_t Mask {};
|
||||
for (auto Reg : PreserveAll_SRAFPR) {
|
||||
Mask |= (1U << Reg.Idx());
|
||||
}
|
||||
return Mask;
|
||||
}()};
|
||||
constexpr uint32_t PreserveAll_SRAFPRMask = 0;
|
||||
|
||||
// Dynamic FPRs
|
||||
// - v0-v7
|
||||
@@ -355,18 +386,16 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr
|
||||
if (EmitterCTX->Config.Is64BitMode()) {
|
||||
StaticRegisters = x64::SRA;
|
||||
GeneralRegisters = x64::RA;
|
||||
GeneralRegistersNotPreserved = x64::NotPreserved_Dynamic;
|
||||
StaticFPRegisters = x64::SRAFPR;
|
||||
GeneralFPRegisters = x64::RAFPR;
|
||||
PairRegisters = x64::RAPairs;
|
||||
#ifdef _M_ARM_64EC
|
||||
ConfiguredDynamicRegisterBase = std::span(x64::RA.begin(), 7);
|
||||
#endif
|
||||
} else {
|
||||
ConfiguredDynamicRegisterBase = std::span(x32::RA.begin() + 6, 8);
|
||||
PairRegisters = x32::RAPairs;
|
||||
|
||||
StaticRegisters = x32::SRA;
|
||||
GeneralRegisters = x32::RA;
|
||||
GeneralRegistersNotPreserved = x32::NotPreserved_Dynamic;
|
||||
|
||||
StaticFPRegisters = x32::SRAFPR;
|
||||
GeneralFPRegisters = x32::RAFPR;
|
||||
@@ -676,10 +705,12 @@ void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs, uint3
|
||||
|
||||
// Now handle PF/AF
|
||||
if (PFAFSpillMask) {
|
||||
auto PFOffset = offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw);
|
||||
[[maybe_unused]] auto AFOffset = offsetof(FEXCore::Core::CpuStateFrame, State.af_raw);
|
||||
LOGMAN_THROW_A_FMT(PFAFSpillMask == PFAFMask, "PF/AF not spilled together");
|
||||
LOGMAN_THROW_A_FMT(AFOffset == PFOffset + 4, "PF/AF are together");
|
||||
|
||||
str(REG_PF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw));
|
||||
str(REG_AF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.af_raw));
|
||||
stp<ARMEmitter::IndexType::OFFSET>(REG_PF.W(), REG_AF.W(), STATE.R(), PFOffset);
|
||||
}
|
||||
|
||||
if (FPRs) {
|
||||
@@ -825,8 +856,7 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
if (PFAFFillMask) {
|
||||
LOGMAN_THROW_A_FMT(PFAFFillMask == PFAFMask, "PF/AF not filled together");
|
||||
|
||||
ldr(REG_PF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw));
|
||||
ldr(REG_AF.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.af_raw));
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(REG_PF.W(), REG_AF.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -930,9 +960,9 @@ void Arm64Emitter::PopGeneralRegisters(std::span<const ARMEmitter::Register> Reg
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::PushDynamicRegsAndLR(ARMEmitter::Register TmpReg) {
|
||||
void Arm64Emitter::PushDynamicRegs(ARMEmitter::Register TmpReg) {
|
||||
const auto CanUseSVE256 = EmitterCTX->HostFeatures.SupportsSVE256;
|
||||
const auto GPRSize = (ConfiguredDynamicRegisterBase.size() + 1) * Core::CPUState::GPR_REG_SIZE;
|
||||
const auto GPRSize = GeneralRegistersNotPreserved.size() * Core::CPUState::GPR_REG_SIZE;
|
||||
const auto FPRRegSize = CanUseSVE256 ? 32 : 16;
|
||||
const auto FPRSize = GeneralFPRegisters.size() * FPRRegSize;
|
||||
const uint64_t SPOffset = AlignUp(GPRSize + FPRSize, 16);
|
||||
@@ -948,25 +978,17 @@ void Arm64Emitter::PushDynamicRegsAndLR(ARMEmitter::Register TmpReg) {
|
||||
PushVectorRegisters(TmpReg, CanUseSVE256, GeneralFPRegisters);
|
||||
|
||||
// Push the general registers.
|
||||
PushGeneralRegisters(TmpReg, ConfiguredDynamicRegisterBase);
|
||||
|
||||
#ifndef _M_ARM_64EC
|
||||
str(ARMEmitter::XReg::lr, TmpReg, 0);
|
||||
#endif
|
||||
PushGeneralRegisters(TmpReg, GeneralRegistersNotPreserved);
|
||||
}
|
||||
|
||||
void Arm64Emitter::PopDynamicRegsAndLR() {
|
||||
void Arm64Emitter::PopDynamicRegs() {
|
||||
const auto CanUseSVE256 = EmitterCTX->HostFeatures.SupportsSVE256;
|
||||
|
||||
// Pop vectors first
|
||||
PopVectorRegisters(CanUseSVE256, GeneralFPRegisters);
|
||||
|
||||
// Pop GPRs second
|
||||
PopGeneralRegisters(ConfiguredDynamicRegisterBase);
|
||||
|
||||
#ifndef _M_ARM_64EC
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
#endif
|
||||
PopGeneralRegisters(GeneralRegistersNotPreserved);
|
||||
}
|
||||
|
||||
void Arm64Emitter::SpillForPreserveAllABICall(ARMEmitter::Register TmpReg, bool FPRs) {
|
||||
|
||||
@@ -104,9 +104,9 @@ protected:
|
||||
|
||||
FEXCore::Context::ContextImpl* EmitterCTX;
|
||||
|
||||
std::span<const ARMEmitter::Register> ConfiguredDynamicRegisterBase {};
|
||||
std::span<const ARMEmitter::Register> StaticRegisters {};
|
||||
std::span<const ARMEmitter::Register> GeneralRegisters {};
|
||||
std::span<const ARMEmitter::Register> GeneralRegistersNotPreserved {};
|
||||
std::span<const ARMEmitter::VRegister> StaticFPRegisters {};
|
||||
std::span<const ARMEmitter::VRegister> GeneralFPRegisters {};
|
||||
uint32_t PairRegisters = 0;
|
||||
@@ -141,8 +141,8 @@ protected:
|
||||
void PopVectorRegisters(bool SVERegs, std::span<const ARMEmitter::VRegister> VRegs);
|
||||
void PopGeneralRegisters(std::span<const ARMEmitter::Register> Regs);
|
||||
|
||||
void PushDynamicRegsAndLR(ARMEmitter::Register TmpReg);
|
||||
void PopDynamicRegsAndLR();
|
||||
void PushDynamicRegs(ARMEmitter::Register TmpReg);
|
||||
void PopDynamicRegs();
|
||||
|
||||
void PushCalleeSavedRegisters();
|
||||
void PopCalleeSavedRegisters();
|
||||
@@ -150,12 +150,12 @@ protected:
|
||||
// Spills and fills SRA/Dynamic registers that are required for Arm64 `preserve_all` ABI.
|
||||
// This ABI changes most registers to be callee saved.
|
||||
// Caller Saved:
|
||||
// - X0-X8, X16-X18.
|
||||
// - X0-X8, X16-X18, X30.
|
||||
// - v0-v7
|
||||
// - For 256-bit SVE hosts: top 128-bits of v8-v31
|
||||
//
|
||||
// Callee Saved:
|
||||
// - X9-X15, X19-X31
|
||||
// - X9-X15, X19-X29, X31
|
||||
// - Low 128-bits of v8-v31
|
||||
void SpillForPreserveAllABICall(ARMEmitter::Register TmpReg, bool FPRs = true);
|
||||
void FillForPreserveAllABICall(bool FPRs = true);
|
||||
@@ -165,7 +165,7 @@ protected:
|
||||
SpillForPreserveAllABICall(TmpReg, FPRs);
|
||||
} else {
|
||||
SpillStaticRegs(TmpReg, FPRs);
|
||||
PushDynamicRegsAndLR(TmpReg);
|
||||
PushDynamicRegs(TmpReg);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -173,7 +173,7 @@ protected:
|
||||
if (SupportsPreserveAllABI) {
|
||||
FillForPreserveAllABICall(FPRs);
|
||||
} else {
|
||||
PopDynamicRegsAndLR();
|
||||
PopDynamicRegs();
|
||||
FillStaticRegs(FPRs);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -48,6 +48,10 @@ namespace CPU {
|
||||
{0x8000'0000'8000'0000ULL, 0x8000'0000'8000'0000ULL}, // NAMED_VECTOR_CVTMAX_I32
|
||||
{0x8000'0000'0000'0000ULL, 0x8000'0000'0000'0000ULL}, // NAMED_VECTOR_CVTMAX_I64
|
||||
{0x0000'0000'0000'0000ULL, 0x0000'0000'0000'8000ULL}, // NAMED_VECTOR_F80_SIGN_MASK
|
||||
{0x5A82'7999'5A82'7999ULL, 0x5A82'7999'5A82'7999ULL}, // NAMED_VECTOR_SHA1RNDS_K0
|
||||
{0x6ED9'EBA1'6ED9'EBA1ULL, 0x6ED9'EBA1'6ED9'EBA1ULL}, // NAMED_VECTOR_SHA1RNDS_K1
|
||||
{0x8F1B'BCDC'8F1B'BCDCULL, 0x8F1B'BCDC'8F1B'BCDCULL}, // NAMED_VECTOR_SHA1RNDS_K2
|
||||
{0xCA62'C1D6'CA62'C1D6ULL, 0xCA62'C1D6'CA62'C1D6ULL}, // NAMED_VECTOR_SHA1RNDS_K3
|
||||
};
|
||||
|
||||
constexpr static auto PSHUFLW_LUT {[]() consteval {
|
||||
@@ -294,7 +298,7 @@ namespace CPU {
|
||||
// Fill in telemetry values
|
||||
for (size_t i = 0; i < FEXCore::Telemetry::TYPE_LAST; ++i) {
|
||||
auto& Telem = FEXCore::Telemetry::GetTelemetryValue(static_cast<FEXCore::Telemetry::TelemetryType>(i));
|
||||
Common.TelemetryValueAddresses[i] = reinterpret_cast<uint64_t>(Telem.GetAddr());
|
||||
Common.TelemetryValueAddresses[i] = reinterpret_cast<uint64_t>(&Telem);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
@@ -87,9 +87,6 @@ namespace CPU {
|
||||
// The length of the guest code for this block.
|
||||
size_t GuestSize;
|
||||
|
||||
// If this block represents a single guest instruction.
|
||||
bool SingleInst;
|
||||
|
||||
// Number of RIP entries for this JIT Code section.
|
||||
uint32_t NumberOfRIPEntries;
|
||||
|
||||
@@ -99,7 +96,10 @@ namespace CPU {
|
||||
// Shared-code modification spin-loop futex.
|
||||
uint32_t SpinLockFutex;
|
||||
|
||||
uint32_t _Pad;
|
||||
// If this block represents a single guest instruction.
|
||||
bool SingleInst;
|
||||
|
||||
uint8_t _Pad[3];
|
||||
};
|
||||
|
||||
/**
|
||||
|
||||
@@ -395,25 +395,7 @@ void CPUIDEmu::SetupFeatures() {
|
||||
XCR0 |= XCR0_AVX;
|
||||
}
|
||||
|
||||
// Override features if the user has specifically called for it.
|
||||
FEX_CONFIG_OPT(CPUIDFeatures, CPUID);
|
||||
if (!CPUIDFeatures()) {
|
||||
// Early exit if no features are overriden.
|
||||
return;
|
||||
}
|
||||
|
||||
#define ENABLE_DISABLE_OPTION(FeatureName, name, enum_name) \
|
||||
do { \
|
||||
const bool Disable##name = (CPUIDFeatures() & FEXCore::Config::CPUID::DISABLE##enum_name) != 0; \
|
||||
const bool Enable##name = (CPUIDFeatures() & FEXCore::Config::CPUID::ENABLE##enum_name) != 0; \
|
||||
LogMan::Throw::AFmt(!(Disable##name && Enable##name), "Disabling and Enabling CPU feature (" #name ") is mutually exclusive"); \
|
||||
const bool AlreadyEnabled = Features.FeatureName; \
|
||||
const bool Result = (AlreadyEnabled | Enable##name) & !Disable##name; \
|
||||
Features.FeatureName = Result; \
|
||||
} while (0)
|
||||
|
||||
ENABLE_DISABLE_OPTION(SHA, SHA, SHA);
|
||||
#undef ENABLE_DISABLE_OPTION
|
||||
Features.SHA = CTX->HostFeatures.SupportsSHA;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0h(uint32_t Leaf) const {
|
||||
|
||||
@@ -380,7 +380,7 @@ bool ContextImpl::InitCore() {
|
||||
SignalDelegation->SetConfig(SignalConfig);
|
||||
|
||||
#ifndef _WIN32
|
||||
#elif !defined(_M_ARM64EC)
|
||||
#elif !defined(_M_ARM_64EC)
|
||||
// WOW64 always needs the interrupt fault check to be enabled.
|
||||
Config.NeedsPendingInterruptFaultCheck = true;
|
||||
#endif
|
||||
@@ -473,6 +473,7 @@ void ContextImpl::DestroyThread(FEXCore::Core::InternalThreadState* Thread) {
|
||||
void ContextImpl::UnlockAfterFork(FEXCore::Core::InternalThreadState* LiveThread, bool Child) {
|
||||
Allocator::UnlockAfterFork(LiveThread, Child);
|
||||
|
||||
Profiler::PostForkAction(Child);
|
||||
if (Child) {
|
||||
CodeInvalidationMutex.StealAndDropActiveLocks();
|
||||
if (Config.StrictInProcessSplitLocks) {
|
||||
@@ -496,10 +497,6 @@ void ContextImpl::LockBeforeFork(FEXCore::Core::InternalThreadState* Thread) {
|
||||
}
|
||||
#endif
|
||||
|
||||
void ContextImpl::AddBlockMapping(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, void* Ptr) {
|
||||
Thread->LookupCache->AddBlockMapping(Address, Ptr);
|
||||
}
|
||||
|
||||
void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread) {
|
||||
FEXCORE_PROFILE_INSTANT("ClearCodeCache");
|
||||
|
||||
@@ -568,6 +565,16 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
for (size_t j = 0; j < CodeBlocks->size(); ++j) {
|
||||
const FEXCore::Frontend::Decoder::DecodedBlocks& Block = CodeBlocks->at(j);
|
||||
|
||||
bool BlockInForceTSOValidRange = false;
|
||||
auto InstForceTSOIt = ForceTSOInstructions.end();
|
||||
if (ForceTSOValidRanges.Contains({Block.Entry, Block.Entry + Block.Size})) {
|
||||
if (auto It = ForceTSOInstructions.lower_bound(Block.Entry); *It < Block.Entry + Block.Size) {
|
||||
InstForceTSOIt = It;
|
||||
BlockInForceTSOValidRange = true;
|
||||
}
|
||||
}
|
||||
|
||||
// Set the block entry point
|
||||
Thread->OpDispatcher->SetNewBlockIfChanged(Block.Entry);
|
||||
|
||||
@@ -584,6 +591,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < InstsInBlock; ++i) {
|
||||
uint64_t InstAddress = Block.Entry + BlockInstructionsLength;
|
||||
const FEXCore::X86Tables::X86InstInfo* TableInfo {nullptr};
|
||||
const FEXCore::X86Tables::DecodedInst* DecodedInfo {nullptr};
|
||||
|
||||
@@ -606,7 +614,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
Thread->OpDispatcher->FlushRegisterCache(true);
|
||||
|
||||
if (ExtendedDebugInfo || Thread->OpDispatcher->CanHaveSideEffects(TableInfo, DecodedInfo)) {
|
||||
Thread->OpDispatcher->_GuestOpcode(Block.Entry + BlockInstructionsLength - GuestRIP);
|
||||
Thread->OpDispatcher->_GuestOpcode(InstAddress - GuestRIP);
|
||||
}
|
||||
|
||||
if (Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL) {
|
||||
@@ -623,7 +631,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
Thread->OpDispatcher->SetCurrentCodeBlock(CodeWasChangedBlock);
|
||||
Thread->OpDispatcher->_ThreadRemoveCodeEntry();
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, Block.Entry + BlockInstructionsLength - GuestRIP));
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, InstAddress - GuestRIP));
|
||||
|
||||
auto NextOpBlock = Thread->OpDispatcher->CreateNewCodeBlockAfter(CurrentBlock);
|
||||
|
||||
@@ -635,17 +643,27 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
auto Fn = TableInfo->OpcodeDispatcher;
|
||||
Thread->OpDispatcher->ResetHandledLock();
|
||||
Thread->OpDispatcher->ResetDecodeFailure();
|
||||
IR::ForceTSOMode ForceTSO =
|
||||
BlockInForceTSOValidRange ?
|
||||
(InstForceTSOIt != ForceTSOInstructions.end() && *InstForceTSOIt == InstAddress ? IR::ForceTSOMode::ForceEnabled :
|
||||
IR::ForceTSOMode::ForceDisabled) :
|
||||
IR::ForceTSOMode::NoOverride;
|
||||
Thread->OpDispatcher->SetForceTSO(ForceTSO);
|
||||
std::invoke(Fn, Thread->OpDispatcher, DecodedInfo);
|
||||
if (Thread->OpDispatcher->HadDecodeFailure()) {
|
||||
HadDispatchError = true;
|
||||
} else {
|
||||
if (Thread->OpDispatcher->HasHandledLock() != IsLocked) {
|
||||
HadDispatchError = true;
|
||||
LogMan::Msg::EFmt("Missing LOCK HANDLER at 0x{:x}{{'{}'}}", Block.Entry + BlockInstructionsLength, TableInfo->Name ?: "UND");
|
||||
LogMan::Msg::EFmt("Missing LOCK HANDLER at 0x{:x}{{'{}'}}", InstAddress, TableInfo->Name ?: "UND");
|
||||
}
|
||||
BlockInstructionsLength += DecodedInfo->InstSize;
|
||||
TotalInstructionsLength += DecodedInfo->InstSize;
|
||||
++TotalInstructions;
|
||||
|
||||
// Walk InstForceTSOIt forward past the handled instruction
|
||||
InstForceTSOIt =
|
||||
std::find_if(InstForceTSOIt, ForceTSOInstructions.end(), [&](auto Val) { return Val >= Block.Entry + BlockInstructionsLength; });
|
||||
}
|
||||
} else {
|
||||
// Invalid instruction
|
||||
@@ -773,8 +791,9 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
}
|
||||
|
||||
uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP, uint64_t MaxInst) {
|
||||
FEXCORE_PROFILE_SCOPED("CompileBlock");
|
||||
auto Thread = Frame->Thread;
|
||||
FEXCORE_PROFILE_SCOPED("CompileBlock");
|
||||
FEXCORE_PROFILE_ACCUMULATION(Thread, AccumulatedJITTime);
|
||||
|
||||
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
|
||||
auto lk = GuardSignalDeferringSection<std::shared_lock>(CodeInvalidationMutex, Thread);
|
||||
@@ -843,7 +862,7 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
|
||||
// Insert to lookup cache
|
||||
// Pages containing this block are added via AddBlockExecutableRange before each page gets accessed in the frontend
|
||||
AddBlockMapping(Thread, GuestRIP, CodePtr);
|
||||
Thread->LookupCache->AddBlockMapping(GuestRIP, CodePtr);
|
||||
|
||||
return (uintptr_t)CodePtr;
|
||||
}
|
||||
@@ -907,19 +926,10 @@ void ContextImpl::MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) {
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::ThreadAddBlockLink(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestDestination,
|
||||
FEXCore::Context::ExitFunctionLinkData* HostLink, const FEXCore::Context::BlockDelinkerFunc& delinker) {
|
||||
auto lk = GuardSignalDeferringSection<std::shared_lock>(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
Thread->LookupCache->AddBlockLink(GuestDestination, HostLink, delinker);
|
||||
}
|
||||
|
||||
void ContextImpl::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) {
|
||||
LogMan::Throw::AFmt(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to "
|
||||
"be unique_locked here");
|
||||
|
||||
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
|
||||
|
||||
Thread->LookupCache->Erase(Thread->CurrentFrame, GuestRIP);
|
||||
}
|
||||
|
||||
@@ -982,6 +992,19 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::AddForceTSOInformation(const IntervalList<uint64_t>& ValidRanges, fextl::set<uint64_t>&& Instructions) {
|
||||
LogMan::Throw::AFmt(CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to be unique_locked here");
|
||||
ForceTSOValidRanges.Insert(ValidRanges);
|
||||
ForceTSOInstructions.merge(Instructions);
|
||||
}
|
||||
|
||||
void ContextImpl::RemoveForceTSOInformation(uint64_t Address, uint64_t Size) {
|
||||
LogMan::Throw::AFmt(CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to be unique_locked here");
|
||||
|
||||
ForceTSOValidRanges.Remove({Address, Address + Size});
|
||||
ForceTSOInstructions.erase(ForceTSOInstructions.lower_bound(Address), ForceTSOInstructions.upper_bound(Address + Size));
|
||||
}
|
||||
|
||||
void ContextImpl::RemoveCustomIREntrypoint(uintptr_t Entrypoint) {
|
||||
LOGMAN_THROW_A_FMT(Config.Is64BitMode || !(Entrypoint >> 32), "64-bit Entrypoint in 32-bit mode {:x}", Entrypoint);
|
||||
|
||||
|
||||
@@ -505,7 +505,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
auto EmitLongALUOpHandler = [&](auto R, auto Offset) {
|
||||
auto Address = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR(TMP4);
|
||||
PushDynamicRegs(TMP4);
|
||||
SpillStaticRegs(TMP4);
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
@@ -529,7 +529,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
FillStaticRegs();
|
||||
|
||||
// Fix the stack and any values that were stepped on
|
||||
PopDynamicRegsAndLR();
|
||||
PopDynamicRegs();
|
||||
|
||||
// Go back to our code block
|
||||
ret();
|
||||
|
||||
@@ -471,7 +471,9 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
|
||||
size_t CurrentSrc = 0;
|
||||
|
||||
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_1ST_SRC) != 0) {
|
||||
const auto VEXOperand = Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_SRC_MASK;
|
||||
|
||||
if (VEXOperand == FEXCore::X86Tables::InstFlags::FLAGS_VEX_1ST_SRC) {
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
|
||||
|
||||
@@ -496,7 +498,7 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
++CurrentSrc;
|
||||
}
|
||||
|
||||
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_2ND_SRC) != 0) {
|
||||
if (VEXOperand == FEXCore::X86Tables::InstFlags::FLAGS_VEX_2ND_SRC) {
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::GPR;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.HighBits = false;
|
||||
DecodeInst->Src[CurrentSrc].Data.GPR.GPR = MapVEXToReg(Options.vvvv, HasXMMSrc);
|
||||
@@ -515,7 +517,7 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
++CurrentSrc;
|
||||
}
|
||||
|
||||
if ((Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_DST) != 0) {
|
||||
if (VEXOperand == FEXCore::X86Tables::InstFlags::FLAGS_VEX_DST) {
|
||||
CurrentDest->Type = DecodedOperand::OpType::GPR;
|
||||
CurrentDest->Data.GPR.HighBits = false;
|
||||
CurrentDest->Data.GPR.GPR = MapVEXToReg(Options.vvvv, HasXMMDst);
|
||||
@@ -607,7 +609,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
uint16_t LocalOp = OPD(Info->Type, PrefixType, ModRM.reg);
|
||||
FEXCore::X86Tables::X86InstInfo* LocalInfo = &SecondInstGroupOps[LocalOp];
|
||||
#undef OPD
|
||||
if (LocalInfo->Type == FEXCore::X86Tables::TYPE_SECOND_GROUP_MODRM) {
|
||||
if (LocalInfo->Type == FEXCore::X86Tables::TYPE_SECOND_GROUP_MODRM && ModRM.mod == 0b11) {
|
||||
// Everything in this group is privileged instructions aside from XGETBV
|
||||
constexpr std::array<uint8_t, 8> RegToField = {
|
||||
255, 0, 1, 2, 255, 255, 255, 3,
|
||||
@@ -690,7 +692,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
return NormalOp(LocalInfo, Op, options);
|
||||
}
|
||||
} else if (Info->Type == FEXCore::X86Tables::TYPE_GROUP_EVEX) {
|
||||
FEXCORE_TELEMETRY_SET(EVEXOpTelem, 1);
|
||||
FEXCORE_TELEMETRY_SET(TYPE_USES_EVEX_OPS, 1);
|
||||
// EVEX unsupported
|
||||
return false;
|
||||
}
|
||||
@@ -912,12 +914,6 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (DecodeInst->Dest.IsGPR()) {
|
||||
LOGMAN_THROW_A_FMT(DecodeInst->Dest.Data.GPR.GPR != FEXCore::X86State::REG_INVALID, "Destination GPR was invalid");
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
void Decoder::BranchTargetInMultiblockRange() {
|
||||
|
||||
@@ -90,10 +90,10 @@ private:
|
||||
Utils::FixedSizePooledAllocation<FEXCore::X86Tables::DecodedInst*, 5000, 500> PoolObject;
|
||||
size_t DecodedSize {};
|
||||
|
||||
const uint8_t* InstStream;
|
||||
const uint8_t* InstStream {};
|
||||
|
||||
static constexpr size_t MAX_INST_SIZE = 15;
|
||||
uint8_t InstructionSize;
|
||||
uint8_t InstructionSize {};
|
||||
std::array<uint8_t, MAX_INST_SIZE> Instruction;
|
||||
FEXCore::X86Tables::DecodedInst* DecodeInst;
|
||||
|
||||
@@ -124,7 +124,5 @@ private:
|
||||
};
|
||||
|
||||
const uint8_t* AdjustAddrForSpecialRegion(const uint8_t* _InstStream, uint64_t EntryPoint, uint64_t RIP);
|
||||
|
||||
FEXCORE_TELEMETRY_INIT(EVEXOpTelem, TYPE_USES_EVEX_OPS);
|
||||
};
|
||||
} // namespace FEXCore::Frontend
|
||||
@@ -1,10 +1,13 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include "Common/SoftFloat.h"
|
||||
#include "Common/SoftFloat-3e/softfloat.h"
|
||||
|
||||
#include "Interface/Core/Interpreter/Fallbacks/FallbackOpHandler.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static softfloat_state SoftFloatStateFromFCW(uint16_t FCW, bool Force80BitPrecision = false) {
|
||||
softfloat_state State {};
|
||||
@@ -35,12 +38,14 @@ FEXCORE_PRESERVE_ALL_ATTR static softfloat_state SoftFloatStateFromFCW(uint16_t
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTTO> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle4(uint16_t FCW, float src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle4(uint16_t FCW, float src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat(&State, src);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle8(uint16_t FCW, double src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle8(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat(&State, src);
|
||||
}
|
||||
@@ -48,7 +53,8 @@ struct OpHandlers<IR::OP_F80CVTTO> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CMP> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static uint64_t handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static uint64_t handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
|
||||
bool eq, lt, nan;
|
||||
@@ -70,37 +76,43 @@ struct OpHandlers<IR::OP_F80CMP> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVT> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static float handle4(uint16_t FCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static float handle4(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return src.ToF32(&State);
|
||||
return X80SoftFloat(src).ToF32(&State);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle8(uint16_t FCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle8(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return src.ToF64(&State);
|
||||
return X80SoftFloat(src).ToF64(&State);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTINT> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int16_t handle2(uint16_t FCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int16_t handle2(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return src.ToI16(&State);
|
||||
return X80SoftFloat(src).ToI16(&State);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t handle4(uint16_t FCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t handle4(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return src.ToI32(&State);
|
||||
return X80SoftFloat(src).ToI32(&State);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int64_t handle8(uint16_t FCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int64_t handle8(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return src.ToI64(&State);
|
||||
return X80SoftFloat(src).ToI64(&State);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int16_t handle2t(uint16_t FCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int16_t handle2t(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
auto rv = extF80_to_i32(&State, src, softfloat_round_minMag, false);
|
||||
auto rv = extF80_to_i32(&State, X80SoftFloat(src), softfloat_round_minMag, false);
|
||||
|
||||
if (rv > INT16_MAX || rv < INT16_MIN) {
|
||||
///< Indefinite value for 16-bit conversions.
|
||||
@@ -110,31 +122,36 @@ struct OpHandlers<IR::OP_F80CVTINT> {
|
||||
}
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t handle4t(uint16_t FCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t handle4t(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return extF80_to_i32(&State, src, softfloat_round_minMag, false);
|
||||
return extF80_to_i32(&State, X80SoftFloat(src), softfloat_round_minMag, false);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int64_t handle8t(uint16_t FCW, X80SoftFloat src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int64_t handle8t(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return extF80_to_i64(&State, src, softfloat_round_minMag, false);
|
||||
return extF80_to_i64(&State, X80SoftFloat(src), softfloat_round_minMag, false);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTTOINT> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle2(uint16_t FCW, int16_t src) {
|
||||
return src;
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle2(uint16_t FCW, int16_t src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return X80SoftFloat(src);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle4(uint16_t FCW, int32_t src) {
|
||||
return src;
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle4(uint16_t FCW, int32_t src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return X80SoftFloat(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ROUND> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FRNDINT(&State, Src1);
|
||||
}
|
||||
@@ -142,7 +159,8 @@ struct OpHandlers<IR::OP_F80ROUND> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80F2XM1> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::F2XM1(&State, Src1);
|
||||
}
|
||||
@@ -150,7 +168,8 @@ struct OpHandlers<IR::OP_F80F2XM1> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80TAN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FTAN(&State, Src1);
|
||||
}
|
||||
@@ -158,7 +177,8 @@ struct OpHandlers<IR::OP_F80TAN> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SQRT> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FSQRT(&State, Src1);
|
||||
}
|
||||
@@ -166,7 +186,8 @@ struct OpHandlers<IR::OP_F80SQRT> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SIN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FSIN(&State, Src1);
|
||||
}
|
||||
@@ -174,7 +195,8 @@ struct OpHandlers<IR::OP_F80SIN> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80COS> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FCOS(&State, Src1);
|
||||
}
|
||||
@@ -182,21 +204,24 @@ struct OpHandlers<IR::OP_F80COS> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80XTRACT_EXP> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return X80SoftFloat::FXTRACT_EXP(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80XTRACT_SIG> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return X80SoftFloat::FXTRACT_SIG(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ADD> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FADD(&State, Src1, Src2);
|
||||
}
|
||||
@@ -204,7 +229,8 @@ struct OpHandlers<IR::OP_F80ADD> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SUB> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FSUB(&State, Src1, Src2);
|
||||
}
|
||||
@@ -212,7 +238,8 @@ struct OpHandlers<IR::OP_F80SUB> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80MUL> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FMUL(&State, Src1, Src2);
|
||||
}
|
||||
@@ -220,7 +247,8 @@ struct OpHandlers<IR::OP_F80MUL> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80DIV> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
return X80SoftFloat::FDIV(&State, Src1, Src2);
|
||||
}
|
||||
@@ -228,7 +256,8 @@ struct OpHandlers<IR::OP_F80DIV> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FYL2X> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FYL2X(&State, Src1, Src2);
|
||||
}
|
||||
@@ -236,7 +265,8 @@ struct OpHandlers<IR::OP_F80FYL2X> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ATAN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FATAN(&State, Src1, Src2);
|
||||
}
|
||||
@@ -244,7 +274,8 @@ struct OpHandlers<IR::OP_F80ATAN> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FPREM1> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FREM1(&State, Src1, Src2);
|
||||
}
|
||||
@@ -252,7 +283,8 @@ struct OpHandlers<IR::OP_F80FPREM1> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FPREM> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FREM(&State, Src1, Src2);
|
||||
}
|
||||
@@ -260,7 +292,8 @@ struct OpHandlers<IR::OP_F80FPREM> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SCALE> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return X80SoftFloat::FSCALE(&State, Src1, Src2);
|
||||
}
|
||||
@@ -268,63 +301,72 @@ struct OpHandlers<IR::OP_F80SCALE> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SIN> {
|
||||
static double handle(uint16_t FCW, double src) {
|
||||
static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return sin(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64COS> {
|
||||
static double handle(uint16_t FCW, double src) {
|
||||
static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return cos(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64TAN> {
|
||||
static double handle(uint16_t FCW, double src) {
|
||||
static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return tan(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64F2XM1> {
|
||||
static double handle(uint16_t FCW, double src) {
|
||||
static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return exp2(src) - 1.0;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64ATAN> {
|
||||
static double handle(uint16_t FCW, double src1, double src2) {
|
||||
static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return atan2(src1, src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FPREM> {
|
||||
static double handle(uint16_t FCW, double src1, double src2) {
|
||||
static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return fmod(src1, src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FPREM1> {
|
||||
static double handle(uint16_t FCW, double src1, double src2) {
|
||||
static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return remainder(src1, src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FYL2X> {
|
||||
static double handle(uint16_t FCW, double src1, double src2) {
|
||||
static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return src2 * log2(src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SCALE> {
|
||||
static double handle(uint16_t FCW, double src1, double src2) {
|
||||
static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
if (src1 == 0.0) { // src1 might be +/- zero
|
||||
return src1; // this will return negative or positive zero if when appropriate
|
||||
}
|
||||
@@ -335,7 +377,9 @@ struct OpHandlers<IR::OP_F64SCALE> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80BCDSTORE> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src1) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1q, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
X80SoftFloat Src1 = Src1q;
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW);
|
||||
bool Negative = Src1.Sign;
|
||||
|
||||
@@ -376,7 +420,8 @@ struct OpHandlers<IR::OP_F80BCDSTORE> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80BCDLOAD> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat handle(uint16_t FCW, X80SoftFloat Src) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
uint8_t* Src1 = reinterpret_cast<uint8_t*>(&Src);
|
||||
uint64_t BCD {};
|
||||
// We walk through each uint8_t and pull out the BCD encoding
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
@@ -15,13 +16,14 @@ static FallbackInfo GetFallbackInfo(R (*fn)(Args...), FEXCore::Core::FallbackHan
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(double (*fn)(uint16_t, double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_I16_F64, (void*)fn, HandlerIndex, false};
|
||||
FallbackInfo GetFallbackInfo(double (*fn)(uint16_t, double, FEXCore::Core::CpuStateFrame*), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_I16_F64_PTR, (void*)fn, HandlerIndex, false};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(double (*fn)(uint16_t, double, double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_I16_F64_F64, (void*)fn, HandlerIndex, false};
|
||||
FallbackInfo
|
||||
GetFallbackInfo(double (*fn)(uint16_t, double, double, FEXCore::Core::CpuStateFrame*), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_I16_F64_F64_PTR, (void*)fn, HandlerIndex, false};
|
||||
}
|
||||
|
||||
void InterpreterOps::FillFallbackIndexPointers(uint64_t* Info) {
|
||||
@@ -86,11 +88,11 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case IR::OpSize::i32Bit: {
|
||||
*Info = {FABI_F80_I16_F32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4, SupportsPreserveAllABI};
|
||||
*Info = {FABI_F80_I16_F32_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
*Info = {FABI_F80_I16_F64, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8, SupportsPreserveAllABI};
|
||||
*Info = {FABI_F80_I16_F64_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
@@ -100,11 +102,11 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
case IR::OP_F80CVT: {
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i32Bit: {
|
||||
*Info = {FABI_F32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4, SupportsPreserveAllABI};
|
||||
*Info = {FABI_F32_I16_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
*Info = {FABI_F64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8, Core::OPINDEX_F80CVT_8, SupportsPreserveAllABI};
|
||||
*Info = {FABI_F64_I16_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8, Core::OPINDEX_F80CVT_8, SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
@@ -117,28 +119,31 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i16Bit: {
|
||||
if (Op->Truncate) {
|
||||
*Info = {FABI_I16_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t, Core::OPINDEX_F80CVTINT_TRUNC2,
|
||||
*Info = {FABI_I16_I16_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t, Core::OPINDEX_F80CVTINT_TRUNC2,
|
||||
SupportsPreserveAllABI};
|
||||
} else {
|
||||
*Info = {FABI_I16_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2, Core::OPINDEX_F80CVTINT_2, SupportsPreserveAllABI};
|
||||
*Info = {FABI_I16_I16_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2, Core::OPINDEX_F80CVTINT_2,
|
||||
SupportsPreserveAllABI};
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case IR::OpSize::i32Bit: {
|
||||
if (Op->Truncate) {
|
||||
*Info = {FABI_I32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t, Core::OPINDEX_F80CVTINT_TRUNC4,
|
||||
*Info = {FABI_I32_I16_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t, Core::OPINDEX_F80CVTINT_TRUNC4,
|
||||
SupportsPreserveAllABI};
|
||||
} else {
|
||||
*Info = {FABI_I32_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4, Core::OPINDEX_F80CVTINT_4, SupportsPreserveAllABI};
|
||||
*Info = {FABI_I32_I16_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4, Core::OPINDEX_F80CVTINT_4,
|
||||
SupportsPreserveAllABI};
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
if (Op->Truncate) {
|
||||
*Info = {FABI_I64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t, Core::OPINDEX_F80CVTINT_TRUNC8,
|
||||
*Info = {FABI_I64_I16_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t, Core::OPINDEX_F80CVTINT_TRUNC8,
|
||||
SupportsPreserveAllABI};
|
||||
} else {
|
||||
*Info = {FABI_I64_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8, Core::OPINDEX_F80CVTINT_8, SupportsPreserveAllABI};
|
||||
*Info = {FABI_I64_I16_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8, Core::OPINDEX_F80CVTINT_8,
|
||||
SupportsPreserveAllABI};
|
||||
}
|
||||
return true;
|
||||
}
|
||||
@@ -147,7 +152,7 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CMP: {
|
||||
*Info = {FABI_I64_I16_F80_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle,
|
||||
*Info = {FABI_I64_I16_F80_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle,
|
||||
(Core::FallbackHandlerIndex)(Core::OPINDEX_F80CMP), SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
@@ -157,11 +162,13 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case IR::OpSize::i16Bit: {
|
||||
*Info = {FABI_F80_I16_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2, Core::OPINDEX_F80CVTTOINT_2, SupportsPreserveAllABI};
|
||||
*Info = {FABI_F80_I16_I16_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2, Core::OPINDEX_F80CVTTOINT_2,
|
||||
SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
case IR::OpSize::i32Bit: {
|
||||
*Info = {FABI_F80_I16_I32, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4, Core::OPINDEX_F80CVTTOINT_4, SupportsPreserveAllABI};
|
||||
*Info = {FABI_F80_I16_I32_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4, Core::OPINDEX_F80CVTTOINT_4,
|
||||
SupportsPreserveAllABI};
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
@@ -169,16 +176,16 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
break;
|
||||
}
|
||||
|
||||
#define COMMON_UNARY_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = {FABI_F80_I16_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, SupportsPreserveAllABI}; \
|
||||
return true; \
|
||||
#define COMMON_UNARY_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = {FABI_F80_I16_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, SupportsPreserveAllABI}; \
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_BINARY_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = {FABI_F80_I16_F80_F80, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, SupportsPreserveAllABI}; \
|
||||
return true; \
|
||||
#define COMMON_BINARY_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = {FABI_F80_I16_F80_F80_PTR, (void*)&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle, Core::OPINDEX_F80##OP, SupportsPreserveAllABI}; \
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_F64_OP(OP) \
|
||||
@@ -229,7 +236,7 @@ bool InterpreterOps::GetFallbackHandler(bool SupportsPreserveAllABI, const IR::I
|
||||
SupportsPreserveAllABI};
|
||||
return true;
|
||||
case IR::OP_VPCMPISTRX:
|
||||
*Info = {FABI_I32_I128_I128_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_VPCMPISTRX>::handle, Core::OPINDEX_VPCMPISTRX, SupportsPreserveAllABI};
|
||||
*Info = {FABI_I32_V128_V128_I16, (void*)&FEXCore::CPU::OpHandlers<IR::OP_VPCMPISTRX>::handle, Core::OPINDEX_VPCMPISTRX, SupportsPreserveAllABI};
|
||||
return true;
|
||||
|
||||
default: break;
|
||||
|
||||
@@ -0,0 +1,91 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Core/Interpreter/Fallbacks/VectorFallbacks.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
#include <arm_neon.h>
|
||||
#endif
|
||||
|
||||
#include <cstring>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#ifdef _M_ARM_64
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t GetImplicitLength(FEXCore::VectorRegType data, uint16_t control) {
|
||||
const auto is_using_words = (control & 1) != 0;
|
||||
|
||||
if (is_using_words) {
|
||||
uint16x8_t a = vreinterpretq_u16_u8(data);
|
||||
uint16x8_t VIndexes {};
|
||||
const uint16x8_t VIndex16 = vdupq_n_u16(8);
|
||||
uint16_t Indexes[8] = {
|
||||
0, 1, 2, 3, 4, 5, 6, 7,
|
||||
};
|
||||
memcpy(&VIndexes, Indexes, sizeof(VIndexes));
|
||||
auto MaskResult = vceqzq_u16(a);
|
||||
auto SelectResult = vbslq_u16(MaskResult, VIndexes, VIndex16);
|
||||
return vminvq_u16(SelectResult);
|
||||
} else {
|
||||
uint8x16_t VIndexes {};
|
||||
const uint8x16_t VIndex16 = vdupq_n_u8(16);
|
||||
uint8_t Indexes[16] = {
|
||||
0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15,
|
||||
};
|
||||
memcpy(&VIndexes, Indexes, sizeof(VIndexes));
|
||||
auto MaskResult = vceqzq_u8(data);
|
||||
auto SelectResult = vbslq_u8(MaskResult, VIndexes, VIndex16);
|
||||
return vminvq_u8(SelectResult);
|
||||
}
|
||||
}
|
||||
#else
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t GetImplicitLength(FEXCore::VectorRegType data, uint16_t control) {
|
||||
const auto* data_u8 = reinterpret_cast<const uint8_t*>(&data);
|
||||
const auto is_using_words = (control & 1) != 0;
|
||||
|
||||
int32_t length = 0;
|
||||
|
||||
if (is_using_words) {
|
||||
const auto get_word = [data_u8](int32_t index) {
|
||||
const auto* src = data_u8 + (index * sizeof(uint16_t));
|
||||
|
||||
uint16_t element {};
|
||||
std::memcpy(&element, src, sizeof(uint16_t));
|
||||
return element;
|
||||
};
|
||||
|
||||
while (length < 8 && get_word(length) != 0) {
|
||||
length++;
|
||||
}
|
||||
} else {
|
||||
while (length < 16 && data_u8[length] != 0) {
|
||||
length++;
|
||||
}
|
||||
}
|
||||
|
||||
return length;
|
||||
}
|
||||
#endif
|
||||
|
||||
// Essentially the same in terms of behavior with VPCMPESTRX instructions,
|
||||
// with the only difference being that the length of the string is encoded
|
||||
// as part of the data vectors passed in.
|
||||
//
|
||||
// i.e. Length is determined by the presence of a NUL (all-zero) character
|
||||
// within the data.
|
||||
//
|
||||
// If no NUL character exists, then the length of the strings are assumed
|
||||
// to be the max length possible for the given character size specified
|
||||
// in the control flags (16 characters for 8-bit, and 8 characters for 16-bit).
|
||||
//
|
||||
FEXCORE_PRESERVE_ALL_ATTR uint32_t OpHandlers<IR::OP_VPCMPISTRX>::handle(FEXCore::VectorRegType lhs, FEXCore::VectorRegType rhs, uint16_t control) {
|
||||
// Subtract by 1 in order to make validity limits 0-based
|
||||
const auto valid_lhs = GetImplicitLength(lhs, control) - 1;
|
||||
const auto valid_rhs = GetImplicitLength(rhs, control) - 1;
|
||||
__uint128_t lhs_i;
|
||||
memcpy(&lhs_i, &lhs, sizeof(lhs_i));
|
||||
__uint128_t rhs_i;
|
||||
memcpy(&rhs_i, &rhs, sizeof(rhs_i));
|
||||
|
||||
return OpHandlers<IR::OP_VPCMPESTRX>::MainBody(lhs_i, valid_lhs, rhs_i, valid_rhs, control);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <algorithm>
|
||||
@@ -6,9 +7,9 @@
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
#include "Interface/Core/Interpreter/Fallbacks/FallbackOpHandler.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Common/VectorRegType.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
@@ -344,51 +345,7 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_VPCMPISTRX> {
|
||||
// Essentially the same in terms of behavior with VPCMPESTRX instructions,
|
||||
// with the only difference being that the length of the string is encoded
|
||||
// as part of the data vectors passed in.
|
||||
//
|
||||
// i.e. Length is determined by the presence of a NUL (all-zero) character
|
||||
// within the data.
|
||||
//
|
||||
// If no NUL character exists, then the length of the strings are assumed
|
||||
// to be the max length possible for the given character size specified
|
||||
// in the control flags (16 characters for 8-bit, and 8 characters for 16-bit).
|
||||
//
|
||||
FEXCORE_PRESERVE_ALL_ATTR static uint32_t handle(__uint128_t lhs, __uint128_t rhs, uint16_t control) {
|
||||
// Subtract by 1 in order to make validity limits 0-based
|
||||
const auto valid_lhs = GetImplicitLength(lhs, control) - 1;
|
||||
const auto valid_rhs = GetImplicitLength(rhs, control) - 1;
|
||||
|
||||
return OpHandlers<IR::OP_VPCMPESTRX>::MainBody(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static int32_t GetImplicitLength(const __uint128_t& data, uint16_t control) {
|
||||
const auto* data_u8 = reinterpret_cast<const uint8_t*>(&data);
|
||||
const auto is_using_words = (control & 1) != 0;
|
||||
|
||||
int32_t length = 0;
|
||||
|
||||
if (is_using_words) {
|
||||
const auto get_word = [data_u8](int32_t index) {
|
||||
const auto* src = data_u8 + (index * sizeof(uint16_t));
|
||||
|
||||
uint16_t element {};
|
||||
std::memcpy(&element, src, sizeof(uint16_t));
|
||||
return element;
|
||||
};
|
||||
|
||||
while (length < 8 && get_word(length) != 0) {
|
||||
length++;
|
||||
}
|
||||
} else {
|
||||
while (length < 16 && data_u8[length] != 0) {
|
||||
length++;
|
||||
}
|
||||
}
|
||||
|
||||
return length;
|
||||
}
|
||||
FEXCORE_PRESERVE_ALL_ATTR static uint32_t handle(VectorRegType lhs, VectorRegType rhs, uint16_t control);
|
||||
};
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -1,8 +1,6 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <array>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
@@ -16,22 +14,22 @@ struct IROp_Header;
|
||||
namespace FEXCore::CPU {
|
||||
enum FallbackABI {
|
||||
FABI_UNKNOWN,
|
||||
FABI_F80_I16_F32,
|
||||
FABI_F80_I16_F64,
|
||||
FABI_F80_I16_I16,
|
||||
FABI_F80_I16_I32,
|
||||
FABI_F32_I16_F80,
|
||||
FABI_F64_I16_F80,
|
||||
FABI_F64_I16_F64,
|
||||
FABI_F64_I16_F64_F64,
|
||||
FABI_I16_I16_F80,
|
||||
FABI_I32_I16_F80,
|
||||
FABI_I64_I16_F80,
|
||||
FABI_I64_I16_F80_F80,
|
||||
FABI_F80_I16_F80,
|
||||
FABI_F80_I16_F80_F80,
|
||||
FABI_F80_I16_F32_PTR,
|
||||
FABI_F80_I16_F64_PTR,
|
||||
FABI_F80_I16_I16_PTR,
|
||||
FABI_F80_I16_I32_PTR,
|
||||
FABI_F32_I16_F80_PTR,
|
||||
FABI_F64_I16_F80_PTR,
|
||||
FABI_F64_I16_F64_PTR,
|
||||
FABI_F64_I16_F64_F64_PTR,
|
||||
FABI_I16_I16_F80_PTR,
|
||||
FABI_I32_I16_F80_PTR,
|
||||
FABI_I64_I16_F80_PTR,
|
||||
FABI_I64_I16_F80_F80_PTR,
|
||||
FABI_F80_I16_F80_PTR,
|
||||
FABI_F80_I16_F80_F80_PTR,
|
||||
FABI_I32_I64_I64_I128_I128_I16,
|
||||
FABI_I32_I128_I128_I16,
|
||||
FABI_I32_V128_V128_I16,
|
||||
};
|
||||
|
||||
struct FallbackInfo {
|
||||
|
||||
@@ -1386,7 +1386,10 @@ DEF_OP(Bfi) {
|
||||
bfi(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
} else if (Dst != Src) {
|
||||
// If the destination isn't the source then we can move the DstSrc and insert directly.
|
||||
mov(EmitSize, Dst, SrcDst);
|
||||
//
|
||||
// The move is 64-bit to allow register renaming, the upper bits don't
|
||||
// matter because of the bfi's EmitSize.
|
||||
mov(ARMEmitter::Size::i64Bit, Dst, SrcDst);
|
||||
bfi(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
} else {
|
||||
// Destination didn't match the dst source register.
|
||||
@@ -1555,7 +1558,7 @@ DEF_OP(VExtractToGPR) {
|
||||
|
||||
const auto Offset = ElementSizeBits * Op->Index;
|
||||
[[maybe_unused]] const auto Is256Bit = Offset >= SSERegBitSize;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
|
||||
@@ -101,11 +101,16 @@ DEF_OP(CAS) {
|
||||
auto Expected = GetReg(Op->Expected.ID());
|
||||
auto Desired = GetReg(Op->Desired.ID());
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Dst = GetReg(Node);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mov(EmitSize, TMP2, Expected);
|
||||
casal(SubEmitSize, TMP2, Desired, MemSrc);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
if (Expected == Dst && Dst != MemSrc && Dst != Desired) {
|
||||
casal(SubEmitSize, Dst, Desired, MemSrc);
|
||||
} else {
|
||||
mov(EmitSize, TMP2, Expected);
|
||||
casal(SubEmitSize, TMP2, Desired, MemSrc);
|
||||
mov(EmitSize, Dst, TMP2.R());
|
||||
}
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
ARMEmitter::ForwardLabel LoopNotExpected;
|
||||
@@ -122,11 +127,11 @@ DEF_OP(CAS) {
|
||||
b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
|
||||
stlxr(SubEmitSize, TMP3, Desired, MemSrc);
|
||||
cbnz(EmitSize, TMP3, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), Expected);
|
||||
mov(EmitSize, Dst, Expected);
|
||||
b(&LoopExpected);
|
||||
|
||||
Bind(&LoopNotExpected);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
mov(EmitSize, Dst, TMP2.R());
|
||||
// exclusive monitor needs to be cleared here
|
||||
// Might have hit the case where ldaxr was hit but stlxr wasn't
|
||||
clrex();
|
||||
@@ -286,7 +291,6 @@ DEF_OP(AtomicSwap) {
|
||||
const auto SubEmitSize = OpSize == IR::OpSize::i64Bit ? ARMEmitter::SubRegSize::i64Bit :
|
||||
OpSize == IR::OpSize::i32Bit ? ARMEmitter::SubRegSize::i32Bit :
|
||||
OpSize == IR::OpSize::i16Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
OpSize == IR::OpSize::i8Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
|
||||
@@ -150,7 +150,7 @@ DEF_OP(Syscall) {
|
||||
// X2: Pointer to SyscallArguments
|
||||
|
||||
FEXCore::IR::SyscallFlags Flags = Op->Flags;
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
PushDynamicRegs(TMP1);
|
||||
|
||||
uint32_t GPRSpillMask = ~0U;
|
||||
uint32_t FPRSpillMask = ~0U;
|
||||
@@ -201,7 +201,7 @@ DEF_OP(Syscall) {
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
PopDynamicRegs();
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NORETURNEDRESULT) != FEXCore::IR::SyscallFlags::NORETURNEDRESULT) {
|
||||
// Move result to its destination register.
|
||||
@@ -314,7 +314,7 @@ DEF_OP(Thunk) {
|
||||
|
||||
SpillStaticRegs(TMP1); // spill to ctx before ra64 spill
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
PushDynamicRegs(TMP1);
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetReg(Op->ArgPtr.ID()));
|
||||
|
||||
@@ -326,7 +326,7 @@ DEF_OP(Thunk) {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
PopDynamicRegs();
|
||||
|
||||
FillStaticRegs(); // load from ctx after ra64 refill
|
||||
}
|
||||
@@ -378,7 +378,7 @@ DEF_OP(ValidateCode) {
|
||||
}
|
||||
|
||||
DEF_OP(ThreadRemoveCodeEntry) {
|
||||
PushDynamicRegsAndLR(TMP4);
|
||||
PushDynamicRegs(TMP4);
|
||||
SpillStaticRegs(TMP4);
|
||||
|
||||
// Arguments are passed as follows:
|
||||
@@ -397,7 +397,7 @@ DEF_OP(ThreadRemoveCodeEntry) {
|
||||
FillStaticRegs();
|
||||
|
||||
// Fix the stack and any values that were stepped on
|
||||
PopDynamicRegsAndLR();
|
||||
PopDynamicRegs();
|
||||
}
|
||||
|
||||
DEF_OP(CPUID) {
|
||||
@@ -406,7 +406,7 @@ DEF_OP(CPUID) {
|
||||
mov(ARMEmitter::Size::i64Bit, TMP2, GetReg(Op->Function.ID()));
|
||||
mov(ARMEmitter::Size::i64Bit, TMP3, GetReg(Op->Leaf.ID()));
|
||||
|
||||
PushDynamicRegsAndLR(TMP4);
|
||||
PushDynamicRegs(TMP4);
|
||||
SpillStaticRegs(TMP4);
|
||||
|
||||
// x0 = CPUID Handler
|
||||
@@ -433,7 +433,7 @@ DEF_OP(CPUID) {
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
PopDynamicRegs();
|
||||
|
||||
// Results are in x0, x1
|
||||
// Results want to be 4xi32 scalars
|
||||
@@ -446,7 +446,7 @@ DEF_OP(CPUID) {
|
||||
DEF_OP(XGetBV) {
|
||||
auto Op = IROp->C<IR::IROp_XGetBV>();
|
||||
|
||||
PushDynamicRegsAndLR(TMP4);
|
||||
PushDynamicRegs(TMP4);
|
||||
SpillStaticRegs(TMP4);
|
||||
|
||||
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, GetReg(Op->Function.ID()));
|
||||
@@ -467,7 +467,7 @@ DEF_OP(XGetBV) {
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
PopDynamicRegs();
|
||||
|
||||
// Results are in x0, need to split into i32 parts
|
||||
mov(ARMEmitter::Size::i32Bit, GetReg(Op->OutEAX.ID()), TMP1);
|
||||
|
||||
@@ -6,6 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
@@ -16,7 +17,7 @@ DEF_OP(VInsGPR) {
|
||||
const auto DestIdx = Op->DestIdx;
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp);
|
||||
const auto ElementsPer128Bit = IR::NumElements(IR::OpSize::i128Bit, ElementSize);
|
||||
@@ -104,6 +105,16 @@ DEF_OP(VCastFromGPR) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VLoadTwoGPRs) {
|
||||
const auto Op = IROp->C<IR::IROp_VLoadTwoGPRs>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto SrcLower = GetReg(Op->Lower.ID());
|
||||
const auto SrcUpper = GetReg(Op->Upper.ID());
|
||||
fmov(ARMEmitter::Size::i64Bit, Dst.D(), SrcLower);
|
||||
fmov(ARMEmitter::Size::i64Bit, Dst.D(), SrcUpper, true);
|
||||
}
|
||||
|
||||
DEF_OP(VDupFromGPR) {
|
||||
const auto Op = IROp->C<IR::IROp_VDupFromGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -112,7 +123,7 @@ DEF_OP(VDupFromGPR) {
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
@@ -206,7 +217,7 @@ DEF_OP(Vector_SToF) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -239,7 +250,7 @@ DEF_OP(Vector_FToZS) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -270,7 +281,7 @@ DEF_OP(Vector_FToS) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
|
||||
@@ -301,7 +312,7 @@ DEF_OP(Vector_FToF) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Conv = (IR::OpSizeToSize(ElementSize) << 8) | IR::OpSizeToSize(Op->SrcElementSize);
|
||||
|
||||
@@ -404,7 +415,7 @@ DEF_OP(Vector_FToI) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
@@ -459,13 +470,68 @@ DEF_OP(Vector_FToI) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToISized) {
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToISized>();
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = IROp->Size == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit, "256-bit not wired up, though we could change that");
|
||||
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFRINTTS, "Need FRINTTS for Vector_FToISized");
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
|
||||
if (ElementSize == IROp->Size) {
|
||||
// See above
|
||||
#define ROUNDING_FN(name) \
|
||||
if (ElementSize == IR::OpSize::i32Bit) { \
|
||||
name(Dst.S(), Vector.S()); \
|
||||
} else if (ElementSize == IR::OpSize::i64Bit) { \
|
||||
name(Dst.D(), Vector.D()); \
|
||||
} else { \
|
||||
FEX_UNREACHABLE; \
|
||||
}
|
||||
|
||||
if (Op->IntSize == IR::OpSize::i64Bit) {
|
||||
if (Op->HostRound) {
|
||||
ROUNDING_FN(frint64x);
|
||||
} else {
|
||||
ROUNDING_FN(frint64z);
|
||||
}
|
||||
} else {
|
||||
if (Op->HostRound) {
|
||||
ROUNDING_FN(frint32x);
|
||||
} else {
|
||||
ROUNDING_FN(frint32z);
|
||||
}
|
||||
}
|
||||
|
||||
#undef ROUNDING_FN
|
||||
} else {
|
||||
if (Op->IntSize == IR::OpSize::i64Bit) {
|
||||
if (Op->HostRound) {
|
||||
frint64x(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
} else {
|
||||
frint64z(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
}
|
||||
} else {
|
||||
if (Op->HostRound) {
|
||||
frint32x(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
} else {
|
||||
frint32z(SubEmitSize, Dst.Q(), Vector.Q());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_F64ToI32) {
|
||||
const auto Op = IROp->C<IR::IROp_Vector_F64ToI32>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Round = Op->Round;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
|
||||
@@ -169,6 +169,125 @@ DEF_OP(VSha1H) {
|
||||
sha1h(Dst.S(), Src.S());
|
||||
}
|
||||
|
||||
DEF_OP(VSha1C) {
|
||||
auto Op = IROp->C<IR::IROp_VSha1C>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src3 = GetVReg(Op->Src3.ID());
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha1c(Dst, Src2.S(), Src3);
|
||||
} else if (Dst != Src2 && Dst != Src3) {
|
||||
mov(Dst.Q(), Src1.Q());
|
||||
sha1c(Dst, Src2.S(), Src3);
|
||||
} else {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
sha1c(VTMP1, Src2.S(), Src3);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSha1M) {
|
||||
auto Op = IROp->C<IR::IROp_VSha1M>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src3 = GetVReg(Op->Src3.ID());
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha1m(Dst, Src2.S(), Src3);
|
||||
} else if (Dst != Src2 && Dst != Src3) {
|
||||
mov(Dst.Q(), Src1.Q());
|
||||
sha1m(Dst, Src2.S(), Src3);
|
||||
} else {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
sha1m(VTMP1, Src2.S(), Src3);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSha1P) {
|
||||
auto Op = IROp->C<IR::IROp_VSha1P>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src3 = GetVReg(Op->Src3.ID());
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha1p(Dst, Src2.S(), Src3);
|
||||
} else if (Dst != Src2 && Dst != Src3) {
|
||||
mov(Dst.Q(), Src1.Q());
|
||||
sha1p(Dst, Src2.S(), Src3);
|
||||
} else {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
sha1p(VTMP1, Src2.S(), Src3);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSha1SU1) {
|
||||
auto Op = IROp->C<IR::IROp_VSha1SU1>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha1su1(Dst, Src2);
|
||||
} else if (Dst != Src2) {
|
||||
mov(Dst.Q(), Src1.Q());
|
||||
sha1su1(Dst, Src2);
|
||||
} else {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
sha1su1(VTMP1, Src2);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSha256H) {
|
||||
auto Op = IROp->C<IR::IROp_VSha256H>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src3 = GetVReg(Op->Src3.ID());
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha256h(Dst, Src2, Src3);
|
||||
} else if (Dst != Src2 && Dst != Src3) {
|
||||
mov(Dst.Q(), Src1.Q());
|
||||
sha256h(Dst, Src2, Src3);
|
||||
} else {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
sha256h(VTMP1, Src2, Src3);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSha256H2) {
|
||||
auto Op = IROp->C<IR::IROp_VSha256H2>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src3 = GetVReg(Op->Src3.ID());
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha256h2(Dst, Src2, Src3);
|
||||
} else if (Dst != Src2 && Dst != Src3) {
|
||||
mov(Dst.Q(), Src1.Q());
|
||||
sha256h2(Dst, Src2, Src3);
|
||||
} else {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
sha256h2(VTMP1, Src2, Src3);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSha256U0) {
|
||||
auto Op = IROp->C<IR::IROp_VSha256U0>();
|
||||
|
||||
@@ -185,6 +304,23 @@ DEF_OP(VSha256U0) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSha256U1) {
|
||||
auto Op = IROp->C<IR::IROp_VSha256U1>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
|
||||
if (Dst != Src1 && Dst != Src2) {
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), 0);
|
||||
sha256su1(Dst, Src1, Src2);
|
||||
} else {
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), 0);
|
||||
sha256su1(VTMP1, Src1, Src2);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(PCLMUL) {
|
||||
const auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -11,6 +11,7 @@ desc: Main glue logic of the arm64 splatter backend
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Common/SoftFloat.h"
|
||||
#include "FEXCore/Utils/Telemetry.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
@@ -87,16 +88,13 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
} else {
|
||||
auto FillF80Result = [&]() {
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1, ARMEmitter::XReg::x0);
|
||||
mov(TMP2, ARMEmitter::XReg::x1);
|
||||
mov(VTMP1.Q(), ARMEmitter::VReg::v0.Q());
|
||||
}
|
||||
|
||||
FillForABICall(Info.SupportsPreserveAllABI, true);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
eor(Dst.Q(), Dst.Q(), Dst.Q());
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, TMP1);
|
||||
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 4, TMP2);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
};
|
||||
|
||||
auto FillF64Result = [&]() {
|
||||
@@ -120,52 +118,16 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
};
|
||||
|
||||
switch (Info.ABI) {
|
||||
case FABI_F80_I16_F32: {
|
||||
case FABI_F80_I16_F32_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
fmov(ARMEmitter::SReg::s0, Src1.S());
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, float>(ARMEmitter::Reg::r1);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
FillF80Result();
|
||||
} break;
|
||||
|
||||
case FABI_F80_I16_F64: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
mov(ARMEmitter::DReg::d0, Src1.D());
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, double>(ARMEmitter::Reg::r1);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
FillF80Result();
|
||||
} break;
|
||||
|
||||
case FABI_F80_I16_I16:
|
||||
case FABI_F80_I16_I32: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetReg(IROp->Args[0].ID());
|
||||
if (Info.ABI == FABI_F80_I16_I16) {
|
||||
sxth(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, Src1);
|
||||
} else {
|
||||
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, Src1);
|
||||
}
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, uint32_t>(ARMEmitter::Reg::r2);
|
||||
GenerateIndirectRuntimeCall<FEXCore::VectorRegType, uint16_t, float, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
@@ -173,20 +135,59 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
FillF80Result();
|
||||
} break;
|
||||
|
||||
case FABI_F32_I16_F80: {
|
||||
case FABI_F80_I16_F64_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
mov(ARMEmitter::DReg::d0, Src1.D());
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<FEXCore::VectorRegType, uint16_t, double, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
FillF80Result();
|
||||
} break;
|
||||
|
||||
case FABI_F80_I16_I16_PTR:
|
||||
case FABI_F80_I16_I32_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetReg(IROp->Args[0].ID());
|
||||
if (Info.ABI == FABI_F80_I16_I16_PTR) {
|
||||
sxth(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, Src1);
|
||||
} else {
|
||||
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, Src1);
|
||||
}
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
mov(ARMEmitter::XReg::x2, STATE);
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<FEXCore::VectorRegType, uint16_t, uint32_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
FillF80Result();
|
||||
} break;
|
||||
|
||||
case FABI_F32_I16_F80_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
mov(ARMEmitter::VReg::v0.Q(), Src1.Q());
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<float, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
GenerateIndirectRuntimeCall<float, uint16_t, FEXCore::VectorRegType, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
@@ -198,43 +199,45 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
fmov(Dst.S(), VTMP1.S());
|
||||
} break;
|
||||
|
||||
case FABI_F64_I16_F80: {
|
||||
case FABI_F64_I16_F80_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
mov(ARMEmitter::VReg::v0.Q(), Src1.Q());
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, FEXCore::VectorRegType, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
FillF64Result();
|
||||
} break;
|
||||
|
||||
case FABI_F64_I16_F64: {
|
||||
case FABI_F64_I16_F64_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
mov(ARMEmitter::DReg::d0, Src1.D());
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, double>(ARMEmitter::Reg::r1);
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, double, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
FillF64Result();
|
||||
} break;
|
||||
|
||||
case FABI_F64_I16_F64_F64: {
|
||||
case FABI_F64_I16_F64_F64_PTR: {
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src2 = GetVReg(IROp->Args[1].ID());
|
||||
|
||||
@@ -249,30 +252,31 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
}
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, double, double>(ARMEmitter::Reg::r1);
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, double, double, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
FillF64Result();
|
||||
} break;
|
||||
|
||||
case FABI_I16_I16_F80: {
|
||||
case FABI_I16_I16_F80_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
mov(ARMEmitter::VReg::v0.Q(), Src1.Q());
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint16_t, FEXCore::VectorRegType, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
@@ -283,38 +287,38 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
const auto Dst = GetReg(Node);
|
||||
sxth(ARMEmitter::Size::i64Bit, Dst, TMP1);
|
||||
} break;
|
||||
case FABI_I32_I16_F80: {
|
||||
case FABI_I32_I16_F80_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
mov(ARMEmitter::VReg::v0.Q(), Src1.Q());
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint16_t, FEXCore::VectorRegType, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
FillI32Result();
|
||||
} break;
|
||||
case FABI_I64_I16_F80: {
|
||||
case FABI_I64_I16_F80_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
mov(ARMEmitter::VReg::v0.Q(), Src1.Q());
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint16_t, FEXCore::VectorRegType, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
@@ -325,24 +329,28 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
const auto Dst = GetReg(Node);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst, TMP1);
|
||||
} break;
|
||||
case FABI_I64_I16_F80_F80: {
|
||||
case FABI_I64_I16_F80_F80_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src2 = GetVReg(IROp->Args[1].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r3, Src2, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r4, Src2, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x5, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint16_t, uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r5);
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
mov(ARMEmitter::VReg::v1.Q(), Src2.Q());
|
||||
mov(ARMEmitter::VReg::v0.Q(), VTMP1.Q());
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r5);
|
||||
mov(ARMEmitter::VReg::v0.Q(), Src1.Q());
|
||||
mov(ARMEmitter::VReg::v1.Q(), Src2.Q());
|
||||
}
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint16_t, FEXCore::VectorRegType, FEXCore::VectorRegType, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
@@ -353,42 +361,47 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
const auto Dst = GetReg(Node);
|
||||
mov(ARMEmitter::Size::i64Bit, Dst, TMP1);
|
||||
} break;
|
||||
case FABI_F80_I16_F80: {
|
||||
case FABI_F80_I16_F80_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
mov(ARMEmitter::VReg::v0.Q(), Src1.Q());
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
GenerateIndirectRuntimeCall<FEXCore::VectorRegType, uint16_t, FEXCore::VectorRegType, uint64_t>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
FillF80Result();
|
||||
} break;
|
||||
case FABI_F80_I16_F80_F80: {
|
||||
case FABI_F80_I16_F80_F80_PTR: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src2 = GetVReg(IROp->Args[1].ID());
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r3, Src2, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r4, Src2, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x5, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r5);
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
mov(ARMEmitter::VReg::v1.Q(), Src2.Q());
|
||||
mov(ARMEmitter::VReg::v0.Q(), VTMP1.Q());
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r5);
|
||||
mov(ARMEmitter::VReg::v0.Q(), Src1.Q());
|
||||
mov(ARMEmitter::VReg::v1.Q(), Src2.Q());
|
||||
}
|
||||
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<FEXCore::VectorRegType, uint16_t, FEXCore::VectorRegType, FEXCore::VectorRegType, uint64_t>(
|
||||
ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
}
|
||||
|
||||
FillF80Result();
|
||||
@@ -430,7 +443,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
|
||||
FillI32Result();
|
||||
} break;
|
||||
case FABI_I32_I128_I128_I16: {
|
||||
case FABI_I32_V128_V128_I16: {
|
||||
SpillForABICall(Info.SupportsPreserveAllABI, TMP1, true);
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPISTRX>();
|
||||
@@ -439,19 +452,22 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
const auto Src2 = GetVReg(Op->RHS.ID());
|
||||
const auto Control = Op->Control;
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 1);
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r2, Src2, 0);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r3, Src2, 1);
|
||||
|
||||
movz(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r4, Control);
|
||||
|
||||
ldr(ARMEmitter::XReg::x5, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint64_t, uint64_t, uint64_t, uint64_t, uint16_t>(ARMEmitter::Reg::r5);
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
mov(ARMEmitter::VReg::v1.Q(), Src2.Q());
|
||||
mov(ARMEmitter::VReg::v0.Q(), VTMP1.Q());
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r5);
|
||||
mov(ARMEmitter::VReg::v0.Q(), Src1.Q());
|
||||
mov(ARMEmitter::VReg::v1.Q(), Src2.Q());
|
||||
}
|
||||
|
||||
movz(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r0, Control);
|
||||
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint32_t, FEXCore::VectorRegType, FEXCore::VectorRegType, uint16_t>(ARMEmitter::Reg::r1);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
FillI32Result();
|
||||
@@ -466,7 +482,6 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
static void DirectBlockDelinker(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
|
||||
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
|
||||
uintptr_t branch = (uintptr_t)(Record)-8;
|
||||
|
||||
@@ -57,10 +57,10 @@ private:
|
||||
const bool HostSupportsRPRES {};
|
||||
const bool HostSupportsAFP {};
|
||||
|
||||
ARMEmitter::BiDirectionalLabel* PendingTargetLabel;
|
||||
FEXCore::Context::ContextImpl* CTX;
|
||||
const FEXCore::IR::IRListView* IR;
|
||||
uint64_t Entry;
|
||||
ARMEmitter::BiDirectionalLabel* PendingTargetLabel {};
|
||||
FEXCore::Context::ContextImpl* CTX {};
|
||||
const FEXCore::IR::IRListView* IR {};
|
||||
uint64_t Entry {};
|
||||
CPUBackend::CompiledCode CodeData {};
|
||||
|
||||
fextl::map<IR::NodeID, ARMEmitter::BiDirectionalLabel> JumpTargets;
|
||||
@@ -257,9 +257,9 @@ private:
|
||||
|
||||
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
|
||||
void EmitDetectionString();
|
||||
IR::RegisterAllocationPass* RAPass;
|
||||
const IR::RegisterAllocationData* RAData;
|
||||
FEXCore::Core::DebugData* DebugData;
|
||||
IR::RegisterAllocationPass* RAPass {};
|
||||
const IR::RegisterAllocationData* RAData {};
|
||||
FEXCore::Core::DebugData* DebugData {};
|
||||
|
||||
void ResetStack();
|
||||
/**
|
||||
|
||||
@@ -642,10 +642,10 @@ ARMEmitter::SVEMemOperand Arm64JITCore::GenerateSVEMemOperand(IR::OpSize AccessS
|
||||
}
|
||||
|
||||
const auto SignedConst = static_cast<int64_t>(Const);
|
||||
const auto SignedAVXSize = static_cast<int64_t>(Core::CPUState::XMM_AVX_REG_SIZE);
|
||||
const auto SignedSVESize = static_cast<int64_t>(HostSupportsSVE256 ? Core::CPUState::XMM_AVX_REG_SIZE : Core::CPUState::XMM_SSE_REG_SIZE);
|
||||
|
||||
const auto IsCleanlyDivisible = (SignedConst % SignedAVXSize) == 0;
|
||||
const auto Index = SignedConst / SignedAVXSize;
|
||||
const auto IsCleanlyDivisible = (SignedConst % SignedSVESize) == 0;
|
||||
const auto Index = SignedConst / SignedSVESize;
|
||||
|
||||
// SVE's immediate variants of load stores are quite limited in terms
|
||||
// of immediate range. They also operate on a by-vector-length basis.
|
||||
@@ -759,7 +759,8 @@ DEF_OP(LoadMemTSO) {
|
||||
const auto Dst = GetReg(Node);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
LOGMAN_THROW_A_FMT(IsInlineConstant(Op->Offset, &Offset), "expected immediate");
|
||||
[[maybe_unused]] bool IsInline = IsInlineConstant(Op->Offset, &Offset);
|
||||
LOGMAN_THROW_A_FMT(IsInline, "expected immediate");
|
||||
}
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
@@ -836,7 +837,7 @@ DEF_OP(VLoadVectorMasked) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
const auto CMPPredicate = ARMEmitter::PReg::p0;
|
||||
@@ -891,7 +892,15 @@ DEF_OP(VLoadVectorMasked) {
|
||||
auto WorkingReg = TMP1;
|
||||
auto TempMemReg = MemReg;
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, TempDst.Q(), 0);
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid(), "Complex addressing requested and not supported!");
|
||||
uint64_t Const {};
|
||||
if (Op->Offset.IsInvalid()) {
|
||||
// Intentional no-op.
|
||||
} else if (IsInlineConstant(Op->Offset, &Const)) {
|
||||
TempMemReg = TMP2;
|
||||
add(ARMEmitter::Size::i64Bit, TMP2, MemReg, Const);
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Complex addressing requested and not supported!");
|
||||
}
|
||||
|
||||
const uint64_t ElementSizeInBits = IR::OpSizeAsBits(IROp->ElementSize);
|
||||
for (size_t i = 0; i < NumElements; ++i) {
|
||||
@@ -931,7 +940,7 @@ DEF_OP(VStoreVectorMasked) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
const auto CMPPredicate = ARMEmitter::PReg::p0;
|
||||
@@ -983,7 +992,16 @@ DEF_OP(VStoreVectorMasked) {
|
||||
// Use VTMP1 as the temporary destination
|
||||
auto WorkingReg = TMP1;
|
||||
auto TempMemReg = MemReg;
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid(), "Complex addressing requested and not supported!");
|
||||
|
||||
uint64_t Const {};
|
||||
if (Op->Offset.IsInvalid()) {
|
||||
// Intentional no-op.
|
||||
} else if (IsInlineConstant(Op->Offset, &Const)) {
|
||||
TempMemReg = TMP2;
|
||||
add(ARMEmitter::Size::i64Bit, TMP2, MemReg, Const);
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Complex addressing requested and not supported!");
|
||||
}
|
||||
|
||||
const uint64_t ElementSizeInBits = IR::OpSizeAsBits(IROp->ElementSize);
|
||||
for (size_t i = 0; i < NumElements; ++i) {
|
||||
@@ -1020,6 +1038,7 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
|
||||
ARMEmitter::VRegister VectorIndexLow, std::optional<ARMEmitter::VRegister> VectorIndexHigh,
|
||||
ARMEmitter::VRegister MaskReg, IR::OpSize VectorIndexSize, size_t DataElementOffsetStart,
|
||||
size_t IndexElementOffsetStart, uint8_t OffsetScale) {
|
||||
LOGMAN_THROW_A_FMT(ElementSize >= IR::OpSize::i8Bit && ElementSize <= IR::OpSize::i64Bit, "Invalid element size");
|
||||
|
||||
const auto PerformSMove = [this](IR::OpSize ElementSize, const ARMEmitter::Register Dst, const ARMEmitter::VRegister Vector, int index) {
|
||||
switch (ElementSize) {
|
||||
@@ -1149,7 +1168,7 @@ DEF_OP(VLoadVectorGatherMasked) {
|
||||
/// - AddrBase also doesn't need to exist
|
||||
/// - If the instruction is using 64-bit vector indexing or 32-bit addresses where the top-bit isn't set then this is valid!
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto IncomingDst = GetVReg(Op->Incoming.ID());
|
||||
@@ -1380,7 +1399,7 @@ DEF_OP(VBroadcastFromMem) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -1547,7 +1566,7 @@ DEF_OP(StoreMem) {
|
||||
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: strb(Src, MemSrc); break;
|
||||
case IR::OpSize::i16Bit: strh(Src, MemSrc); break;
|
||||
@@ -1658,8 +1677,8 @@ DEF_OP(StoreMemPair) {
|
||||
const auto Addr = GetReg(Op->Addr.ID());
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src1 = GetReg(Op->Value1.ID());
|
||||
const auto Src2 = GetReg(Op->Value2.ID());
|
||||
const auto Src1 = GetZeroableReg(Op->Value1);
|
||||
const auto Src2 = GetZeroableReg(Op->Value2);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i32Bit: stp<ARMEmitter::IndexType::OFFSET>(Src1.W(), Src2.W(), Addr, Op->Offset); break;
|
||||
case IR::OpSize::i64Bit: stp<ARMEmitter::IndexType::OFFSET>(Src1.X(), Src2.X(), Addr, Op->Offset); break;
|
||||
@@ -1691,10 +1710,11 @@ DEF_OP(StoreMemTSO) {
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
LOGMAN_THROW_A_FMT(IsInlineConstant(Op->Offset, &Offset), "expected immediate");
|
||||
[[maybe_unused]] bool IsInline = IsInlineConstant(Op->Offset, &Offset);
|
||||
LOGMAN_THROW_A_FMT(IsInline, "expected immediate");
|
||||
}
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
@@ -1711,7 +1731,7 @@ DEF_OP(StoreMemTSO) {
|
||||
}
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
@@ -1763,7 +1783,7 @@ DEF_OP(MemSet) {
|
||||
const bool IsAtomic = CTX->IsMemcpyAtomicTSOEnabled();
|
||||
const auto Size = IR::OpSizeToSize(Op->Size);
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto Value = GetReg(Op->Value.ID());
|
||||
const auto Value = GetZeroableReg(Op->Value);
|
||||
const auto Length = GetReg(Op->Length.ID());
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
@@ -2312,7 +2332,7 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
auto MemReg = GetReg(Op->Addr.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
if (!IsInlineConstant(Op->Offset, &Offset)) {
|
||||
@@ -2332,7 +2352,7 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
}
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP1, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: stlrb(Src, MemReg); break;
|
||||
@@ -2504,7 +2524,7 @@ DEF_OP(VStoreNonTemporal) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
|
||||
|
||||
const auto Value = GetVReg(Op->Value.ID());
|
||||
@@ -2546,7 +2566,7 @@ DEF_OP(VLoadNonTemporal) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || (Is256Bit && HostSupportsSVE256), "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
|
||||
@@ -166,7 +166,7 @@ DEF_OP(PopRoundingMode) {
|
||||
DEF_OP(Print) {
|
||||
auto Op = IROp->C<IR::IROp_Print>();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
PushDynamicRegs(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
if (IsGPR(Op->Value.ID())) {
|
||||
@@ -189,7 +189,7 @@ DEF_OP(Print) {
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
PopDynamicRegsAndLR();
|
||||
PopDynamicRegs();
|
||||
}
|
||||
|
||||
DEF_OP(ProcessorID) {
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -27,7 +27,6 @@ $end_info$
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <bit>
|
||||
#include <cstdint>
|
||||
#include <tuple>
|
||||
|
||||
@@ -1000,6 +999,7 @@ void OpDispatchBuilder::TESTOp(OpcodeArgs, uint32_t SrcIndex) {
|
||||
Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
const auto Size = OpSizeFromDst(Op);
|
||||
LOGMAN_THROW_A_FMT(Size >= IR::OpSize::i8Bit && Size <= IR::OpSize::i64Bit, "Invalid size");
|
||||
|
||||
uint64_t Const;
|
||||
bool AlwaysNonnegative = false;
|
||||
@@ -1092,8 +1092,8 @@ void OpDispatchBuilder::CQOOp(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::XCHGOp(OpcodeArgs) {
|
||||
// Load both the source and the destination
|
||||
if (Op->OP == 0x90 && GetSrcSize(Op) >= 4 && Op->Src[0].IsGPR() && Op->Src[0].Data.GPR.GPR == FEXCore::X86State::REG_RAX &&
|
||||
Op->Dest.IsGPR() && Op->Dest.Data.GPR.GPR == FEXCore::X86State::REG_RAX) {
|
||||
if (Op->OP == 0x90 && Op->Src[0].IsGPR() && Op->Src[0].Data.GPR.GPR == FEXCore::X86State::REG_RAX && Op->Dest.IsGPR() &&
|
||||
Op->Dest.Data.GPR.GPR == FEXCore::X86State::REG_RAX) {
|
||||
// This is one heck of a sucky special case
|
||||
// If we are the 0x90 XCHG opcode (Meaning source is GPR RAX)
|
||||
// and destination register is ALSO RAX
|
||||
@@ -1103,6 +1103,14 @@ void OpDispatchBuilder::XCHGOp(OpcodeArgs) {
|
||||
// But this would result in a zext on 64bit, which would ruin the no-op nature of the instruction
|
||||
// So x86-64 spec mandates this special case that even though it is a 32bit instruction and
|
||||
// is supposed to zext the result, it is a true no-op
|
||||
//
|
||||
// x86 spec text here:
|
||||
//
|
||||
// XCHG (E)AX, (E)AX (encoded instruction byte is 90H) is an alias for
|
||||
// NOP regardless of data size prefixes, including REX.W.
|
||||
//
|
||||
// Note that also includes 16-bit so we don't gate this on size. The
|
||||
// sequence (66 90) is a valid two-byte nop that we also ignore.
|
||||
if (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX) {
|
||||
// If this instruction has a REP prefix then this is architecturally
|
||||
// defined to be a `PAUSE` instruction. On older processors this ends up
|
||||
@@ -1422,14 +1430,15 @@ void OpDispatchBuilder::SHLDOp(OpcodeArgs) {
|
||||
// Calculate flags early.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto Size = GetSrcBitSize(Op);
|
||||
|
||||
// Allow garbage on the Src if it will be ignored by the Lshr below
|
||||
Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = Size >= 32});
|
||||
Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags);
|
||||
|
||||
// Allow garbage on the shift, we're masking it anyway.
|
||||
Ref Shift = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
const auto Size = GetSrcBitSize(Op);
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op.
|
||||
if (Size == 64) {
|
||||
Shift = _And(OpSize::i64Bit, Shift, _InlineConstant(0x3F));
|
||||
@@ -1538,7 +1547,7 @@ void OpDispatchBuilder::SHRDImmediateOp(OpcodeArgs) {
|
||||
Ref ShiftRight = _Constant(Shift);
|
||||
auto ShiftLeft = _Constant(Size - Shift);
|
||||
|
||||
auto Tmp1 = _Lshr(Size == 64 ? OpSize::i64Bit : OpSize::i32Bit, Dest, ShiftRight);
|
||||
auto Tmp1 = _Lshr(OpSize::i32Bit, Dest, ShiftRight);
|
||||
auto Tmp2 = _Lshl(OpSize::i64Bit, Src, ShiftLeft);
|
||||
|
||||
Res = _Or(OpSize::i64Bit, Tmp1, Tmp2);
|
||||
@@ -1589,7 +1598,7 @@ void OpDispatchBuilder::RotateOp(OpcodeArgs, bool Left, bool IsImmediate, bool I
|
||||
|
||||
const uint32_t Size = GetSrcBitSize(Op);
|
||||
const auto OpSize = Size == 64 ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
uint64_t UnmaskedConst;
|
||||
uint64_t UnmaskedConst {};
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op. But it's
|
||||
// equivalent to mask to the actual size of the op, that way we can bound
|
||||
@@ -2176,7 +2185,7 @@ void OpDispatchBuilder::RCRSmallerOp(OpcodeArgs) {
|
||||
|
||||
StoreResult(GPRClass, Op, Res, OpSize::iInvalid);
|
||||
|
||||
uint64_t SrcConst;
|
||||
uint64_t SrcConst = 0;
|
||||
bool IsSrcConst = IsValueConstant(WrapNode(Src), &SrcConst);
|
||||
SrcConst &= 0x1f;
|
||||
|
||||
@@ -2400,7 +2409,10 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
unsigned LshrSize = std::max<uint8_t>(IR::OpSizeToSize(OpSize::i32Bit), Size / 8);
|
||||
auto BitSelect = (Size == (LshrSize * 8)) ? Src : _And(OpSize::i64Bit, Src, _Constant(Mask));
|
||||
|
||||
// OF/SF/ZF/AF/PF undefined.
|
||||
// OF/SF/AF/PF undefined. ZF must be preserved. We choose to preserve OF/SF
|
||||
// too since we just use an rmif to insert into CF directly. We could
|
||||
// optimize perhaps.
|
||||
//
|
||||
// Set CF before the action to save a move, except for complements where we
|
||||
// can reuse the invert.
|
||||
if (Action != BTAction::BTComplement) {
|
||||
@@ -2408,7 +2420,8 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
Value = _Lshr(IR::SizeToOpSize(LshrSize), Value, BitSelect);
|
||||
}
|
||||
|
||||
SetCFDirect_InvalidateNZV(Value, ConstantShift, Value);
|
||||
SetRFLAG(Value, X86State::RFLAG_CF_RAW_LOC, ConstantShift, true);
|
||||
CFInverted = false;
|
||||
}
|
||||
|
||||
switch (Action) {
|
||||
@@ -2441,7 +2454,9 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
Value = Dest;
|
||||
}
|
||||
|
||||
SetCFInverted_InvalidateNZV(Value, ConstantShift, true);
|
||||
SetRFLAG(Value, X86State::RFLAG_CF_RAW_LOC, ConstantShift, true);
|
||||
CFInverted = true;
|
||||
|
||||
StoreResult(GPRClass, Op, Dest, OpSize::iInvalid);
|
||||
break;
|
||||
}
|
||||
@@ -2473,7 +2488,7 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
|
||||
if (DestIsLockedMem(Op)) {
|
||||
HandledLock = true;
|
||||
Value = _AtomicFetchCLR(OpSize::i8Bit, BitMask, LoadEffectiveAddress(Address, true));
|
||||
Value = _AtomicFetchCLR(OpSize::i8Bit, BitMask, LoadEffectiveAddress(this, Address, CTX->GetGPROpSize(), true));
|
||||
} else {
|
||||
Value = _LoadMemAutoTSO(GPRClass, OpSize::i8Bit, Address, OpSize::i8Bit);
|
||||
|
||||
@@ -2488,7 +2503,7 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
|
||||
if (DestIsLockedMem(Op)) {
|
||||
HandledLock = true;
|
||||
Value = _AtomicFetchOr(OpSize::i8Bit, BitMask, LoadEffectiveAddress(Address, true));
|
||||
Value = _AtomicFetchOr(OpSize::i8Bit, BitMask, LoadEffectiveAddress(this, Address, CTX->GetGPROpSize(), true));
|
||||
} else {
|
||||
Value = _LoadMemAutoTSO(GPRClass, OpSize::i8Bit, Address, OpSize::i8Bit);
|
||||
|
||||
@@ -2503,7 +2518,7 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
|
||||
if (DestIsLockedMem(Op)) {
|
||||
HandledLock = true;
|
||||
Value = _AtomicFetchXor(OpSize::i8Bit, BitMask, LoadEffectiveAddress(Address, true));
|
||||
Value = _AtomicFetchXor(OpSize::i8Bit, BitMask, LoadEffectiveAddress(this, Address, CTX->GetGPROpSize(), true));
|
||||
} else {
|
||||
Value = _LoadMemAutoTSO(GPRClass, OpSize::i8Bit, Address, OpSize::i8Bit);
|
||||
|
||||
@@ -2658,7 +2673,7 @@ void OpDispatchBuilder::MULOp(OpcodeArgs) {
|
||||
|
||||
Ref Src1 = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
Ref Src2 = LoadGPRRegister(X86State::REG_RAX);
|
||||
Ref Result;
|
||||
Ref Result {};
|
||||
|
||||
if (Size != OpSize::i64Bit) {
|
||||
Src1 = _Bfe(OpSize::i64Bit, SizeBits, 0, Src1);
|
||||
@@ -2706,6 +2721,7 @@ void OpDispatchBuilder::MULOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::NOTOp(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
const auto SizeBits = IR::OpSizeAsBits(Size);
|
||||
LOGMAN_THROW_A_FMT(Size >= IR::OpSize::i8Bit && Size <= IR::OpSize::i64Bit, "Invalid size");
|
||||
|
||||
Ref MaskConst {};
|
||||
if (Size == OpSize::i64Bit) {
|
||||
@@ -3001,6 +3017,22 @@ void OpDispatchBuilder::SGDTOp(OpcodeArgs) {
|
||||
_StoreMemAutoTSO(GPRClass, GDTStoreSize, AddressMode {.Base = DestAddress, .Offset = 2, .AddrSize = OpSize::i64Bit}, _Constant(GDTAddress));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SIDTOp(OpcodeArgs) {
|
||||
auto DestAddress = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
|
||||
// See SGDTOp, matches Linux in reported values
|
||||
uint64_t IDTAddress = 0xFFFFFE0000000000ULL;
|
||||
auto IDTStoreSize = OpSize::i64Bit;
|
||||
if (!CTX->Config.Is64BitMode) {
|
||||
// Mask off upper bits if 32-bit result.
|
||||
IDTAddress &= ~0U;
|
||||
IDTStoreSize = OpSize::i32Bit;
|
||||
}
|
||||
|
||||
_StoreMemAutoTSO(GPRClass, OpSize::i16Bit, DestAddress, _Constant(0xfff));
|
||||
_StoreMemAutoTSO(GPRClass, IDTStoreSize, AddressMode {.Base = DestAddress, .Offset = 2, .AddrSize = OpSize::i64Bit}, _Constant(IDTAddress));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SMSWOp(OpcodeArgs) {
|
||||
const bool IsMemDst = DestIsMem(Op);
|
||||
|
||||
@@ -3297,7 +3329,7 @@ void OpDispatchBuilder::CMPSOp(OpcodeArgs) {
|
||||
SetCurrentCodeBlock(BeforeLoop);
|
||||
StartNewBlock();
|
||||
|
||||
ForeachDirection([this, Op, Size, REPE](int PtrDir) {
|
||||
ForeachDirection([this, Op, Size, REPE](int32_t PtrDir) {
|
||||
IRPair<IROp_CondJump> InnerJump;
|
||||
auto JumpIntoLoop = Jump();
|
||||
|
||||
@@ -3330,11 +3362,11 @@ void OpDispatchBuilder::CMPSOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RCX, TailCounter);
|
||||
|
||||
// Offset the pointer
|
||||
Dest_RDI = _Add(OpSize::i64Bit, Dest_RDI, _Constant(PtrDir * IR::OpSizeToSize(Size)));
|
||||
Dest_RDI = _Add(OpSize::i64Bit, Dest_RDI, _Constant(PtrDir * static_cast<int32_t>(IR::OpSizeToSize(Size))));
|
||||
StoreGPRRegister(X86State::REG_RDI, Dest_RDI);
|
||||
|
||||
// Offset second pointer
|
||||
Dest_RSI = _Add(OpSize::i64Bit, Dest_RSI, _Constant(PtrDir * IR::OpSizeToSize(Size)));
|
||||
Dest_RSI = _Add(OpSize::i64Bit, Dest_RSI, _Constant(PtrDir * static_cast<int32_t>(IR::OpSizeToSize(Size))));
|
||||
StoreGPRRegister(X86State::REG_RSI, Dest_RSI);
|
||||
|
||||
// If TailCounter != 0, compare sources.
|
||||
@@ -3396,7 +3428,7 @@ void OpDispatchBuilder::LODSOp(OpcodeArgs) {
|
||||
// Calculate flags early. because end of block
|
||||
CalculateDeferredFlags();
|
||||
|
||||
ForeachDirection([this, Op, Size](int PtrDir) {
|
||||
ForeachDirection([this, Op, Size](int32_t PtrDir) {
|
||||
// XXX: Theoretically LODS could be optimized to
|
||||
// RSI += {-}(RCX * Size)
|
||||
// RAX = [RSI - Size]
|
||||
@@ -3440,7 +3472,7 @@ void OpDispatchBuilder::LODSOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RCX, TailCounter);
|
||||
|
||||
// Offset the pointer
|
||||
TailDest_RSI = _Add(OpSize::i64Bit, TailDest_RSI, _Constant(PtrDir * IR::OpSizeToSize(Size)));
|
||||
TailDest_RSI = _Add(OpSize::i64Bit, TailDest_RSI, _Constant(PtrDir * static_cast<int32_t>(IR::OpSizeToSize(Size))));
|
||||
StoreGPRRegister(X86State::REG_RSI, TailDest_RSI);
|
||||
|
||||
// Jump back to the start, we have more work to do
|
||||
@@ -3480,7 +3512,7 @@ void OpDispatchBuilder::SCASOp(OpcodeArgs) {
|
||||
// Calculate flags early. because end of block
|
||||
CalculateDeferredFlags();
|
||||
|
||||
ForeachDirection([this, Op, Size](int Dir) {
|
||||
ForeachDirection([this, Op, Size](int32_t Dir) {
|
||||
bool REPE = Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX;
|
||||
|
||||
auto JumpStart = Jump();
|
||||
@@ -3524,7 +3556,7 @@ void OpDispatchBuilder::SCASOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RCX, TailCounter);
|
||||
|
||||
// Offset the pointer
|
||||
TailDest_RDI = _Add(OpSize::i64Bit, TailDest_RDI, _Constant(Dir * IR::OpSizeToSize(Size)));
|
||||
TailDest_RDI = _Add(OpSize::i64Bit, TailDest_RDI, _Constant(Dir * static_cast<int32_t>(IR::OpSizeToSize(Size))));
|
||||
StoreGPRRegister(X86State::REG_RDI, TailDest_RDI);
|
||||
|
||||
CalculateDeferredFlags();
|
||||
@@ -3772,7 +3804,7 @@ void OpDispatchBuilder::CMPXCHGOp(OpcodeArgs) {
|
||||
|
||||
if (GPRSize == OpSize::i64Bit && Size == OpSize::i32Bit) {
|
||||
Src1 = LoadSource_WithOpSize(GPRClass, Op, Op->Dest, GPRSize, Op->Flags, {.AllowUpperGarbage = true});
|
||||
Src1Lower = _Bfe(GPRSize, IR::OpSizeAsBits(Size), 0, Src1);
|
||||
Src1Lower = Trivial ? Src1 : _Bfe(GPRSize, IR::OpSizeAsBits(Size), 0, Src1);
|
||||
} else {
|
||||
Src1 = LoadSource_WithOpSize(GPRClass, Op, Op->Dest, Size, Op->Flags, {.AllowUpperGarbage = true});
|
||||
Src1Lower = Src1;
|
||||
@@ -3809,15 +3841,9 @@ void OpDispatchBuilder::CMPXCHGOp(OpcodeArgs) {
|
||||
Ref Src2 = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
HandledLock = Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK;
|
||||
|
||||
Ref Src3 {};
|
||||
Ref Src3Lower {};
|
||||
if (GPRSize == OpSize::i64Bit && Size == OpSize::i32Bit) {
|
||||
Src3 = LoadGPRRegister(X86State::REG_RAX);
|
||||
Src3Lower = _Bfe(OpSize::i32Bit, 32, 0, Src3);
|
||||
} else {
|
||||
Src3 = LoadGPRRegister(X86State::REG_RAX, Size);
|
||||
Src3Lower = Src3;
|
||||
}
|
||||
auto Src3 = LoadGPRRegister(X86State::REG_RAX);
|
||||
auto Src3Lower = _Bfe(OpSize::i64Bit, OpSizeAsBits(Size), 0, Src3);
|
||||
|
||||
// If this is a memory location then we want the pointer to it
|
||||
Ref Src1 = MakeSegmentAddress(Op, Op->Dest);
|
||||
|
||||
@@ -3825,7 +3851,7 @@ void OpDispatchBuilder::CMPXCHGOp(OpcodeArgs) {
|
||||
// if (DataSrc == Src3) { *Src1 == Src2; } Src2 = DataSrc
|
||||
// This will write to memory! Careful!
|
||||
// Third operand must be a calculated guest memory address
|
||||
Ref CASResult = _CAS(Size, Src3Lower, Src2, Src1);
|
||||
Ref CASResult = _CAS(Size, Src3, Src2, Src1);
|
||||
Ref RAXResult = CASResult;
|
||||
|
||||
CalculateFlags_SUB(OpSizeFromSrc(Op), Src3Lower, CASResult);
|
||||
@@ -3996,7 +4022,7 @@ Ref OpDispatchBuilder::GetSegment(uint32_t Flags, uint32_t DefaultPrefix, bool O
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX:
|
||||
SegmentResult = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, gs_cached));
|
||||
break;
|
||||
default: break; // Do nothing
|
||||
default: return nullptr;
|
||||
}
|
||||
|
||||
CheckLegacySegmentRead(SegmentResult, Prefix);
|
||||
@@ -4126,94 +4152,6 @@ void OpDispatchBuilder::UpdatePrefixFromSegment(Ref Segment, uint32_t SegmentReg
|
||||
}
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::LoadEffectiveAddress(AddressMode A, bool AddSegmentBase, bool AllowUpperGarbage) {
|
||||
const auto GPRSize = CTX->GetGPROpSize();
|
||||
Ref Tmp = A.Base;
|
||||
|
||||
if (A.Offset) {
|
||||
Ref Offset = _Constant(A.Offset);
|
||||
Tmp = Tmp ? _Add(GPRSize, Tmp, Offset) : Offset;
|
||||
}
|
||||
|
||||
if (A.Index) {
|
||||
if (A.IndexScale != 1) {
|
||||
LOGMAN_THROW_A_FMT((A.IndexScale & (A.IndexScale - 1)) == 0, "power of two");
|
||||
uint32_t Log2 = FEXCore::ilog2(A.IndexScale);
|
||||
|
||||
if (Tmp) {
|
||||
Tmp = _AddShift(GPRSize, Tmp, A.Index, ShiftType::LSL, Log2);
|
||||
} else {
|
||||
Tmp = _Lshl(GPRSize, A.Index, _Constant(Log2));
|
||||
}
|
||||
} else {
|
||||
Tmp = Tmp ? _Add(GPRSize, Tmp, A.Index) : A.Index;
|
||||
}
|
||||
}
|
||||
|
||||
// For 64-bit AddrSize can be 32-bit or 64-bit
|
||||
// For 32-bit AddrSize can be 32-bit or 16-bit
|
||||
//
|
||||
// If the AddrSize is not the GPRSize then we need to clear the upper bits.
|
||||
if ((A.AddrSize < GPRSize) && !AllowUpperGarbage && Tmp) {
|
||||
Tmp = _Bfe(GPRSize, IR::OpSizeAsBits(A.AddrSize), 0, Tmp);
|
||||
}
|
||||
|
||||
if (A.Segment && AddSegmentBase) {
|
||||
Tmp = Tmp ? _Add(GPRSize, Tmp, A.Segment) : A.Segment;
|
||||
}
|
||||
|
||||
return Tmp ?: _Constant(0);
|
||||
}
|
||||
|
||||
AddressMode OpDispatchBuilder::SelectAddressMode(AddressMode A, bool AtomicTSO, bool Vector, IR::OpSize AccessSize) {
|
||||
const auto GPRSize = CTX->GetGPROpSize();
|
||||
|
||||
// In the future this also needs to account for LRCPC3.
|
||||
bool SupportsRegIndex = Vector || !AtomicTSO;
|
||||
|
||||
// Try a constant offset. For 64-bit, this maps directly. For 32-bit, this
|
||||
// works only for displacements with magnitude < 16KB, since those bottom
|
||||
// addresses are reserved and therefore wrap around is invalid.
|
||||
//
|
||||
// TODO: Also handle GPR TSO if we can guarantee the constant inlines.
|
||||
if (SupportsRegIndex) {
|
||||
if ((A.Base || A.Segment) && A.Offset) {
|
||||
const bool Const_16K = A.Offset > -16384 && A.Offset < 16384 && A.AddrSize == OpSize::i32Bit && GPRSize == OpSize::i32Bit;
|
||||
|
||||
if ((A.AddrSize == OpSize::i64Bit) || Const_16K) {
|
||||
// Peel off the offset
|
||||
AddressMode B = A;
|
||||
B.Offset = 0;
|
||||
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(B, true /* AddSegmentBase */, false),
|
||||
.Index = _Constant(A.Offset),
|
||||
.IndexType = MEM_OFFSET_SXTX,
|
||||
.IndexScale = 1,
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
// Try a (possibly scaled) register index.
|
||||
if (A.AddrSize == OpSize::i64Bit && A.Base && (A.Index || A.Segment) && !A.Offset &&
|
||||
(A.IndexScale == 1 || A.IndexScale == IR::OpSizeToSize(AccessSize))) {
|
||||
if (A.Index && A.Segment) {
|
||||
A.Base = _Add(GPRSize, A.Base, A.Segment);
|
||||
} else if (A.Segment) {
|
||||
A.Index = A.Segment;
|
||||
A.IndexScale = 1;
|
||||
}
|
||||
return A;
|
||||
}
|
||||
}
|
||||
|
||||
// Fallback on software address calculation
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(A, true),
|
||||
.Index = InvalidNode,
|
||||
};
|
||||
}
|
||||
|
||||
AddressMode OpDispatchBuilder::DecodeAddress(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand,
|
||||
MemoryAccessType AccessType, bool IsLoad) {
|
||||
const auto GPRSize = CTX->GetGPROpSize();
|
||||
@@ -4311,7 +4249,7 @@ Ref OpDispatchBuilder::LoadSource_WithOpSize(RegisterClassType Class, const X86T
|
||||
|
||||
if ((IsOperandMem(Operand, true) && LoadData) || ForceLoad) {
|
||||
if (OpSize == OpSize::f80Bit) {
|
||||
Ref MemSrc = LoadEffectiveAddress(A, true);
|
||||
Ref MemSrc = LoadEffectiveAddress(this, A, CTX->GetGPROpSize(), true);
|
||||
if (CTX->HostFeatures.SupportsSVE128 || CTX->HostFeatures.SupportsSVE256) {
|
||||
return _LoadMemX87SVEOptPredicate(OpSize::i128Bit, OpSize::i16Bit, MemSrc);
|
||||
} else {
|
||||
@@ -4323,7 +4261,7 @@ Ref OpDispatchBuilder::LoadSource_WithOpSize(RegisterClassType Class, const X86T
|
||||
|
||||
return _LoadMemAutoTSO(Class, OpSize, A, Align == OpSize::iInvalid ? OpSize : Align);
|
||||
} else {
|
||||
return LoadEffectiveAddress(A, false, AllowUpperGarbage);
|
||||
return LoadEffectiveAddress(this, A, CTX->GetGPROpSize(), false, AllowUpperGarbage);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4444,7 +4382,7 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
AddressMode A = DecodeAddress(Op, Operand, AccessType, false /* IsLoad */);
|
||||
|
||||
if (OpSize == OpSize::f80Bit) {
|
||||
Ref MemStoreDst = LoadEffectiveAddress(A, true);
|
||||
Ref MemStoreDst = LoadEffectiveAddress(this, A, CTX->GetGPROpSize(), true);
|
||||
if (CTX->HostFeatures.SupportsSVE128 || CTX->HostFeatures.SupportsSVE256) {
|
||||
_StoreMemX87SVEOptPredicate(OpSize::i128Bit, OpSize::i16Bit, Src, MemStoreDst);
|
||||
} else {
|
||||
@@ -4607,6 +4545,12 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::I
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::LSLOp(OpcodeArgs) {
|
||||
// Emulate by always returning failure, this deviates from both Linux and Windows but
|
||||
// shouldn't be depended on by anything.
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(_Constant(0));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::INTOp(OpcodeArgs) {
|
||||
IR::BreakDefinition Reason;
|
||||
bool SetRIPToNext = false;
|
||||
@@ -5496,9 +5440,9 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
|
||||
// 1 = Invalid
|
||||
|
||||
{OPDReg(0xD9, 2) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FSTF64, OpSize::i32Bit>},
|
||||
{OPDReg(0xD9, 2) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FST, OpSize::i32Bit>},
|
||||
|
||||
{OPDReg(0xD9, 3) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FSTF64, OpSize::i32Bit>},
|
||||
{OPDReg(0xD9, 3) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FST, OpSize::i32Bit>},
|
||||
|
||||
{OPDReg(0xD9, 4) | 0x00, 8, &OpDispatchBuilder::X87LDENVF64},
|
||||
|
||||
@@ -5593,7 +5537,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
|
||||
// 6 = Invalid
|
||||
|
||||
{OPDReg(0xDB, 7) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FSTF64, OpSize::f80Bit>},
|
||||
{OPDReg(0xDB, 7) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FST, OpSize::f80Bit>},
|
||||
|
||||
|
||||
{OPD(0xDB, 0xC0), 8, &OpDispatchBuilder::X87FCMOV},
|
||||
@@ -5644,9 +5588,9 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
|
||||
{OPDReg(0xDD, 1) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FISTF64, true>},
|
||||
|
||||
{OPDReg(0xDD, 2) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FSTF64, OpSize::i64Bit>},
|
||||
{OPDReg(0xDD, 2) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FST, OpSize::i64Bit>},
|
||||
|
||||
{OPDReg(0xDD, 3) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FSTF64, OpSize::i64Bit>},
|
||||
{OPDReg(0xDD, 3) | 0x00, 8, &OpDispatchBuilder::Bind<&OpDispatchBuilder::FST, OpSize::i64Bit>},
|
||||
|
||||
{OPDReg(0xDD, 4) | 0x00, 8, &OpDispatchBuilder::X87FRSTOR},
|
||||
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Core/Addressing.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
@@ -46,6 +47,12 @@ enum class BTAction {
|
||||
BTComplement,
|
||||
};
|
||||
|
||||
enum class ForceTSOMode {
|
||||
NoOverride,
|
||||
ForceDisabled,
|
||||
ForceEnabled,
|
||||
};
|
||||
|
||||
struct LoadSourceOptions {
|
||||
// Alignment of the load in bytes. iInvalid signifies opsize aligned.
|
||||
IR::OpSize Align = OpSize::iInvalid;
|
||||
@@ -72,19 +79,6 @@ struct LoadSourceOptions {
|
||||
bool AllowUpperGarbage = false;
|
||||
};
|
||||
|
||||
struct AddressMode {
|
||||
Ref Segment {nullptr};
|
||||
Ref Base {nullptr};
|
||||
Ref Index {nullptr};
|
||||
MemOffsetType IndexType = MEM_OFFSET_SXTX;
|
||||
uint8_t IndexScale = 1;
|
||||
int64_t Offset = 0;
|
||||
|
||||
// Size in bytes for the address calculation. 8 for an arm64 hardware mode.
|
||||
IR::OpSize AddrSize;
|
||||
bool NonTSO;
|
||||
};
|
||||
|
||||
class OpDispatchBuilder final : public IREmitter {
|
||||
friend class FEXCore::IR::Pass;
|
||||
friend class FEXCore::IR::PassManager;
|
||||
@@ -273,6 +267,13 @@ public:
|
||||
return HandledLock;
|
||||
}
|
||||
|
||||
void SetForceTSO(ForceTSOMode Mode) {
|
||||
ForceTSO = Mode;
|
||||
}
|
||||
ForceTSOMode GetForceTSO() const {
|
||||
return ForceTSO;
|
||||
}
|
||||
|
||||
void SetDumpIR(bool DumpIR) {
|
||||
ShouldDump = DumpIR;
|
||||
}
|
||||
@@ -302,6 +303,7 @@ public:
|
||||
void MOVVectorUnalignedOp(OpcodeArgs);
|
||||
void MOVVectorNTOp(OpcodeArgs);
|
||||
void ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp, unsigned SrcIdx);
|
||||
void LSLOp(OpcodeArgs);
|
||||
void INTOp(OpcodeArgs);
|
||||
void SyscallOp(OpcodeArgs, bool IsSyscallInst);
|
||||
void ThunkOp(OpcodeArgs);
|
||||
@@ -417,6 +419,7 @@ public:
|
||||
void EnterOp(OpcodeArgs);
|
||||
|
||||
void SGDTOp(OpcodeArgs);
|
||||
void SIDTOp(OpcodeArgs);
|
||||
void SMSWOp(OpcodeArgs);
|
||||
|
||||
enum class VectorOpType {
|
||||
@@ -434,6 +437,7 @@ public:
|
||||
|
||||
void VectorALUROp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
|
||||
void VectorUnaryOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
|
||||
void RSqrt3DNowOp(OpcodeArgs, bool Duplicate);
|
||||
template<FEXCore::IR::IROps IROp, IR::OpSize ElementSize>
|
||||
void VectorUnaryDuplicateOp(OpcodeArgs);
|
||||
|
||||
@@ -750,7 +754,6 @@ public:
|
||||
void FLDF64_Const(OpcodeArgs, uint64_t Num);
|
||||
void FLDF64(OpcodeArgs, IR::OpSize Width);
|
||||
void FMULF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpResult ResInST0);
|
||||
void FSTF64(OpcodeArgs, IR::OpSize Width);
|
||||
void FSUBF64(OpcodeArgs, IR::OpSize Width, bool Integer, bool Reverse, OpResult ResInST0);
|
||||
void FTSTF64(OpcodeArgs);
|
||||
void X87FLDCWF64(OpcodeArgs);
|
||||
@@ -901,6 +904,15 @@ public:
|
||||
return Pair;
|
||||
}
|
||||
|
||||
Ref SHADataShuffle(Ref Src) {
|
||||
// SHA data shuffle matches PSHUFD shuffle where elements are inverted.
|
||||
// Because this shuffle mask gets reused multiple times per instruction, it's always a win to load the mask once and reuse it.
|
||||
const uint32_t Shuffle = 0b00'01'10'11;
|
||||
auto LookupIndexes =
|
||||
LoadAndCacheIndexedNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFD, Shuffle * 16);
|
||||
return _VTBL1(OpSize::i128Bit, Src, LookupIndexes);
|
||||
}
|
||||
|
||||
RefPair AVX128_LoadSource_WithOpSize(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags,
|
||||
bool NeedsHigh, MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
|
||||
|
||||
@@ -1203,6 +1215,7 @@ public:
|
||||
uint64_t NextBit = (1ull << (Index - 1));
|
||||
uint32_t Offset = CacheIndexToContextOffset(Index);
|
||||
auto Class = CacheIndexClass(Index);
|
||||
LOGMAN_THROW_A_FMT(Offset != ~0U, "Invalid offset");
|
||||
|
||||
// Use stp where possible to store multiple values at a time. This accelerates AVX.
|
||||
// TODO: this is all really confusing because of backwards iteration,
|
||||
@@ -1320,6 +1333,7 @@ private:
|
||||
bool HandledLock {false};
|
||||
bool DecodeFailure {false};
|
||||
bool NeedsBlockEnd {false};
|
||||
ForceTSOMode ForceTSO {ForceTSOMode::NoOverride};
|
||||
// Used during new op bringup
|
||||
bool ShouldDump {false};
|
||||
|
||||
@@ -1480,9 +1494,6 @@ private:
|
||||
|
||||
Ref GetRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset = 0);
|
||||
|
||||
Ref LoadEffectiveAddress(AddressMode A, bool AddSegmentBase, bool AllowUpperGarbage = false);
|
||||
AddressMode SelectAddressMode(AddressMode A, bool AtomicTSO, bool Vector, IR::OpSize AccessSize);
|
||||
|
||||
bool IsOperandMem(const X86Tables::DecodedOperand& Operand, bool Load) {
|
||||
// Literals are immediates as sources but memory addresses as destinations.
|
||||
return !(Load && Operand.IsLiteral()) && !Operand.IsGPR();
|
||||
@@ -1723,27 +1734,6 @@ private:
|
||||
CFInverted = true;
|
||||
}
|
||||
|
||||
// As above but with
|
||||
//
|
||||
// x - 1
|
||||
//
|
||||
// If x = 0, hardware C is not set. If x = 1, hardware C is set.
|
||||
void SetCFInverted_InvalidateNZV(Ref Value, unsigned ValueOffset = 0, bool MustMask = false) {
|
||||
if (CTX->HostFeatures.SupportsFlagM) {
|
||||
// This turns into a single rmif
|
||||
SetCFInverted(Value, ValueOffset, MustMask);
|
||||
} else {
|
||||
// Do math on flagm
|
||||
if (ValueOffset || MustMask) {
|
||||
Value = _Bfe(OpSize::i64Bit, 1, ValueOffset, Value);
|
||||
}
|
||||
|
||||
HandleNZCVWrite();
|
||||
_SubNZCV(OpSize::i32Bit, Value, _InlineConstant(1));
|
||||
CFInverted = true;
|
||||
}
|
||||
}
|
||||
|
||||
void SetCFInverted(Ref Value, unsigned ValueOffset = 0, bool MustMask = false) {
|
||||
SetRFLAG(Value, X86State::RFLAG_CF_RAW_LOC, ValueOffset, MustMask);
|
||||
CFInverted = true;
|
||||
@@ -1825,12 +1815,12 @@ private:
|
||||
static const int AVXHigh0Index = 48;
|
||||
static const int AVXHigh15Index = 63;
|
||||
|
||||
int CacheIndexToContextOffset(int Index) {
|
||||
uint32_t CacheIndexToContextOffset(int Index) {
|
||||
switch (Index) {
|
||||
case MM0Index ... MM7Index: return offsetof(FEXCore::Core::CPUState, mm[Index - MM0Index]);
|
||||
case AVXHigh0Index ... AVXHigh15Index: return offsetof(FEXCore::Core::CPUState, avx_high[Index - AVXHigh0Index][0]);
|
||||
case AbridgedFTWIndex: return offsetof(FEXCore::Core::CPUState, AbridgedFTW);
|
||||
default: return -1;
|
||||
default: return ~0U;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2367,11 +2357,15 @@ private:
|
||||
bool BlockSetRIP {false};
|
||||
|
||||
bool Multiblock {};
|
||||
uint64_t Entry;
|
||||
uint64_t Entry {};
|
||||
IROp_IRHeader* CurrentHeader {};
|
||||
|
||||
bool IsTSOEnabled(FEXCore::IR::RegisterClassType Class) {
|
||||
if (Class == FPRClass) {
|
||||
if (ForceTSO == ForceTSOMode::ForceEnabled) {
|
||||
return true;
|
||||
} else if (ForceTSO == ForceTSOMode::ForceDisabled) {
|
||||
return false;
|
||||
} else if (Class == FPRClass) {
|
||||
return CTX->IsVectorAtomicTSOEnabled();
|
||||
} else {
|
||||
return CTX->IsAtomicTSOEnabled();
|
||||
@@ -2396,7 +2390,7 @@ private:
|
||||
|
||||
Ref _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, AddressMode A, IR::OpSize Align = IR::OpSize::i8Bit) {
|
||||
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
A = SelectAddressMode(A, AtomicTSO, Class != GPRClass, Size);
|
||||
A = SelectAddressMode(this, A, CTX->GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, AtomicTSO, Class != GPRClass, Size);
|
||||
|
||||
if (AtomicTSO) {
|
||||
return _LoadMemTSO(Class, Size, A.Base, A.Index, Align, A.IndexType, A.IndexScale);
|
||||
@@ -2416,7 +2410,7 @@ private:
|
||||
A.Offset = 0;
|
||||
}
|
||||
|
||||
Out.Base = LoadEffectiveAddress(A, true, false);
|
||||
Out.Base = LoadEffectiveAddress(this, A, CTX->GetGPROpSize(), true, false);
|
||||
return Out;
|
||||
}
|
||||
|
||||
@@ -2447,7 +2441,7 @@ private:
|
||||
|
||||
Ref _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, AddressMode A, Ref Value, IR::OpSize Align = IR::OpSize::i8Bit) {
|
||||
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
A = SelectAddressMode(A, AtomicTSO, Class != GPRClass, Size);
|
||||
A = SelectAddressMode(this, A, CTX->GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, AtomicTSO, Class != GPRClass, Size);
|
||||
|
||||
if (AtomicTSO) {
|
||||
return _StoreMemTSO(Class, Size, Value, A.Base, A.Index, Align, A.IndexType, A.IndexScale);
|
||||
|
||||
@@ -783,7 +783,7 @@ void OpDispatchBuilder::AVX128_VZERO(OpcodeArgs) {
|
||||
if (IsVZEROALL) {
|
||||
// NOTE: Despite the name being VZEROALL, this will still only ever
|
||||
// zero out up to the first 16 registers (even on AVX-512, where we have 32 registers)
|
||||
Ref ZeroVector;
|
||||
Ref ZeroVector {};
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; i++) {
|
||||
// Explicitly not caching named vector zero. This ensures that every register gets movi #0.0 directly.
|
||||
@@ -1326,10 +1326,12 @@ void OpDispatchBuilder::AVX128_MOVMSK(OpcodeArgs) {
|
||||
};
|
||||
|
||||
Ref GPR {};
|
||||
if (SrcSize == OpSize::i128Bit && ElementSize == OpSize::i64Bit) {
|
||||
GPR = Mask8Byte(Src.Low);
|
||||
} else if (SrcSize == OpSize::i128Bit && ElementSize == OpSize::i32Bit) {
|
||||
GPR = Mask4Byte(Src.Low);
|
||||
if (Is128Bit) {
|
||||
if (ElementSize == OpSize::i64Bit) {
|
||||
GPR = Mask8Byte(Src.Low);
|
||||
} else {
|
||||
GPR = Mask4Byte(Src.Low);
|
||||
}
|
||||
} else if (ElementSize == OpSize::i32Bit) {
|
||||
auto GPRLow = Mask4Byte(Src.Low);
|
||||
auto GPRHigh = Mask4Byte(Src.High);
|
||||
@@ -1670,7 +1672,6 @@ void OpDispatchBuilder::AVX128_VEXTRACT128(OpcodeArgs) {
|
||||
const auto DstIsXMM = Op->Dest.IsGPR();
|
||||
const auto Selector = Op->Src[1].Literal() & 0b1;
|
||||
|
||||
///< TODO: Once we support loading only upper-half of the ymm register we can load the half depending on selection literal.
|
||||
auto Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, true);
|
||||
|
||||
RefPair Result {};
|
||||
@@ -2032,7 +2033,18 @@ void OpDispatchBuilder::AVX128_VPALIGNR(OpcodeArgs) {
|
||||
return Src2;
|
||||
}
|
||||
|
||||
return _VExtr(OpSize::i128Bit, OpSize::i8Bit, Src1, Src2, Index);
|
||||
if (Index == 16) {
|
||||
return Src1;
|
||||
}
|
||||
|
||||
auto SanitizedIndex = Index;
|
||||
if (Index > 16) {
|
||||
Src2 = Src1;
|
||||
Src1 = LoadZeroVector(OpSize::i128Bit);
|
||||
SanitizedIndex -= 16;
|
||||
}
|
||||
|
||||
return _VExtr(OpSize::i128Bit, OpSize::i8Bit, Src1, Src2, SanitizedIndex);
|
||||
});
|
||||
}
|
||||
|
||||
@@ -2052,9 +2064,7 @@ void OpDispatchBuilder::AVX128_VMASKMOVImpl(OpcodeArgs, IR::OpSize ElementSize,
|
||||
auto Data = AVX128_LoadSource_WithOpSize(Op, DataOp, Op->Flags, !Is128Bit);
|
||||
_VStoreVectorMasked(OpSize::i128Bit, ElementSize, Mask.Low, Data.Low, Address, Invalid(), MEM_OFFSET_SXTX, 1);
|
||||
if (!Is128Bit) {
|
||||
///< TODO: This can be cleaner if AVX128_LoadSource_WithOpSize could return both constructed addresses.
|
||||
auto AddressHigh = _Add(OpSize::i64Bit, Address, _Constant(16));
|
||||
_VStoreVectorMasked(OpSize::i128Bit, ElementSize, Mask.High, Data.High, AddressHigh, Invalid(), MEM_OFFSET_SXTX, 1);
|
||||
_VStoreVectorMasked(OpSize::i128Bit, ElementSize, Mask.High, Data.High, Address, _InlineConstant(16), MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
} else {
|
||||
auto Address = MakeAddress(DataOp);
|
||||
@@ -2065,9 +2075,7 @@ void OpDispatchBuilder::AVX128_VMASKMOVImpl(OpcodeArgs, IR::OpSize ElementSize,
|
||||
if (Is128Bit) {
|
||||
Result.High = LoadZeroVector(OpSize::i128Bit);
|
||||
} else {
|
||||
///< TODO: This can be cleaner if AVX128_LoadSource_WithOpSize could return both constructed addresses.
|
||||
auto AddressHigh = _Add(OpSize::i64Bit, Address, _Constant(16));
|
||||
Result.High = _VLoadVectorMasked(OpSize::i128Bit, ElementSize, Mask.High, AddressHigh, Invalid(), MEM_OFFSET_SXTX, 1);
|
||||
Result.High = _VLoadVectorMasked(OpSize::i128Bit, ElementSize, Mask.High, Address, _InlineConstant(16), MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
}
|
||||
|
||||
@@ -62,29 +62,36 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
// This instruction mostly matches ARMv8's SHA1SU1 instruction but one of the elements are flipped in an unexpected way.
|
||||
// Do all the work without it.
|
||||
Ref Result;
|
||||
if (CTX->HostFeatures.SupportsSHA) {
|
||||
// ARM SHA1 mostly matches x86 semantics, except the input and outputs are both flipped from elements 0,1,2,3 to 3,2,1,0.
|
||||
auto Src1 = SHADataShuffle(Dest);
|
||||
auto Src2 = SHADataShuffle(Src);
|
||||
|
||||
const auto ZeroRegister = LoadZeroVector(OpSize::i32Bit);
|
||||
// The result is swizzled differently than expected
|
||||
Result = SHADataShuffle(_VSha1SU1(Src1, Src2));
|
||||
} else {
|
||||
// Shift the incoming source left by a 32-bit element, inserting Zeros.
|
||||
// This could be slightly improved to use a VInsGPR with the zero register.
|
||||
const auto ZeroRegister = LoadZeroVector(OpSize::i32Bit);
|
||||
auto Src2Shift = _VExtr(OpSize::i128Bit, OpSize::i8Bit, Src, ZeroRegister, 12);
|
||||
auto Xor1 = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, Src2Shift);
|
||||
|
||||
// Shift the incoming source left by a 32-bit element, inserting Zeros.
|
||||
// This could be slightly improved to use a VInsGPR with the zero register.
|
||||
auto Src2Shift = _VExtr(OpSize::i128Bit, OpSize::i8Bit, Src, ZeroRegister, 12);
|
||||
auto Xor1 = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, Src2Shift);
|
||||
// Emulate rotate.
|
||||
auto ShiftLeftXor1 = _VShlI(OpSize::i128Bit, OpSize::i32Bit, Xor1, 1);
|
||||
auto RotatedXor1 = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeftXor1, Xor1, 31);
|
||||
|
||||
// Emulate rotate.
|
||||
auto ShiftLeftXor1 = _VShlI(OpSize::i128Bit, OpSize::i32Bit, Xor1, 1);
|
||||
auto RotatedXor1 = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeftXor1, Xor1, 31);
|
||||
// Element0 didn't get XOR'd with anything, so do it now.
|
||||
auto ExtractUpper = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, RotatedXor1, 3);
|
||||
auto XorLower = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, ExtractUpper);
|
||||
|
||||
// Element0 didn't get XOR'd with anything, so do it now.
|
||||
auto ExtractUpper = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, RotatedXor1, 3);
|
||||
auto XorLower = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, ExtractUpper);
|
||||
// Emulate rotate.
|
||||
auto ShiftLeftXorLower = _VShlI(OpSize::i128Bit, OpSize::i32Bit, XorLower, 1);
|
||||
auto RotatedXorLower = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeftXorLower, XorLower, 31);
|
||||
|
||||
// Emulate rotate.
|
||||
auto ShiftLeftXorLower = _VShlI(OpSize::i128Bit, OpSize::i32Bit, XorLower, 1);
|
||||
auto RotatedXorLower = _VUShraI(OpSize::i128Bit, OpSize::i32Bit, ShiftLeftXorLower, XorLower, 31);
|
||||
Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 0, 0, RotatedXor1, RotatedXorLower);
|
||||
}
|
||||
|
||||
auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 0, 0, RotatedXor1, RotatedXorLower);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
@@ -92,16 +99,16 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
using FnType = Ref (*)(OpDispatchBuilder&, Ref, Ref, Ref);
|
||||
|
||||
const auto f0 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref {
|
||||
const auto f0 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref { // sha1c?
|
||||
return Self._Xor(OpSize::i32Bit, Self._And(OpSize::i32Bit, B, C), Self._Andn(OpSize::i32Bit, D, B));
|
||||
};
|
||||
const auto f1 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref {
|
||||
const auto f1 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref { // sha1p with different key
|
||||
return Self._Xor(OpSize::i32Bit, Self._Xor(OpSize::i32Bit, B, C), D);
|
||||
};
|
||||
const auto f2 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref {
|
||||
const auto f2 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref { // sha1m
|
||||
return Self.BitwiseAtLeastTwo(B, C, D);
|
||||
};
|
||||
const auto f3 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref {
|
||||
const auto f3 = [](OpDispatchBuilder& Self, Ref B, Ref C, Ref D) -> Ref { // sha1p
|
||||
return Self._Xor(OpSize::i32Bit, Self._Xor(OpSize::i32Bit, B, C), D);
|
||||
};
|
||||
|
||||
@@ -119,60 +126,92 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
f3,
|
||||
};
|
||||
|
||||
const uint64_t Imm8 = Op->Src[1].Literal() & 0b11;
|
||||
const FnType Fn = fn_array[Imm8];
|
||||
auto K = _Constant(OpSize::i32Bit, k_array[Imm8]);
|
||||
|
||||
const uint64_t Imm8 = Op->Src[1].Literal() & 0b11;
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto W0E = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
|
||||
Ref Result {};
|
||||
if (CTX->HostFeatures.SupportsSHA) {
|
||||
Ref ConstantVector {};
|
||||
switch (Imm8) {
|
||||
case 0:
|
||||
ConstantVector = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K0);
|
||||
break;
|
||||
case 1:
|
||||
ConstantVector = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K1);
|
||||
break;
|
||||
case 2:
|
||||
ConstantVector = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K2);
|
||||
break;
|
||||
case 3:
|
||||
ConstantVector = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K3);
|
||||
break;
|
||||
}
|
||||
|
||||
using RoundResult = std::tuple<Ref, Ref, Ref, Ref, Ref>;
|
||||
const auto ZeroRegister = LoadZeroVector(OpSize::i32Bit);
|
||||
|
||||
const auto Round0 = [&]() -> RoundResult {
|
||||
auto A = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto B = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2);
|
||||
auto C = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
|
||||
auto D = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0);
|
||||
Ref Src1 = SHADataShuffle(Dest);
|
||||
Ref Src2 = SHADataShuffle(Src);
|
||||
Src2 = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src2, ConstantVector);
|
||||
|
||||
auto A1 =
|
||||
_Add(OpSize::i32Bit,
|
||||
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 27))), W0E), K);
|
||||
auto B1 = A;
|
||||
auto C1 = _Ror(OpSize::i32Bit, B, _Constant(OpSize::i32Bit, 2));
|
||||
auto D1 = C;
|
||||
auto E1 = D;
|
||||
switch (Imm8) {
|
||||
case 0: Result = SHADataShuffle(_VSha1C(Src1, ZeroRegister, Src2)); break;
|
||||
case 2: Result = SHADataShuffle(_VSha1M(Src1, ZeroRegister, Src2)); break;
|
||||
case 1:
|
||||
case 3: Result = SHADataShuffle(_VSha1P(Src1, ZeroRegister, Src2)); break;
|
||||
}
|
||||
} else {
|
||||
const FnType Fn = fn_array[Imm8];
|
||||
auto K = _Constant(OpSize::i32Bit, k_array[Imm8]);
|
||||
auto W0E = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
|
||||
|
||||
return {A1, B1, C1, D1, E1};
|
||||
};
|
||||
const auto Round1To3 = [&](Ref A, Ref B, Ref C, Ref D, Ref E, Ref Src, unsigned W_idx) -> RoundResult {
|
||||
// Kill W and E at the beginning
|
||||
auto W = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, W_idx);
|
||||
auto Q = _Add(OpSize::i32Bit, W, E);
|
||||
using RoundResult = std::tuple<Ref, Ref, Ref, Ref, Ref>;
|
||||
|
||||
auto ANext =
|
||||
_Add(OpSize::i32Bit,
|
||||
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 27))), Q), K);
|
||||
auto BNext = A;
|
||||
auto CNext = _Ror(OpSize::i32Bit, B, _Constant(OpSize::i32Bit, 2));
|
||||
auto DNext = C;
|
||||
auto ENext = D;
|
||||
const auto Round0 = [&]() -> RoundResult {
|
||||
auto A = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto B = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2);
|
||||
auto C = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
|
||||
auto D = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0);
|
||||
|
||||
return {ANext, BNext, CNext, DNext, ENext};
|
||||
};
|
||||
auto A1 =
|
||||
_Add(OpSize::i32Bit,
|
||||
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 27))), W0E), K);
|
||||
auto B1 = A;
|
||||
auto C1 = _Ror(OpSize::i32Bit, B, _Constant(OpSize::i32Bit, 2));
|
||||
auto D1 = C;
|
||||
auto E1 = D;
|
||||
|
||||
auto [A1, B1, C1, D1, E1] = Round0();
|
||||
auto [A2, B2, C2, D2, E2] = Round1To3(A1, B1, C1, D1, E1, Src, 2);
|
||||
auto [A3, B3, C3, D3, E3] = Round1To3(A2, B2, C2, D2, E2, Src, 1);
|
||||
auto Final = Round1To3(A3, B3, C3, D3, E3, Src, 0);
|
||||
return {A1, B1, C1, D1, E1};
|
||||
};
|
||||
const auto Round1To3 = [&](Ref A, Ref B, Ref C, Ref D, Ref E, Ref Src, unsigned W_idx) -> RoundResult {
|
||||
// Kill W and E at the beginning
|
||||
auto W = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, W_idx);
|
||||
auto Q = _Add(OpSize::i32Bit, W, E);
|
||||
|
||||
auto Dest3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, std::get<0>(Final));
|
||||
auto Dest2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, Dest3, std::get<1>(Final));
|
||||
auto Dest1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, Dest2, std::get<2>(Final));
|
||||
auto Dest0 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, Dest1, std::get<3>(Final));
|
||||
auto ANext =
|
||||
_Add(OpSize::i32Bit,
|
||||
_Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Fn(*this, B, C, D), _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 27))), Q), K);
|
||||
auto BNext = A;
|
||||
auto CNext = _Ror(OpSize::i32Bit, B, _Constant(OpSize::i32Bit, 2));
|
||||
auto DNext = C;
|
||||
auto ENext = D;
|
||||
|
||||
StoreResult(FPRClass, Op, Dest0, OpSize::iInvalid);
|
||||
return {ANext, BNext, CNext, DNext, ENext};
|
||||
};
|
||||
|
||||
auto [A1, B1, C1, D1, E1] = Round0();
|
||||
auto [A2, B2, C2, D2, E2] = Round1To3(A1, B1, C1, D1, E1, Src, 2);
|
||||
auto [A3, B3, C3, D3, E3] = Round1To3(A2, B2, C2, D2, E2, Src, 1);
|
||||
auto Final = Round1To3(A3, B3, C3, D3, E3, Src, 0);
|
||||
|
||||
auto Dest3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, std::get<0>(Final));
|
||||
auto Dest2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, Dest3, std::get<1>(Final));
|
||||
auto Dest1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, Dest2, std::get<2>(Final));
|
||||
Result = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, Dest1, std::get<3>(Final));
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
|
||||
@@ -222,19 +261,28 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto W14 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 2);
|
||||
auto W15 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
|
||||
auto W16 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0), Sigma1(W14));
|
||||
auto W17 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1), Sigma1(W15));
|
||||
auto W18 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2), Sigma1(W16));
|
||||
auto W19 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3), Sigma1(W17));
|
||||
Ref Result;
|
||||
if (CTX->HostFeatures.SupportsSHA) {
|
||||
auto Src1 = _VExtr(OpSize::i128Bit, OpSize::i32Bit, Dest, Dest, 3);
|
||||
auto DupDst = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto Src2 = _VZip2(OpSize::i128Bit, OpSize::i64Bit, DupDst, Src);
|
||||
|
||||
auto D3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, W19);
|
||||
auto D2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, D3, W18);
|
||||
auto D1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, D2, W17);
|
||||
auto D0 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, D1, W16);
|
||||
Result = _VSha256U1(Src1, Src2);
|
||||
} else {
|
||||
auto W14 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 2);
|
||||
auto W15 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
|
||||
auto W16 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0), Sigma1(W14));
|
||||
auto W17 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1), Sigma1(W15));
|
||||
auto W18 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2), Sigma1(W16));
|
||||
auto W19 = _Add(OpSize::i32Bit, _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3), Sigma1(W17));
|
||||
|
||||
StoreResult(FPRClass, Op, D0, OpSize::iInvalid);
|
||||
auto D3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, W19);
|
||||
auto D2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, D3, W18);
|
||||
auto D1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, D2, W17);
|
||||
Result = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, D1, W16);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::BitwiseAtLeastTwo(Ref A, Ref B, Ref C) {
|
||||
@@ -248,63 +296,88 @@ Ref OpDispatchBuilder::BitwiseAtLeastTwo(Ref A, Ref B, Ref C) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
const auto Ch = [this](Ref E, Ref F, Ref G) -> Ref {
|
||||
return _Xor(OpSize::i32Bit, _And(OpSize::i32Bit, E, F), _Andn(OpSize::i32Bit, G, E));
|
||||
};
|
||||
const auto Sigma0 = [this](Ref A) -> Ref {
|
||||
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 2)), A, ShiftType::ROR, 13),
|
||||
A, ShiftType::ROR, 22);
|
||||
};
|
||||
const auto Sigma1 = [this](Ref E) -> Ref {
|
||||
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, E, _Constant(OpSize::i32Bit, 6)), E, ShiftType::ROR, 11),
|
||||
E, ShiftType::ROR, 25);
|
||||
};
|
||||
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
// Hardcoded to XMM0
|
||||
auto XMM0 = LoadXMMRegister(0);
|
||||
|
||||
auto E0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 1);
|
||||
auto F0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 0);
|
||||
auto G0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
|
||||
Ref Q0 = _Add(OpSize::i32Bit, Ch(E0, F0, G0), Sigma1(E0));
|
||||
Ref Result;
|
||||
if (CTX->HostFeatures.SupportsSHA) {
|
||||
auto shuffle_abcd = [this](Ref Src1, Ref Src2) -> Ref {
|
||||
// Generates a suitable SHA256 `abcd` configuration from x86 format.
|
||||
auto Tmp = _VZip2(OpSize::i128Bit, OpSize::i64Bit, Src2, Src1);
|
||||
return _VRev64(OpSize::i128Bit, OpSize::i32Bit, Tmp);
|
||||
};
|
||||
|
||||
auto WK0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, XMM0, 0);
|
||||
Q0 = _Add(OpSize::i32Bit, Q0, WK0);
|
||||
auto shuffle_efgh = [this](Ref Src1, Ref Src2) -> Ref {
|
||||
// Generates a suitable SHA256 `efgh` configuration from x86 format.
|
||||
auto Tmp = _VZip(OpSize::i128Bit, OpSize::i64Bit, Src2, Src1);
|
||||
return _VRev64(OpSize::i128Bit, OpSize::i32Bit, Tmp);
|
||||
};
|
||||
|
||||
auto H0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0);
|
||||
Q0 = _Add(OpSize::i32Bit, Q0, H0);
|
||||
auto ABCD = shuffle_abcd(Dest, Src);
|
||||
auto EFGH = shuffle_efgh(Dest, Src);
|
||||
|
||||
auto A0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
|
||||
auto B0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 2);
|
||||
auto C0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto A1 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Q0, BitwiseAtLeastTwo(A0, B0, C0)), Sigma0(A0));
|
||||
// x86 uses only the bottom 64-bits of the key, so duplicate to match ARM64 semantics.
|
||||
auto Key = _VDupElement(OpSize::i128Bit, OpSize::i64Bit, XMM0, 0);
|
||||
|
||||
auto D0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2);
|
||||
auto E1 = _Add(OpSize::i32Bit, Q0, D0);
|
||||
auto A = _VSha256H(ABCD, EFGH, Key);
|
||||
auto B = _VSha256H2(EFGH, ABCD, Key);
|
||||
Result = shuffle_abcd(A, B);
|
||||
} else {
|
||||
const auto Ch = [this](Ref E, Ref F, Ref G) -> Ref {
|
||||
return _Xor(OpSize::i32Bit, _And(OpSize::i32Bit, E, F), _Andn(OpSize::i32Bit, G, E));
|
||||
};
|
||||
const auto Sigma0 = [this](Ref A) -> Ref {
|
||||
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, A, _Constant(OpSize::i32Bit, 2)), A, ShiftType::ROR, 13),
|
||||
A, ShiftType::ROR, 22);
|
||||
};
|
||||
const auto Sigma1 = [this](Ref E) -> Ref {
|
||||
return _XorShift(OpSize::i32Bit, _XorShift(OpSize::i32Bit, _Ror(OpSize::i32Bit, E, _Constant(OpSize::i32Bit, 6)), E, ShiftType::ROR, 11),
|
||||
E, ShiftType::ROR, 25);
|
||||
};
|
||||
|
||||
Ref Q1 = _Add(OpSize::i32Bit, Ch(E1, E0, F0), Sigma1(E1));
|
||||
auto E0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 1);
|
||||
auto F0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 0);
|
||||
auto G0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
|
||||
Ref Q0 = _Add(OpSize::i32Bit, Ch(E0, F0, G0), Sigma1(E0));
|
||||
|
||||
auto WK1 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, XMM0, 1);
|
||||
Q1 = _Add(OpSize::i32Bit, Q1, WK1);
|
||||
auto WK0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, XMM0, 0);
|
||||
Q0 = _Add(OpSize::i32Bit, Q0, WK0);
|
||||
|
||||
// Rematerialize G0. Costs a move but saves spilling, coming out ahead.
|
||||
G0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
|
||||
Q1 = _Add(OpSize::i32Bit, Q1, G0);
|
||||
auto H0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 0);
|
||||
Q0 = _Add(OpSize::i32Bit, Q0, H0);
|
||||
|
||||
auto A2 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Q1, BitwiseAtLeastTwo(A1, A0, B0)), Sigma0(A1));
|
||||
auto A0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 3);
|
||||
auto B0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Src, 2);
|
||||
auto C0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto A1 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Q0, BitwiseAtLeastTwo(A0, B0, C0)), Sigma0(A0));
|
||||
|
||||
// Rematerialize C0. As with G0.
|
||||
C0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto E2 = _Add(OpSize::i32Bit, Q1, C0);
|
||||
auto D0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 2);
|
||||
auto E1 = _Add(OpSize::i32Bit, Q0, D0);
|
||||
|
||||
auto Res3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, A2);
|
||||
auto Res2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, Res3, A1);
|
||||
auto Res1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, Res2, E2);
|
||||
auto Res0 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, Res1, E1);
|
||||
Ref Q1 = _Add(OpSize::i32Bit, Ch(E1, E0, F0), Sigma1(E1));
|
||||
|
||||
StoreResult(FPRClass, Op, Res0, OpSize::iInvalid);
|
||||
auto WK1 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, XMM0, 1);
|
||||
Q1 = _Add(OpSize::i32Bit, Q1, WK1);
|
||||
|
||||
// Rematerialize G0. Costs a move but saves spilling, coming out ahead.
|
||||
G0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 1);
|
||||
Q1 = _Add(OpSize::i32Bit, Q1, G0);
|
||||
|
||||
auto A2 = _Add(OpSize::i32Bit, _Add(OpSize::i32Bit, Q1, BitwiseAtLeastTwo(A1, A0, B0)), Sigma0(A1));
|
||||
|
||||
// Rematerialize C0. As with G0.
|
||||
C0 = _VExtractToGPR(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
auto E2 = _Add(OpSize::i32Bit, Q1, C0);
|
||||
|
||||
auto Res3 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 3, Dest, A2);
|
||||
auto Res2 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 2, Res3, A1);
|
||||
auto Res1 = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 1, Res2, E2);
|
||||
Result = _VInsGPR(OpSize::i128Bit, OpSize::i32Bit, 0, Res1, E1);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
|
||||
|
||||
@@ -9,16 +9,16 @@ constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDisp
|
||||
{0x1C, 1, &OpDispatchBuilder::PF2IWOp},
|
||||
{0x1D, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false>},
|
||||
|
||||
{0x86, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRECP, OpSize::i32Bit>},
|
||||
{0x87, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRSQRT, OpSize::i32Bit>},
|
||||
{0x86, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRECPPRECISION, OpSize::i32Bit>},
|
||||
{0x87, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RSqrt3DNowOp, false>},
|
||||
|
||||
{0x8A, 1, &OpDispatchBuilder::PFNACCOp},
|
||||
{0x8E, 1, &OpDispatchBuilder::PFPNACCOp},
|
||||
|
||||
{0x90, 1, &OpDispatchBuilder::VPFCMPOp<1>},
|
||||
{0x94, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMIN, OpSize::i32Bit>},
|
||||
{0x96, 1, &OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRECP, OpSize::i32Bit>},
|
||||
{0x97, 1, &OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRSQRT, OpSize::i32Bit>},
|
||||
{0x96, 1, &OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRECPPRECISION, OpSize::i32Bit>},
|
||||
{0x97, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RSqrt3DNowOp, true>},
|
||||
|
||||
{0x9A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFSUB, OpSize::i32Bit>},
|
||||
{0x9E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADD, OpSize::i32Bit>},
|
||||
|
||||
@@ -135,6 +135,7 @@ Ref OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateOF(IR::OpSize SrcSize, Ref Res, Ref Src1, Ref Src2, bool Sub) {
|
||||
LOGMAN_THROW_A_FMT(SrcSize >= IR::OpSize::i8Bit && SrcSize <= IR::OpSize::i64Bit, "Invalid size");
|
||||
const auto OpSize = SrcSize == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
const uint64_t SignBit = IR::OpSizeAsBits(SrcSize) - 1;
|
||||
Ref Anded = nullptr;
|
||||
|
||||
@@ -21,6 +21,11 @@ constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDis
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_66, 0), 1, &OpDispatchBuilder::SGDTOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F2, 0), 1, &OpDispatchBuilder::SGDTOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_NONE, 1), 1, &OpDispatchBuilder::SIDTOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F3, 1), 1, &OpDispatchBuilder::SIDTOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_66, 1), 1, &OpDispatchBuilder::SIDTOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F2, 1), 1, &OpDispatchBuilder::SIDTOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_NONE, 3), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F3, 3), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_66, 3), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
@@ -36,6 +41,11 @@ constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDis
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_66, 6), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F2, 6), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_NONE, 7), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F3, 7), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_66, 7), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_7, PF_F2, 7), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
|
||||
// GROUP 8
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_NONE, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTNone>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_8, PF_F3, 4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::BTOp, 1, BTAction::BTNone>},
|
||||
|
||||
@@ -5,6 +5,7 @@
|
||||
namespace FEXCore::IR {
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_TwoByteOpTable[] = {
|
||||
// Instructions
|
||||
{0x03, 1, &OpDispatchBuilder::LSLOp},
|
||||
{0x06, 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{0x07, 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{0x0B, 1, &OpDispatchBuilder::INTOp},
|
||||
@@ -19,7 +20,6 @@ constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDisp
|
||||
{0x32, 2, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{0x34, 3, &OpDispatchBuilder::UnimplementedOp},
|
||||
|
||||
{0x3F, 1, &OpDispatchBuilder::ThunkOp},
|
||||
{0x40, 16, &OpDispatchBuilder::CMOVOp},
|
||||
{0x6E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVBetweenGPR_FPR, OpDispatchBuilder::VectorOpType::MMX>},
|
||||
{0x6F, 1, &OpDispatchBuilder::MOVQMMXOp},
|
||||
@@ -143,8 +143,11 @@ constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDisp
|
||||
{0xFD, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, OpSize::i16Bit>},
|
||||
{0xFE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, OpSize::i32Bit>},
|
||||
|
||||
#ifndef _WIN32
|
||||
// FEX reserved instructions
|
||||
{0x37, 1, &OpDispatchBuilder::CallbackReturnOp},
|
||||
{0x3F, 1, &OpDispatchBuilder::ThunkOp},
|
||||
#endif
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint8_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> OpDispatch_SecondaryRepModTables[] = {
|
||||
|
||||
@@ -626,17 +626,36 @@ void OpDispatchBuilder::AVXInsertScalarFCMPOp(OpcodeArgs) {
|
||||
template void OpDispatchBuilder::AVXInsertScalarFCMPOp<OpSize::i32Bit>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::AVXInsertScalarFCMPOp<OpSize::i64Bit>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::RSqrt3DNowOp(OpcodeArgs, bool Duplicate) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
const auto ElementSize = OpSize::i32Bit;
|
||||
|
||||
Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], Size, Op->Flags);
|
||||
|
||||
// For the sqrt reciprocal in 3DNow!, if the source is negative,
|
||||
// then the result has the same sign as the source but the result is always calculated
|
||||
// as if the source was positive.
|
||||
Ref AbsSrc = _VFAbs(Size, ElementSize, Src);
|
||||
Ref PosRSqrt = _VFRSqrtPrecision(Size, ElementSize, AbsSrc);
|
||||
Ref Result = _VFCopySign(Size, ElementSize, PosRSqrt, Src);
|
||||
|
||||
if (Duplicate) {
|
||||
Result = _VDupElement(Size, ElementSize, Result, 0);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VectorUnaryOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize) {
|
||||
// In the event of a scalar operation and a vector source, then
|
||||
// we can specify the entire vector length in order to avoid
|
||||
// unnecessary sign extension on the element to be operated on.
|
||||
// In the event of a memory operand, we load the exact element size.
|
||||
const auto SrcSize = OpSizeFromSrc(Op);
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
|
||||
DeriveOp(ALUOp, IROp, _VFSqrt(SrcSize, ElementSize, Src));
|
||||
Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], Size, Op->Flags);
|
||||
|
||||
DeriveOp(ALUOp, IROp, _VFSqrt(Size, ElementSize, Src));
|
||||
StoreResult(FPRClass, Op, ALUOp, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
@@ -676,8 +695,8 @@ void OpDispatchBuilder::VectorUnaryDuplicateOp(OpcodeArgs) {
|
||||
VectorUnaryDuplicateOpImpl(Op, IROp, ElementSize);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRSQRT, OpSize::i32Bit>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRECP, OpSize::i32Bit>(OpcodeArgs);
|
||||
// TODO: there's only one instantiation of this template. Lets remove it.
|
||||
template void OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRECPPRECISION, OpSize::i32Bit>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::MOVQOp(OpcodeArgs, VectorOpType VectorType) {
|
||||
const auto SrcSize = Op->Src[0].IsGPR() ? OpSize::i128Bit : OpSizeFromSrc(Op);
|
||||
@@ -967,13 +986,17 @@ Ref OpDispatchBuilder::Single128Bit4ByteVectorShuffle(Ref Src, uint8_t Shuffle)
|
||||
// Special case element duplicate and broadcast to low or high 64-bits.
|
||||
return _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Src, Shuffle & 0b11);
|
||||
}
|
||||
|
||||
case 0b00'00'10'10: {
|
||||
// Weird reverse low elements and broadcast to each half of the register
|
||||
Ref Tmp = _VUnZip(OpSize::i128Bit, OpSize::i32Bit, Src, Src);
|
||||
Tmp = _VRev64(OpSize::i128Bit, OpSize::i32Bit, Tmp);
|
||||
return _VZip(OpSize::i128Bit, OpSize::i32Bit, Tmp, Tmp);
|
||||
}
|
||||
case 0b00'00'11'10: {
|
||||
// First element duplicated and shifted in to the top.
|
||||
auto Dup = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Src, 0);
|
||||
return _VExtr(OpSize::i128Bit, OpSize::i32Bit, Dup, Src, 2);
|
||||
}
|
||||
case 0b00'01'00'01: {
|
||||
///< Weird reversed low elements and broadcast
|
||||
Ref Tmp = _VRev64(OpSize::i128Bit, OpSize::i32Bit, Src);
|
||||
@@ -984,6 +1007,11 @@ Ref OpDispatchBuilder::Single128Bit4ByteVectorShuffle(Ref Src, uint8_t Shuffle)
|
||||
Ref Tmp = _VZip(OpSize::i128Bit, OpSize::i32Bit, Src, Src);
|
||||
return _VExtr(OpSize::i128Bit, OpSize::i8Bit, Tmp, Tmp, 4);
|
||||
}
|
||||
case 0b00'01'10'11: {
|
||||
// Inverse elements
|
||||
Ref Tmp = _VRev64(OpSize::i128Bit, OpSize::i32Bit, Src);
|
||||
return _VExtr(OpSize::i128Bit, OpSize::i32Bit, Tmp, Tmp, 2);
|
||||
}
|
||||
case 0b00'10'00'10: {
|
||||
///< Weird reversed even elements and broadcast
|
||||
Ref Tmp = _VUnZip(OpSize::i128Bit, OpSize::i32Bit, Src, Src);
|
||||
@@ -1102,6 +1130,10 @@ Ref OpDispatchBuilder::Single128Bit4ByteVectorShuffle(Ref Src, uint8_t Shuffle)
|
||||
Ref Tmp = _VZip2(OpSize::i128Bit, OpSize::i32Bit, Src, Src);
|
||||
return _VExtr(OpSize::i128Bit, OpSize::i8Bit, Tmp, Tmp, 8);
|
||||
}
|
||||
case 0b10'11'00'01: {
|
||||
// Reverse each 64-bit lane.
|
||||
return _VRev64(OpSize::i128Bit, OpSize::i32Bit, Src);
|
||||
}
|
||||
case 0b10'11'10'11: {
|
||||
///< Weird top two elements reverse and broadcast
|
||||
Ref Tmp = _VZip2(OpSize::i128Bit, OpSize::i64Bit, Src, Src);
|
||||
@@ -2072,17 +2104,28 @@ Ref OpDispatchBuilder::CVTFPR_To_GPRImpl(OpcodeArgs, Ref Src, IR::OpSize SrcElem
|
||||
// Source Element size is determined by instruction
|
||||
const auto GPRSize = OpSizeFromDst(Op);
|
||||
|
||||
if (HostRoundingMode) {
|
||||
Src = _Vector_FToI(SrcElementSize, SrcElementSize, Src, Round_Host);
|
||||
}
|
||||
Ref Converted = _Float_ToGPR_ZS(GPRSize, SrcElementSize, Src);
|
||||
if (CTX->HostFeatures.SupportsFRINTTS) {
|
||||
// When we have FRINTTS, this is a two-step process. First, we round to the
|
||||
// right integer (where _Vector_FToISized matches x86 semantics), then just
|
||||
// convert that to a GPR.
|
||||
Src = _Vector_FToISized(SrcElementSize, SrcElementSize, Src, HostRoundingMode, GPRSize);
|
||||
return _Float_ToGPR_ZS(GPRSize, SrcElementSize, Src);
|
||||
} else {
|
||||
// When we lack hardware support, we need a bit of a convoluted sequence of
|
||||
// fixups before before and after conversion to emulate x86 semantics.
|
||||
if (HostRoundingMode) {
|
||||
Src = _Vector_FToI(SrcElementSize, SrcElementSize, Src, Round_Host);
|
||||
}
|
||||
|
||||
bool Dst32 = GPRSize == OpSize::i32Bit;
|
||||
Ref MaxI = Dst32 ? _Constant(0x80000000) : _Constant(0x8000000000000000);
|
||||
Ref MaxF = LoadAndCacheNamedVectorConstant(SrcElementSize, (SrcElementSize == OpSize::i32Bit) ?
|
||||
(Dst32 ? NAMED_VECTOR_CVTMAX_F32_I32 : NAMED_VECTOR_CVTMAX_F32_I64) :
|
||||
(Dst32 ? NAMED_VECTOR_CVTMAX_F64_I32 : NAMED_VECTOR_CVTMAX_F64_I64));
|
||||
return _Select(GPRSize, SrcElementSize, CondClassType {FEXCore::IR::COND_FGT}, MaxF, Src, Converted, MaxI);
|
||||
Ref Converted = _Float_ToGPR_ZS(GPRSize, SrcElementSize, Src);
|
||||
|
||||
bool Dst32 = GPRSize == OpSize::i32Bit;
|
||||
Ref MaxI = Dst32 ? _Constant(0x80000000) : _Constant(0x8000000000000000);
|
||||
Ref MaxF = LoadAndCacheNamedVectorConstant(SrcElementSize, (SrcElementSize == OpSize::i32Bit) ?
|
||||
(Dst32 ? NAMED_VECTOR_CVTMAX_F32_I32 : NAMED_VECTOR_CVTMAX_F32_I64) :
|
||||
(Dst32 ? NAMED_VECTOR_CVTMAX_F64_I32 : NAMED_VECTOR_CVTMAX_F64_I64));
|
||||
return _Select(GPRSize, SrcElementSize, CondClassType {FEXCore::IR::COND_FGT}, MaxF, Src, Converted, MaxI);
|
||||
}
|
||||
}
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
|
||||
@@ -2137,25 +2180,39 @@ template void OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, false>(
|
||||
|
||||
Ref OpDispatchBuilder::Vector_CVT_Float_To_Int32Impl(OpcodeArgs, IR::OpSize DstSize, Ref Src, IR::OpSize SrcSize, IR::OpSize SrcElementSize,
|
||||
bool HostRoundingMode, bool ZeroUpperHalf) {
|
||||
if (HostRoundingMode) {
|
||||
Src = _Vector_FToI(SrcSize, SrcElementSize, Src, Round_Host);
|
||||
}
|
||||
|
||||
OpSize OverflowConstSize = ZeroUpperHalf && SrcElementSize == OpSize::i64Bit ? DstSize / 2 : DstSize;
|
||||
Ref MaxI = LoadAndCacheNamedVectorConstant(OverflowConstSize, NAMED_VECTOR_CVTMAX_I32);
|
||||
Ref Converted {}, Cmp {};
|
||||
if (SrcElementSize == OpSize::i64Bit) {
|
||||
Ref MaxF = LoadAndCacheNamedVectorConstant(SrcSize, NAMED_VECTOR_CVTMAX_F64_I32);
|
||||
Converted = _Vector_F64ToI32(DstSize, Src, Round_Towards_Zero, ZeroUpperHalf);
|
||||
|
||||
Cmp = _VFCMPGT(SrcSize, OpSize::i64Bit, MaxF, Src);
|
||||
Cmp = _VUShrNI(DstSize, OpSize::i64Bit, Cmp, 32);
|
||||
if (CTX->HostFeatures.SupportsFRINTTS && SrcSize != OpSize::i256Bit) {
|
||||
// If we have FRINTS, this is the usual 2-step
|
||||
Src = _Vector_FToISized(SrcSize, SrcElementSize, Src, HostRoundingMode, OpSize::i32Bit);
|
||||
Ref Dst = _Vector_FToZS(SrcSize, SrcElementSize, Src);
|
||||
if (SrcElementSize == OpSize::i32Bit) {
|
||||
// Return 32-bit result as-is
|
||||
return Dst;
|
||||
} else {
|
||||
// Down step from 64-bit ints to 32-bit ints
|
||||
return _VUShrNI(DstSize, SrcElementSize, Dst, 0);
|
||||
}
|
||||
} else {
|
||||
Ref MaxF = LoadAndCacheNamedVectorConstant(DstSize, NAMED_VECTOR_CVTMAX_F32_I32);
|
||||
Converted = _Vector_FToZS(DstSize, OpSize::i32Bit, Src);
|
||||
Cmp = _VFCMPGT(DstSize, OpSize::i32Bit, MaxF, Src);
|
||||
// Otherwise, we have to do all the fixups, but vectorized.
|
||||
if (HostRoundingMode) {
|
||||
Src = _Vector_FToI(SrcSize, SrcElementSize, Src, Round_Host);
|
||||
}
|
||||
|
||||
OpSize OverflowConstSize = ZeroUpperHalf && SrcElementSize == OpSize::i64Bit ? DstSize / 2 : DstSize;
|
||||
Ref MaxI = LoadAndCacheNamedVectorConstant(OverflowConstSize, NAMED_VECTOR_CVTMAX_I32);
|
||||
Ref Converted {}, Cmp {};
|
||||
if (SrcElementSize == OpSize::i64Bit) {
|
||||
Ref MaxF = LoadAndCacheNamedVectorConstant(SrcSize, NAMED_VECTOR_CVTMAX_F64_I32);
|
||||
Converted = _Vector_F64ToI32(DstSize, Src, Round_Towards_Zero, ZeroUpperHalf);
|
||||
|
||||
Cmp = _VFCMPGT(SrcSize, OpSize::i64Bit, MaxF, Src);
|
||||
Cmp = _VUShrNI(DstSize, OpSize::i64Bit, Cmp, 32);
|
||||
} else {
|
||||
Ref MaxF = LoadAndCacheNamedVectorConstant(DstSize, NAMED_VECTOR_CVTMAX_F32_I32);
|
||||
Converted = _Vector_FToZS(DstSize, OpSize::i32Bit, Src);
|
||||
Cmp = _VFCMPGT(DstSize, OpSize::i32Bit, MaxF, Src);
|
||||
}
|
||||
return _VBSL(DstSize, Cmp, Converted, MaxI);
|
||||
}
|
||||
return _VBSL(DstSize, Cmp, Converted, MaxI);
|
||||
}
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
|
||||
@@ -3917,6 +3974,7 @@ void OpDispatchBuilder::PTestOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VTESTOpImpl(OpSize SrcSize, IR::OpSize ElementSize, Ref Src1, Ref Src2) {
|
||||
LOGMAN_THROW_A_FMT(ElementSize >= IR::OpSize::i8Bit && ElementSize <= IR::OpSize::i64Bit, "Invalid size");
|
||||
const auto ElementSizeInBits = IR::OpSizeAsBits(ElementSize);
|
||||
const auto MaskConstant = uint64_t {1} << (ElementSizeInBits - 1);
|
||||
|
||||
@@ -4618,8 +4676,7 @@ void OpDispatchBuilder::VPBLENDDOp(OpcodeArgs) {
|
||||
return;
|
||||
}
|
||||
if (Selector == 0xFF && Is256Bit) {
|
||||
Ref Result = Is256Bit ? Src2 : _VMov(OpSize::i128Bit, Src2);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResult(FPRClass, Op, Src2, OpSize::iInvalid);
|
||||
return;
|
||||
}
|
||||
// The only bits we care about from the 8-bit immediate for 128-bit operations
|
||||
@@ -5040,7 +5097,7 @@ void OpDispatchBuilder::VPGATHER(OpcodeArgs) {
|
||||
// Only loads two 32-bit elements in to the lower 64-bits of the first destination.
|
||||
// Bits [255:65] all become zero.
|
||||
Result = _VMov(OpSize::i64Bit, Result);
|
||||
} else if (Is128Bit) {
|
||||
} else {
|
||||
Result = _VMov(OpSize::i128Bit, Result);
|
||||
}
|
||||
} else {
|
||||
|
||||
@@ -9,6 +9,7 @@ $end_info$
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/Core/Addressing.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
@@ -40,7 +41,7 @@ Ref OpDispatchBuilder::GetX87Tag(Ref Value, Ref AbridgedFTW) {
|
||||
|
||||
void OpDispatchBuilder::SetX87FTW(Ref FTW) {
|
||||
Ref X87Empty = _Constant(static_cast<uint8_t>(FPState::X87Tag::Empty));
|
||||
Ref NewAbridgedFTW;
|
||||
Ref NewAbridgedFTW {};
|
||||
|
||||
for (int i = 0; i < 8; i++) {
|
||||
Ref RegTag = _Bfe(OpSize::i32Bit, 2, i * 2, FTW);
|
||||
@@ -124,29 +125,17 @@ void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
auto zeroed_exponent = _Select(COND_EQ, absolute, zero, zero, adjusted_exponent);
|
||||
auto upper = _Or(OpSize::i64Bit, sign, zeroed_exponent);
|
||||
|
||||
Ref ConvertedData = _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, shifted);
|
||||
ConvertedData = _VInsElement(OpSize::i128Bit, OpSize::i64Bit, 1, 0, ConvertedData, _VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, upper));
|
||||
Ref ConvertedData = _VLoadTwoGPRs(shifted, upper);
|
||||
_PushStack(ConvertedData, Data, ReadWidth, false);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FST(OpcodeArgs, IR::OpSize Width) {
|
||||
// Ref Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
// FIXME: Is TSO relevant for x87?
|
||||
const auto SourceSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
AddressMode A = DecodeAddress(Op, Op->Dest, MemoryAccessType::DEFAULT, false);
|
||||
|
||||
// Index scale is a power of 2?
|
||||
LOGMAN_THROW_A_FMT(A.IndexScale > 0 && (A.IndexScale & (A.IndexScale - 1)) == 0, "Invalid index scale");
|
||||
A = SelectAddressMode(this, A, CTX->GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, false, false, Width);
|
||||
_StoreStackMem(SourceSize, Width, A.Base, A.Index, OpSize::iInvalid, A.IndexType, A.IndexScale, /*Float=*/true);
|
||||
|
||||
Ref Addr = A.Base ? A.Base : _Constant(0);
|
||||
if (A.Index) {
|
||||
Ref ScaledIndex = A.Index;
|
||||
if (A.IndexScale > 1) {
|
||||
ScaledIndex = _Lshl(A.AddrSize, ScaledIndex, _Constant(std::log2(A.IndexScale)));
|
||||
}
|
||||
Addr = _Add(A.AddrSize, Addr, ScaledIndex);
|
||||
}
|
||||
|
||||
_StoreStackMem(OpSize::i128Bit, Width, Addr, _Constant(A.Offset), /*Float=*/true);
|
||||
if (Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) {
|
||||
_PopStackDestroy();
|
||||
}
|
||||
@@ -550,8 +539,7 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
|
||||
auto low = _Constant(~0ULL);
|
||||
auto high = _Constant(0xFFFF);
|
||||
Ref Mask = _VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, low);
|
||||
Mask = _VInsGPR(OpSize::i128Bit, OpSize::i64Bit, 1, Mask, high);
|
||||
Ref Mask = _VLoadTwoGPRs(low, high);
|
||||
const auto StoreSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
for (int i = 0; i < 7; ++i) {
|
||||
Ref Reg = _LoadMem(FPRClass, OpSize::i128Bit, Mem, _Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
|
||||
@@ -103,27 +103,6 @@ void OpDispatchBuilder::FILDF64(OpcodeArgs) {
|
||||
_PushStack(ConvertedData, Data, ReadWidth, false);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FSTF64(OpcodeArgs, IR::OpSize Width) {
|
||||
AddressMode A = DecodeAddress(Op, Op->Dest, MemoryAccessType::DEFAULT, false);
|
||||
|
||||
// Index scale is a power of 2?
|
||||
LOGMAN_THROW_A_FMT(A.IndexScale > 0 && (A.IndexScale & (A.IndexScale - 1)) == 0, "Invalid index scale");
|
||||
|
||||
Ref Addr = A.Base ? A.Base : _Constant(0);
|
||||
if (A.Index) {
|
||||
Ref ScaledIndex = A.Index;
|
||||
if (A.IndexScale > 1) {
|
||||
ScaledIndex = _Lshl(A.AddrSize, ScaledIndex, _Constant(std::log2(A.IndexScale)));
|
||||
}
|
||||
Addr = _Add(A.AddrSize, Addr, ScaledIndex);
|
||||
}
|
||||
|
||||
_StoreStackMem(OpSize::i64Bit, Width, Addr, _Constant(A.Offset), /*Float=*/true);
|
||||
if (Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) {
|
||||
_PopStackDestroy();
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FISTF64(OpcodeArgs, bool Truncate) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
|
||||
@@ -67,41 +67,41 @@ std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGroupOps = [
|
||||
{OPD(TYPE_GROUP_6, PF_F2, 7), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
// GROUP 7
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 0), 1, X86InstInfo{"SGDT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 1), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 0), 1, X86InstInfo{"SGDT", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 1), 1, X86InstInfo{"SIDT", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 2), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 3), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 4), 1, X86InstInfo{"SMSW", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 6), 1, X86InstInfo{"LMSW", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 7), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_NONE, 7), 1, X86InstInfo{"INVLPG", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 0), 1, X86InstInfo{"SGDT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 1), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 0), 1, X86InstInfo{"SGDT", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 1), 1, X86InstInfo{"SIDT", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 2), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 3), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 4), 1, X86InstInfo{"SMSW", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 6), 1, X86InstInfo{"LMSW", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 7), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F3, 7), 1, X86InstInfo{"INVLPG", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_7, PF_66, 0), 1, X86InstInfo{"SGDT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 1), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 0), 1, X86InstInfo{"SGDT", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 1), 1, X86InstInfo{"SIDT", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 2), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 3), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 4), 1, X86InstInfo{"SMSW", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 6), 1, X86InstInfo{"LMSW", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 7), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_66, 7), 1, X86InstInfo{"INVLPG", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 0), 1, X86InstInfo{"SGDT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 1), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 0), 1, X86InstInfo{"SGDT", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 1), 1, X86InstInfo{"SIDT", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 2), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 3), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 4), 1, X86InstInfo{"SMSW", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 6), 1, X86InstInfo{"LMSW", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 7), 1, X86InstInfo{"", TYPE_SECOND_GROUP_MODRM, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_7, PF_F2, 7), 1, X86InstInfo{"INVLPG", TYPE_SECOND_GROUP_MODRM, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
|
||||
// GROUP 8
|
||||
{OPD(TYPE_GROUP_8, PF_NONE, 0), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
@@ -23,7 +23,7 @@ auto BaseOpsLambda = []() consteval {
|
||||
{0x01, 1, X86InstInfo{"", TYPE_GROUP_7, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
// These two load segment register data
|
||||
{0x02, 1, X86InstInfo{"LAR", TYPE_UNDEC, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x03, 1, X86InstInfo{"LSL", TYPE_UNDEC, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x03, 1, X86InstInfo{"LSL", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_MODRM | FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x04, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x05, 1, X86InstInfo{"SYSCALL", TYPE_INST, DEFAULT_SYSCALL_FLAGS, 0, nullptr}},
|
||||
{0x06, 1, X86InstInfo{"CLTS", TYPE_INST, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
@@ -260,6 +260,7 @@ auto BaseOpsLambda = []() consteval {
|
||||
{0xFE, 1, X86InstInfo{"PADDD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0xFF, 1, X86InstInfo{"UD0", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
|
||||
#ifndef _WIN32
|
||||
// FEX reserved instructions
|
||||
// Unused x86 encoding instruction.
|
||||
|
||||
@@ -267,6 +268,7 @@ auto BaseOpsLambda = []() consteval {
|
||||
|
||||
// This was originally used by VIA to jump to its alternative instruction set. Used for OP_THUNK
|
||||
{0x3F, 1, X86InstInfo{"ALTINST", TYPE_INST, FLAGS_BLOCK_END | FLAGS_NO_OVERLAY | FLAGS_SETS_RIP, 0, nullptr}},
|
||||
#endif
|
||||
};
|
||||
|
||||
GenerateTable(&Table.at(0), TwoByteOpTable, std::size(TwoByteOpTable));
|
||||
|
||||
@@ -343,14 +343,14 @@ constexpr InstFlagType FLAGS_MODRM = (1ULL << 16);
|
||||
// x87
|
||||
constexpr InstFlagType FLAGS_POP = (1ULL << 20);
|
||||
|
||||
// Whether or not the instruction has a VEX prefix for the first source operand
|
||||
constexpr InstFlagType FLAGS_VEX_1ST_SRC = (1ULL << 21);
|
||||
// Whether or not the instruction has a VEX prefix for the second source operand
|
||||
constexpr InstFlagType FLAGS_VEX_2ND_SRC = (1ULL << 22);
|
||||
// Whether or not the instruction has a VEX prefix for the destination
|
||||
constexpr InstFlagType FLAGS_VEX_DST = (1ULL << 23);
|
||||
// Whether or not the instruction has a VEX prefix for the dest, first, or second source.
|
||||
constexpr InstFlagType FLAGS_VEX_SRC_MASK = (0b11ULL << 21);
|
||||
constexpr InstFlagType FLAGS_VEX_NO_OPERAND = (0b00ULL << 21);
|
||||
constexpr InstFlagType FLAGS_VEX_DST = (0b01ULL << 21);
|
||||
constexpr InstFlagType FLAGS_VEX_1ST_SRC = (0b10ULL << 21);
|
||||
constexpr InstFlagType FLAGS_VEX_2ND_SRC = (0b11ULL << 21);
|
||||
// Whether or not the instruction has a VSIB byte
|
||||
constexpr InstFlagType FLAGS_VEX_VSIB = (1ULL << 24);
|
||||
constexpr InstFlagType FLAGS_VEX_VSIB = (1ULL << 23);
|
||||
|
||||
constexpr InstFlagType FLAGS_SIZE_DST_OFF = 58;
|
||||
constexpr InstFlagType FLAGS_SIZE_SRC_OFF = FLAGS_SIZE_DST_OFF + 3;
|
||||
@@ -440,7 +440,7 @@ struct X86InstInfo {
|
||||
}
|
||||
};
|
||||
|
||||
static_assert(std::is_trivial<X86InstInfo>::value, "X86InstInfo needs to be trivial");
|
||||
static_assert(std::is_trivially_copyable_v<X86InstInfo>);
|
||||
|
||||
constexpr size_t MAX_PRIMARY_TABLE_SIZE = 256;
|
||||
constexpr size_t MAX_SECOND_TABLE_SIZE = 256;
|
||||
|
||||
@@ -338,7 +338,7 @@ bool AOTIRCaptureCache::PostCompileCode(FEXCore::Core::InternalThreadState* Thre
|
||||
|
||||
auto LocalRIP = GuestRIP - AOTIRCacheEntry.VAFileStart;
|
||||
auto LocalStartAddr = StartAddr - AOTIRCacheEntry.VAFileStart;
|
||||
auto FileId = AOTIRCacheEntry.Entry->FileId;
|
||||
const auto& FileId = AOTIRCacheEntry.Entry->FileId;
|
||||
|
||||
// The lambda is converted to std::function. This is tricky to refactor so it doesn't allocate memory through glibc.
|
||||
// NOTE: unique_ptr must be passed as a raw pointer since std::function requires lambda captures to be copyable
|
||||
|
||||
@@ -159,7 +159,7 @@ struct NodeWrapperBase final {
|
||||
operator==(const NodeWrapperBase<Type>&, const NodeWrapperBase<Type>&) = default;
|
||||
};
|
||||
|
||||
static_assert(std::is_trivial_v<NodeWrapperBase<OrderedNode>>);
|
||||
static_assert(std::is_trivially_copyable_v<NodeWrapperBase<OrderedNode>>);
|
||||
|
||||
static_assert(sizeof(NodeWrapperBase<OrderedNode>) == sizeof(uint32_t));
|
||||
|
||||
@@ -355,7 +355,7 @@ private:
|
||||
}
|
||||
};
|
||||
|
||||
static_assert(std::is_trivial_v<OrderedNode>);
|
||||
static_assert(std::is_trivially_constructible_v<OrderedNode>);
|
||||
static_assert(std::is_trivially_copyable_v<OrderedNode>);
|
||||
static_assert(offsetof(OrderedNode, Header) == 0);
|
||||
static_assert(sizeof(OrderedNode) == (sizeof(OrderedNodeHeader) + sizeof(uint32_t)));
|
||||
@@ -439,7 +439,7 @@ struct TypeDefinition final {
|
||||
operator==(const TypeDefinition&, const TypeDefinition&) = default;
|
||||
};
|
||||
|
||||
static_assert(std::is_trivial_v<TypeDefinition>);
|
||||
static_assert(std::is_trivially_copyable_v<TypeDefinition>);
|
||||
|
||||
struct FenceType final {
|
||||
using value_type = uint8_t;
|
||||
@@ -642,7 +642,9 @@ static inline OpSize operator/(IR::OpSize Size, T Divisor) {
|
||||
}
|
||||
|
||||
static inline uint8_t NumElements(IR::OpSize RegisterSize, IR::OpSize ElementSize) {
|
||||
LOGMAN_THROW_A_FMT(RegisterSize != IR::OpSize::iInvalid && ElementSize != IR::OpSize::iInvalid, "Invalid Size");
|
||||
LOGMAN_THROW_A_FMT(RegisterSize != IR::OpSize::iInvalid && ElementSize != IR::OpSize::iInvalid && RegisterSize != IR::OpSize::iUnsized &&
|
||||
ElementSize != IR::OpSize::iUnsized,
|
||||
"Invalid Size");
|
||||
return IR::OpSizeToSize(RegisterSize) / IR::OpSizeToSize(ElementSize);
|
||||
}
|
||||
|
||||
|
||||
@@ -749,8 +749,8 @@
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"EmitValidation": [
|
||||
"Offset % IR::OpSizeToSize(RegisterSize) == 0",
|
||||
"RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i256Bit"
|
||||
"RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i256Bit",
|
||||
"Offset % IR::OpSizeToSize(RegisterSize) == 0"
|
||||
]
|
||||
},
|
||||
"VStoreNonTemporalPair OpSize:#RegisterSize, FPR:$ValueLow, FPR:$ValueHigh, GPR:$Addr, i8:$Offset": {
|
||||
@@ -761,8 +761,8 @@
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"EmitValidation": [
|
||||
"Offset % IR::OpSizeToSize(RegisterSize) == 0",
|
||||
"RegisterSize == FEXCore::IR::OpSize::i128Bit"
|
||||
"RegisterSize == FEXCore::IR::OpSize::i128Bit",
|
||||
"Offset % IR::OpSizeToSize(RegisterSize) == 0"
|
||||
]
|
||||
},
|
||||
"FPR = VLoadNonTemporal OpSize:#RegisterSize, GPR:$Addr, i8:$Offset": {
|
||||
@@ -773,8 +773,8 @@
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"EmitValidation": [
|
||||
"Offset % IR::OpSizeToSize(RegisterSize) == 0",
|
||||
"RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i256Bit"
|
||||
"RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i256Bit",
|
||||
"Offset % IR::OpSizeToSize(RegisterSize) == 0"
|
||||
]
|
||||
}
|
||||
},
|
||||
@@ -789,7 +789,7 @@
|
||||
"Dest = %Expected",
|
||||
"if (deref(%Addr) != %Expected) Dest = deref(%Addr)"
|
||||
],
|
||||
|
||||
"TiedSource": 0,
|
||||
"DestSize": "Size",
|
||||
"ImplicitFlagClobber": true,
|
||||
"EmitValidation": [
|
||||
@@ -1507,7 +1507,8 @@
|
||||
"DestSize": "Size",
|
||||
"TiedSource": 0,
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit",
|
||||
"(Width + lsb) <= IR::OpSizeAsBits(Size)"
|
||||
]
|
||||
},
|
||||
"GPR = Bfxil OpSize:#Size, u8:$Width, u8:$lsb, GPR:$Dest, GPR:$Src": {
|
||||
@@ -1519,7 +1520,8 @@
|
||||
"DestSize": "Size",
|
||||
"TiedSource": 0,
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit",
|
||||
"(Width + lsb) <= IR::OpSizeAsBits(Size)"
|
||||
]
|
||||
},
|
||||
"GPR = Bfe OpSize:#Size, u8:$Width, u8:$lsb, GPR:$Src": {
|
||||
@@ -1529,7 +1531,8 @@
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit",
|
||||
"(Width + lsb) <= IR::OpSizeAsBits(Size)"
|
||||
]
|
||||
},
|
||||
"GPR = Sbfe OpSize:#Size, u8:$Width, u8:$lsb, GPR:$Src": {
|
||||
@@ -1539,7 +1542,8 @@
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit",
|
||||
"(Width + lsb) <= IR::OpSizeAsBits(Size)"
|
||||
]
|
||||
},
|
||||
"GPR = NZCVSelect OpSize:#ResultSize, CondClass:$Cond, GPR:$TrueVal, GPR:$FalseVal": {
|
||||
@@ -1881,6 +1885,12 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VFCopySign OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Desc": ["Returns a vector where each element has has the magniture of each corresponding element in vector1 and the sign of vector 2."],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize",
|
||||
"TiedSource": 0
|
||||
}
|
||||
},
|
||||
"Vector": {
|
||||
@@ -1967,9 +1977,25 @@
|
||||
},
|
||||
|
||||
"FPR = VFRecp OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": [
|
||||
"Reciprocal value - matches the precision required by the x86 spec.",
|
||||
"It has a relative error of at most 1.5 * 2^-12"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFRecpPrecision OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": [
|
||||
"Similar to VFRecp but carrying more precision for 3DNow!",
|
||||
"It provides at least 14 bits precision, with a relative error of at most 2^-14"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize",
|
||||
"EmitValidation": [
|
||||
"RegisterSize == FEXCore::IR::OpSize::i64Bit || RegisterSize == FEXCore::IR::OpSize::i32Bit",
|
||||
"ElementSize == FEXCore::IR::OpSize::i32Bit"
|
||||
]
|
||||
},
|
||||
|
||||
"FPR = VFSqrt OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"DestSize": "RegisterSize",
|
||||
@@ -1977,9 +2003,26 @@
|
||||
},
|
||||
|
||||
"FPR = VFRSqrt OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": [
|
||||
"Reciprocal Square Root - matches the precision required by the x86 spec.",
|
||||
"It has a relative error of at most 1.5 * 2^-12"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VFRSqrtPrecision OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": [
|
||||
"Similar to VFRSqrt but carrying more precision for 3DNow!",
|
||||
"It provides at least 15 bits precision, with a relative error of at most 2^-15"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize",
|
||||
"EmitValidation": [
|
||||
"RegisterSize == FEXCore::IR::OpSize::i64Bit || RegisterSize == FEXCore::IR::OpSize::i32Bit",
|
||||
"ElementSize == FEXCore::IR::OpSize::i32Bit"
|
||||
]
|
||||
},
|
||||
|
||||
"FPR = VCMPEQZ OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
@@ -2008,29 +2051,49 @@
|
||||
"FPR = VShlI OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
"ElementSize": "ElementSize",
|
||||
"EmitValidation": [
|
||||
"ElementSize >= FEXCore::IR::OpSize::i8Bit && ElementSize <= FEXCore::IR::OpSize::i64Bit",
|
||||
"BitShift > 0"
|
||||
]
|
||||
},
|
||||
"FPR = VUShrI OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
"ElementSize": "ElementSize",
|
||||
"EmitValidation": [
|
||||
"ElementSize >= FEXCore::IR::OpSize::i8Bit && ElementSize <= FEXCore::IR::OpSize::i64Bit",
|
||||
"BitShift > 0"
|
||||
]
|
||||
},
|
||||
"FPR = VUShraI OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$DestVector, FPR:$Vector, u8:$BitShift": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
"ElementSize": "ElementSize",
|
||||
"EmitValidation": [
|
||||
"ElementSize >= FEXCore::IR::OpSize::i8Bit && ElementSize <= FEXCore::IR::OpSize::i64Bit",
|
||||
"BitShift > 0 && BitShift <= IR::OpSizeAsBits(ElementSize)"
|
||||
]
|
||||
},
|
||||
"FPR = VSShrI OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
"ElementSize": "ElementSize",
|
||||
"EmitValidation": [
|
||||
"ElementSize >= FEXCore::IR::OpSize::i8Bit && ElementSize <= FEXCore::IR::OpSize::i64Bit",
|
||||
"BitShift > 0"
|
||||
]
|
||||
},
|
||||
|
||||
"FPR = VUShrNI OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, u8:$BitShift": {
|
||||
"TiedSource": 0,
|
||||
"Desc": "Unsigned shifts right each element and then narrows to the next lower element size",
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize >> 1"
|
||||
"ElementSize": "ElementSize >> 1",
|
||||
"EmitValidation": [
|
||||
"ElementSize >= FEXCore::IR::OpSize::i16Bit && ElementSize <= FEXCore::IR::OpSize::i64Bit",
|
||||
"BitShift <= IR::OpSizeAsBits(ElementSize)"
|
||||
]
|
||||
},
|
||||
|
||||
"FPR = VUShrNI2 OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$VectorLower, FPR:$VectorUpper, u8:$BitShift": {
|
||||
@@ -2039,7 +2102,11 @@
|
||||
"Inserts results in to the high elements of the first argument"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize >> 1"
|
||||
"ElementSize": "ElementSize >> 1",
|
||||
"EmitValidation": [
|
||||
"ElementSize >= FEXCore::IR::OpSize::i16Bit && ElementSize <= FEXCore::IR::OpSize::i64Bit",
|
||||
"BitShift > 0 && BitShift <= IR::OpSizeAsBits(ElementSize)"
|
||||
]
|
||||
},
|
||||
"FPR = VSXTL OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": "Sign extends elements from the source element size to the next size up",
|
||||
@@ -2272,11 +2339,13 @@
|
||||
|
||||
"FPR = VFMin OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
"ElementSize": "ElementSize",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VFMax OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
"ElementSize": "ElementSize",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VMul OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
@@ -2378,6 +2447,7 @@
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VInsGPR OpSize:#RegisterSize, OpSize:#ElementSize, u8:$DestIdx, FPR:$DestVector, GPR:$Src": {
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
@@ -2389,6 +2459,7 @@
|
||||
"TmpVector <RegisterSize *2> = concat(Upper:Lower)",
|
||||
"Dest = TmpVector >> (ElementSize * Index * 8); // Or can be thought of `concat(&TmpVector[Index], i128)`"
|
||||
],
|
||||
"TiedSource": 1,
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
@@ -2468,6 +2539,7 @@
|
||||
"If the bit in the field is 1 then the corresponding bit is pulled from VectorTrue",
|
||||
"If the bit in the field is 0 then the corresponding bit is pulled from VectorFalse"
|
||||
],
|
||||
"TiedSource": 0,
|
||||
"DestSize": "RegisterSize"
|
||||
},
|
||||
|
||||
@@ -2551,6 +2623,12 @@
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VLoadTwoGPRs GPR:$Lower, GPR:$Upper": {
|
||||
"Desc": ["Moves two 64-bit registers to a vector register optimally"],
|
||||
"DestSize": "OpSize::i128Bit",
|
||||
"ElementSize": "OpSize::i64Bit"
|
||||
},
|
||||
|
||||
"FPR = Float_FromGPR_S OpSize:#DstElementSize, OpSize:$SrcElementSize, GPR:$Src": {
|
||||
"Desc": ["Scalar op: Converts signed GPR to Scalar float",
|
||||
"Zeroes the upper bits of the vector register"
|
||||
@@ -2619,6 +2697,14 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = Vector_FToISized OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector, i1:$HostRound, OpSize:$IntSize": {
|
||||
"Desc": ["Vector op: Rounds float to sized integral",
|
||||
"Either host rounding or round-to-zero",
|
||||
"Rounding mode determined by argument"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = Vector_F64ToI32 OpSize:#RegisterSize, FPR:$Vector, RoundType:$Round, i1:$EnsureZeroUpperHalf": {
|
||||
"Desc": ["Vector op: Rounds 64-bit float to 32-bit integral with round mode",
|
||||
"Matches CVTPD2DQ/CVTTPD2DQ behaviour"
|
||||
@@ -2656,10 +2742,45 @@
|
||||
"Desc": "Does vector scalar SHA1H instruction",
|
||||
"DestSize": "FEXCore::IR::OpSize::i32Bit"
|
||||
},
|
||||
"FPR = VSha1C FPR:$Src1, FPR:$Src2, FPR:$Src3": {
|
||||
"Desc": "Does vector SHA1C instruction",
|
||||
"DestSize": "FEXCore::IR::OpSize::i128Bit",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VSha1M FPR:$Src1, FPR:$Src2, FPR:$Src3": {
|
||||
"Desc": "Does vector SHA1M instruction",
|
||||
"DestSize": "FEXCore::IR::OpSize::i128Bit",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VSha1P FPR:$Src1, FPR:$Src2, FPR:$Src3": {
|
||||
"Desc": "Does vector SHA1P instruction",
|
||||
"DestSize": "FEXCore::IR::OpSize::i128Bit",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VSha1SU1 FPR:$Src1, FPR:$Src2": {
|
||||
"Desc": "Does vector scalar SHA1H instruction",
|
||||
"DestSize": "FEXCore::IR::OpSize::i128Bit",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VSha256U0 FPR:$Src1, FPR:$Src2": {
|
||||
"Desc": "Does vector scalar VSha256U0 instruction",
|
||||
"DestSize": "FEXCore::IR::OpSize::i128Bit",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VSha256U1 FPR:$Src1, FPR:$Src2": {
|
||||
"Desc": "Does vector scalar VSha256U1 instruction",
|
||||
"DestSize": "FEXCore::IR::OpSize::i128Bit"
|
||||
},
|
||||
"FPR = VSha256H FPR:$Src1, FPR:$Src2, FPR:$Src3": {
|
||||
"Desc": "Does vector scalar VSha256H instruction",
|
||||
"DestSize": "FEXCore::IR::OpSize::i128Bit",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VSha256H2 FPR:$Src1, FPR:$Src2, FPR:$Src3": {
|
||||
"Desc": "Does vector scalar VSha256H2 instruction",
|
||||
"DestSize": "FEXCore::IR::OpSize::i128Bit",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"GPR = CRC32 GPR:$Src1, GPR:$Src2, OpSize:$SrcSize": {
|
||||
"Desc": ["CRC32 using polynomial 0x1EDC6F41"
|
||||
],
|
||||
@@ -2782,7 +2903,7 @@
|
||||
"HasSideEffects": true,
|
||||
"X87": true
|
||||
},
|
||||
"StoreStackMem OpSize:$SourceSize, OpSize:$StoreSize, GPR:$Addr, GPR:$Offset, i1:$Float": {
|
||||
"StoreStackMem OpSize:$SourceSize, OpSize:$StoreSize, GPR:$Addr, GPR:$Offset, OpSize:$Align, MemOffsetType:$OffsetType, u8:$OffsetScale, i1:$Float": {
|
||||
"Desc": [
|
||||
"Takes the top value off the x87 stack and stores it to memory.",
|
||||
"SourceSize is 128bit for F80 values, 64-bit for low precision.",
|
||||
|
||||
@@ -37,7 +37,7 @@ static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView* IR, uint64_t Arg) {
|
||||
*out << "#0x" << std::hex << Arg;
|
||||
*out << "#0x" << std::hex << Arg << std::dec;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
|
||||
@@ -60,6 +60,7 @@ public:
|
||||
#include <FEXCore/IR/IRDefines.inc>
|
||||
IRPair<IROp_Constant> _Constant(IR::OpSize Size, uint64_t Constant) {
|
||||
auto Op = AllocateOp<IROp_Constant, IROps::OP_CONSTANT>();
|
||||
LOGMAN_THROW_A_FMT(Size >= IR::OpSize::i8Bit && Size <= IR::OpSize::i64Bit, "Invalid size");
|
||||
uint64_t Mask = ~0ULL >> (64 - IR::OpSizeAsBits(Size));
|
||||
Op.first->Constant = (Constant & Mask);
|
||||
Op.first->Header.Size = Size;
|
||||
@@ -356,10 +357,10 @@ protected:
|
||||
// These could be combined with a little bit of work to be more efficient with memory usage. Isn't a big deal
|
||||
DualIntrusiveAllocatorThreadPool DualListData;
|
||||
|
||||
Ref InvalidNode;
|
||||
Ref InvalidNode {};
|
||||
Ref CurrentCodeBlock {};
|
||||
fextl::vector<Ref> CodeBlocks;
|
||||
uint64_t Entry;
|
||||
uint64_t Entry {};
|
||||
};
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -98,11 +98,11 @@ protected:
|
||||
DualIntrusiveAllocator(size_t Size)
|
||||
: MemorySize {Size} {}
|
||||
|
||||
uintptr_t Data;
|
||||
uintptr_t List;
|
||||
uintptr_t Data {};
|
||||
uintptr_t List {};
|
||||
size_t DataCurrentOffset {0};
|
||||
size_t ListCurrentOffset {0};
|
||||
size_t MemorySize;
|
||||
size_t MemorySize {};
|
||||
};
|
||||
|
||||
class DualIntrusiveAllocatorMalloc final : public DualIntrusiveAllocator {
|
||||
|
||||
@@ -70,7 +70,7 @@ void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl* ctx) {
|
||||
FEX_CONFIG_OPT(DisablePasses, O0);
|
||||
|
||||
if (!DisablePasses()) {
|
||||
InsertPass(CreateX87StackOptimizationPass(ctx->HostFeatures));
|
||||
InsertPass(CreateX87StackOptimizationPass(ctx->HostFeatures, ctx->GetGPROpSize()));
|
||||
InsertPass(CreateConstProp(ctx->HostFeatures.SupportsTSOImm9, &ctx->CPUID));
|
||||
InsertPass(CreateDeadFlagCalculationEliminination());
|
||||
}
|
||||
|
||||
@@ -80,7 +80,7 @@ public:
|
||||
void Finalize();
|
||||
|
||||
protected:
|
||||
FEXCore::HLE::SyscallHandler* SyscallHandler;
|
||||
FEXCore::HLE::SyscallHandler* SyscallHandler {};
|
||||
|
||||
private:
|
||||
using PassArrayType = fextl::vector<fextl::unique_ptr<Pass>>;
|
||||
|
||||
@@ -20,7 +20,7 @@ class RegisterAllocationData;
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool SupportsTSOImm9, const FEXCore::CPUIDEmu* CPUID);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadFlagCalculationEliminination();
|
||||
fextl::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass();
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateX87StackOptimizationPass(const FEXCore::HostFeatures&);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateX87StackOptimizationPass(const FEXCore::HostFeatures&, OpSize GPROpSize);
|
||||
|
||||
namespace Validation {
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateIRValidation();
|
||||
|
||||
@@ -24,6 +24,7 @@ $end_info$
|
||||
namespace FEXCore::IR {
|
||||
|
||||
uint64_t getMask(IROp_Header* Op) {
|
||||
LOGMAN_THROW_A_FMT(Op->Size >= IR::OpSize::i8Bit && Op->Size <= IR::OpSize::i64Bit, "Invalid mask size");
|
||||
uint64_t NumBits = IR::OpSizeAsBits(Op->Size);
|
||||
return (~0ULL) >> (64 - NumBits);
|
||||
}
|
||||
@@ -92,8 +93,8 @@ private:
|
||||
return InlineIf(IREmit, CurrentIR, CodeNode, IROp, Index, Filter);
|
||||
}
|
||||
|
||||
void InlineMemImmediate(IREmitter* IREmit, const IRListView& IR, Ref CodeNode, IROp_Header* IROp, OrderedNodeWrapper Offset,
|
||||
MemOffsetType OffsetType, const size_t Offset_Index, uint8_t& OffsetScale, bool TSO) {
|
||||
void InlineMemImmediate(IREmitter* IREmit, const IRListView& IR, Ref CodeNode, IR::RegisterClassType RegisterClass, IROp_Header* IROp,
|
||||
OrderedNodeWrapper Offset, MemOffsetType OffsetType, const size_t Offset_Index, uint8_t& OffsetScale, bool TSO) {
|
||||
uint64_t Imm {};
|
||||
if (OffsetType != MEM_OFFSET_SXTX || !IREmit->IsValueConstant(Offset, &Imm)) {
|
||||
return;
|
||||
@@ -107,6 +108,9 @@ private:
|
||||
IsSIMM9 &= (SupportsTSOImm9 || !TSO);
|
||||
|
||||
// Extended offsets for regular loadstore only.
|
||||
LOGMAN_THROW_A_FMT(IROp->Size >= IR::OpSize::i8Bit && IROp->Size <= (RegisterClass == GPRClass ? IR::OpSize::i64Bit : IR::OpSize::i256Bit),
|
||||
"Invalid "
|
||||
"size");
|
||||
bool IsExtended = (Imm & (IR::OpSizeToSize(IROp->Size) - 1)) == 0 && Imm / IR::OpSizeToSize(IROp->Size) <= 4095;
|
||||
IsExtended &= !TSO;
|
||||
|
||||
@@ -419,6 +423,7 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
uint64_t Constant;
|
||||
if (IREmit->IsValueConstant(Op->Src, &Constant)) {
|
||||
LOGMAN_THROW_A_FMT(IROp->Size >= IR::OpSize::i8Bit && IROp->Size <= IR::OpSize::i64Bit, "Invalid size");
|
||||
// SBFE of a constant can be converted to a constant.
|
||||
uint64_t SourceMask = Op->Width == 64 ? ~0ULL : ((1ULL << Op->Width) - 1);
|
||||
uint64_t DestSizeInBits = IR::OpSizeAsBits(IROp->Size);
|
||||
@@ -603,11 +608,41 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
|
||||
case OP_ADC:
|
||||
case OP_ADCWITHFLAGS:
|
||||
case OP_STORECONTEXT:
|
||||
case OP_RMIFNZCV: {
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
break;
|
||||
}
|
||||
case OP_STORECONTEXT: {
|
||||
// For i128Bit, we won't see a normal Constant to inline, but as a special
|
||||
// case we can replace with a 2x64-bit store which can use inline zeroes.
|
||||
if (IROp->Size == OpSize::i128Bit) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
auto Header = IREmit->GetOpHeader(IROp->Args[0]);
|
||||
const auto MAX_STP_OFFSET = (252 * 4);
|
||||
|
||||
if (Op->Offset <= MAX_STP_OFFSET && Header->Op == OP_LOADNAMEDVECTORCONSTANT) {
|
||||
auto Const = Header->C<IR::IROp_LoadNamedVectorConstant>();
|
||||
|
||||
if (Const->Constant == IR::NamedVectorConstant::NAMED_VECTOR_ZERO) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
Ref Zero = IREmit->_Constant(0);
|
||||
Ref STP = IREmit->_StoreContextPair(IR::OpSize::i64Bit, GPRClass, Zero, Zero, Op->Offset);
|
||||
IREmit->Remove(CodeNode);
|
||||
|
||||
// XXX: This works around InlineConstant not having an associated
|
||||
// register class, else we'd just do InlineConstant above.
|
||||
Ref InlineZero = IREmit->_InlineConstant(0);
|
||||
IREmit->ReplaceNodeArgument(STP, 0, InlineZero);
|
||||
IREmit->ReplaceNodeArgument(STP, 1, InlineZero);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_CONDADDNZCV:
|
||||
case OP_CONDSUBNZCV: {
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
@@ -661,27 +696,35 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
|
||||
case OP_LOADMEM: {
|
||||
auto Op = IROp->CW<IR::IROp_LoadMem>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, Op->Class, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
break;
|
||||
}
|
||||
case OP_STOREMEM: {
|
||||
auto Op = IROp->CW<IR::IROp_StoreMem>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, Op->Class, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, Op->Value_Index);
|
||||
break;
|
||||
}
|
||||
case OP_PREFETCH: {
|
||||
auto Op = IROp->CW<IR::IROp_Prefetch>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, GPRClass, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
break;
|
||||
}
|
||||
case OP_LOADMEMTSO: {
|
||||
auto Op = IROp->CW<IR::IROp_LoadMemTSO>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, true);
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, Op->Class, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, true);
|
||||
break;
|
||||
}
|
||||
case OP_STOREMEMTSO: {
|
||||
auto Op = IROp->CW<IR::IROp_StoreMemTSO>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, true);
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, Op->Class, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, true);
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, Op->Value_Index);
|
||||
break;
|
||||
}
|
||||
case OP_STOREMEMPAIR: {
|
||||
auto Op = IROp->CW<IR::IROp_StoreMemPair>();
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, Op->Value1_Index);
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, Op->Value2_Index);
|
||||
break;
|
||||
}
|
||||
case OP_MEMCPY: {
|
||||
@@ -692,6 +735,7 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
case OP_MEMSET: {
|
||||
auto Op = IROp->CW<IR::IROp_MemSet>();
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, Op->Direction_Index);
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, Op->Value_Index);
|
||||
break;
|
||||
}
|
||||
|
||||
|
||||
@@ -27,7 +27,7 @@ private:
|
||||
};
|
||||
|
||||
IRDumper::IRDumper() {
|
||||
const auto DumpIRStr = DumpIR();
|
||||
const auto& DumpIRStr = DumpIR();
|
||||
if (DumpIRStr == "stderr" || DumpIRStr == "stdout" || DumpIRStr == "no") {
|
||||
// Intentionally do nothing
|
||||
} else if (DumpIRStr == "server") {
|
||||
|
||||
@@ -25,8 +25,8 @@ public:
|
||||
|
||||
private:
|
||||
|
||||
BitSet<uint64_t> NodeIsLive;
|
||||
OrderedNode* EntryBlock;
|
||||
BitSet<uint64_t> NodeIsLive {};
|
||||
OrderedNode* EntryBlock {};
|
||||
fextl::unordered_map<IR::NodeID, BlockInfo> OffsetToBlockMap;
|
||||
size_t MaxNodes {};
|
||||
|
||||
|
||||
@@ -248,9 +248,9 @@ private:
|
||||
return nullptr;
|
||||
};
|
||||
|
||||
PhysicalRegister DecodeSRAReg(const IROp_Header* IROp, Ref Node) {
|
||||
RegisterClassType Class;
|
||||
uint8_t Reg;
|
||||
PhysicalRegister DecodeSRAReg(const IROp_Header* IROp) {
|
||||
RegisterClassType Class {};
|
||||
uint8_t Reg {};
|
||||
|
||||
uint8_t FlagOffset = Classes[GPRFixedClass.Val].Count - 2;
|
||||
|
||||
@@ -260,7 +260,6 @@ private:
|
||||
Class = Op->Class;
|
||||
Reg = Op->Reg;
|
||||
} else if (IROp->Op == OP_STOREREGISTER) {
|
||||
LOGMAN_THROW_A_FMT(IROp->Op == OP_STOREREGISTER, "node is SRA");
|
||||
const IROp_StoreRegister* Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
|
||||
Class = Op->Class;
|
||||
@@ -520,7 +519,7 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
// each register, used below. Since we initialized Class->Available,
|
||||
// RegToSSA is otherwise undefined so we can stash our temps there.
|
||||
if (auto Node = DecodeSRANode(IROp, CodeNode); Node != nullptr) {
|
||||
auto Reg = DecodeSRAReg(IROp, Node);
|
||||
auto Reg = DecodeSRAReg(IROp);
|
||||
|
||||
PreferredReg[IR->GetID(Node).Value] = Reg;
|
||||
GetClass(Reg)->RegToSSA[Reg.Reg] = CodeNode;
|
||||
@@ -559,7 +558,7 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
// assumed by the forward pass. Do not reset it.
|
||||
|
||||
// SourcesNextUses is read backwards, this tracks the index
|
||||
unsigned SourceIndex = SourcesNextUses.size();
|
||||
int64_t SourceIndex = SourcesNextUses.size();
|
||||
|
||||
// Forward pass: Assign registers, spilling as we go.
|
||||
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
|
||||
@@ -567,7 +566,7 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
|
||||
// Static registers must be consistent at SRA load/store. Evict to ensure.
|
||||
if (auto Node = DecodeSRANode(IROp, CodeNode); Node != nullptr) {
|
||||
auto Reg = DecodeSRAReg(IROp, Node);
|
||||
auto Reg = DecodeSRAReg(IROp);
|
||||
RegisterClass* Class = &Classes[Reg.Class];
|
||||
|
||||
if (!(Class->Available & (1u << Reg.Reg))) {
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "FEXCore/Utils/LogManager.h"
|
||||
#include "Interface/Core/Interpreter/Fallbacks/FallbackOpHandler.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
@@ -5,8 +6,9 @@
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "FEXCore/Utils/Profiler.h"
|
||||
#include "FEXCore/Utils/MathUtils.h"
|
||||
#include "FEXCore/Core/HostFeatures.h"
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/Addressing.h"
|
||||
|
||||
#include <array>
|
||||
#include <cstddef>
|
||||
@@ -20,7 +22,7 @@
|
||||
// and apply the operations in a block of code. Once the block finishes, we emit the necessary operations
|
||||
// that we recorded onto the virtual stack. This allows us to save a lot of code movement
|
||||
// to and from stack registers, top management and valid flags. It also allows us to
|
||||
// perform memcpy optimizations like the one performed in STORESTACKMEMORY.
|
||||
// perform memcpy optimizations like the one performed in STORESTACKMEM.
|
||||
//
|
||||
// By default we run on the fast path - i.e. we assume all values are in the stack and we have a complete
|
||||
// stack overview. However, if we encounter a value that's not in the virtual stack - maybe it was added
|
||||
@@ -147,8 +149,9 @@ private:
|
||||
|
||||
class X87StackOptimization final : public Pass {
|
||||
public:
|
||||
X87StackOptimization(const FEXCore::HostFeatures& Features)
|
||||
: Features(Features) {
|
||||
X87StackOptimization(const FEXCore::HostFeatures& Features, OpSize GPROpSize)
|
||||
: Features(Features)
|
||||
, GPROpSize(GPROpSize) {
|
||||
FEX_CONFIG_OPT(ReducedPrecision, X87REDUCEDPRECISION);
|
||||
ReducedPrecisionMode = ReducedPrecision;
|
||||
}
|
||||
@@ -156,11 +159,97 @@ public:
|
||||
|
||||
private:
|
||||
const FEXCore::HostFeatures& Features;
|
||||
const OpSize GPROpSize;
|
||||
bool ReducedPrecisionMode;
|
||||
|
||||
// Helpers
|
||||
Ref RotateRight8(uint32_t V, Ref Amount);
|
||||
|
||||
void F80SplitStore_Helper(const IROp_StoreStackMem* Op, Ref StackNode) {
|
||||
Ref AddrNode = IR->GetNode(Op->Addr);
|
||||
Ref Offset = IR->GetNode(Op->Offset);
|
||||
OpSize Align = Op->Align;
|
||||
MemOffsetType OffsetType = Op->OffsetType;
|
||||
uint8_t OffsetScale = Op->OffsetScale;
|
||||
|
||||
IREmit->_StoreMem(FPRClass, OpSize::i64Bit, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
auto Upper = IREmit->_VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, StackNode, 1);
|
||||
|
||||
// Store the Upper part of the register (the remaining 2 bytes) into memory.
|
||||
AddressMode A {.Base = AddrNode,
|
||||
.Index = Op->Offset.IsInvalid() ? nullptr : Offset,
|
||||
.IndexType = MEM_OFFSET_SXTX,
|
||||
.IndexScale = OffsetScale,
|
||||
.Offset = 8,
|
||||
.AddrSize = OpSize::i64Bit};
|
||||
A = SelectAddressMode(IREmit, A, GPROpSize, Features.SupportsTSOImm9, false, false, OpSize::i16Bit);
|
||||
IREmit->_StoreMem(GPRClass, OpSize::i16Bit, Upper, A.Base, A.Index, OpSize::i64Bit, MEM_OFFSET_SXTX, A.IndexScale);
|
||||
}
|
||||
|
||||
void StoreStackMem_Helper(const IROp_StoreStackMem* Op, Ref StackNode) {
|
||||
Ref AddrNode = IR->GetNode(Op->Addr);
|
||||
Ref Offset = IR->GetNode(Op->Offset);
|
||||
OpSize Align = Op->Align;
|
||||
MemOffsetType OffsetType = Op->OffsetType;
|
||||
uint8_t OffsetScale = Op->OffsetScale;
|
||||
|
||||
// Normal Precision Mode
|
||||
switch (Op->StoreSize) {
|
||||
case OpSize::i32Bit:
|
||||
case OpSize::i64Bit: {
|
||||
StackNode = IREmit->_F80CVT(Op->StoreSize, StackNode);
|
||||
IREmit->_StoreMem(FPRClass, Op->StoreSize, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
break;
|
||||
}
|
||||
|
||||
case OpSize::f80Bit: {
|
||||
if (Features.SupportsSVE128 || Features.SupportsSVE256) {
|
||||
AddressMode A {.Base = AddrNode,
|
||||
.Index = Op->Offset.IsInvalid() ? nullptr : Offset,
|
||||
.IndexType = MEM_OFFSET_SXTX,
|
||||
.IndexScale = OffsetScale,
|
||||
.AddrSize = OpSize::i64Bit};
|
||||
AddrNode = LoadEffectiveAddress(IREmit, A, GPROpSize, false);
|
||||
IREmit->_StoreMemX87SVEOptPredicate(OpSize::i128Bit, OpSize::i16Bit, StackNode, AddrNode);
|
||||
} else { // 80bit requires split-store
|
||||
F80SplitStore_Helper(Op, StackNode);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: ERROR_AND_DIE_FMT("Unsupported x87 size");
|
||||
}
|
||||
}
|
||||
|
||||
// Performs a store to memory from a value the stack passed in as StackNode.
|
||||
// This is the version dealing with the reduced precision case.
|
||||
void StoreStackMem_Reduced_Helper(const IROp_StoreStackMem* Op, Ref StackNode) {
|
||||
Ref AddrNode = IR->GetNode(Op->Addr);
|
||||
Ref Offset = IR->GetNode(Op->Offset);
|
||||
OpSize Align = Op->Align;
|
||||
MemOffsetType OffsetType = Op->OffsetType;
|
||||
uint8_t OffsetScale = Op->OffsetScale;
|
||||
|
||||
switch (Op->StoreSize) {
|
||||
case OpSize::i32Bit: {
|
||||
StackNode = IREmit->_Float_FToF(OpSize::i32Bit, OpSize::i64Bit, StackNode);
|
||||
[[fallthrough]];
|
||||
}
|
||||
case OpSize::i64Bit: {
|
||||
IREmit->_StoreMem(FPRClass, Op->StoreSize, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
break;
|
||||
}
|
||||
|
||||
// 80bit requires split-store
|
||||
case OpSize::f80Bit: {
|
||||
StackNode = IREmit->_F80CVTTo(StackNode, OpSize::i64Bit);
|
||||
F80SplitStore_Helper(Op, StackNode);
|
||||
break;
|
||||
}
|
||||
default: ERROR_AND_DIE_FMT("Unsupported x87 size");
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
// Helper to check if a Ref is a Zero constant
|
||||
bool IsZero(Ref Node) {
|
||||
auto Header = IR->GetOp<IR::IROp_Header>(Node);
|
||||
@@ -172,7 +261,6 @@ private:
|
||||
return Const->Constant == 0;
|
||||
}
|
||||
|
||||
|
||||
// Handles a Unary operation.
|
||||
// Takes the op we are handling, the Node for the reduced precision case and the node for the normal case.
|
||||
// Depending on the type of Op64, we might need to pass a couple of extra constant arguments, this happens
|
||||
@@ -242,12 +330,12 @@ private:
|
||||
|
||||
// Cache for Constants
|
||||
// ConstantPoll[i] has IREmit->_Constant(i);
|
||||
std::array<Ref, 8> ConstantPool;
|
||||
std::array<Ref, 8> ConstantPool {};
|
||||
Ref GetConstant(ssize_t Offset);
|
||||
|
||||
// Cached value for Top
|
||||
// If slowpath is false, then TopCache is nullptr.
|
||||
std::array<Ref, 8> TopOffsetCache;
|
||||
std::array<Ref, 8> TopOffsetCache {};
|
||||
// Are we on the slow path?
|
||||
// Once we enter the slow path, we never come out.
|
||||
// This just simplifies the code atm. If there's a need to return to the fast path in the future
|
||||
@@ -257,7 +345,7 @@ private:
|
||||
bool SlowPath = false;
|
||||
// Keeping IREmitter not to pass arguments around
|
||||
IREmitter* IREmit = nullptr;
|
||||
IRListView* IR;
|
||||
IRListView* IR = nullptr;
|
||||
};
|
||||
|
||||
inline void X87StackOptimization::InvalidateCaches() {
|
||||
@@ -800,6 +888,9 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
Ref StackNode = SlowPath ? LoadStackValueAtOffset_Slow() : Value->StackDataNode;
|
||||
Ref AddrNode = CurrentIR.GetNode(Op->Addr);
|
||||
Ref Offset = CurrentIR.GetNode(Op->Offset);
|
||||
OpSize Align = Op->Align;
|
||||
MemOffsetType OffsetType = Op->OffsetType;
|
||||
uint8_t OffsetScale = Op->OffsetScale;
|
||||
|
||||
// On the fast path we can optimize memory copies.
|
||||
// If we are doing:
|
||||
@@ -811,51 +902,17 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
// or similar. As long as the source size and dest size are one and the same.
|
||||
// This will avoid any conversions between source and stack element size and conversion back.
|
||||
if (!SlowPath && Value->Source && Value->Source->first == Op->StoreSize && Value->InterpretAsFloat) {
|
||||
IREmit->_StoreMem(Value->InterpretAsFloat ? FPRClass : GPRClass, Op->StoreSize, Value->Source->second, AddrNode, Offset,
|
||||
OpSize::iInvalid, MEM_OFFSET_SXTX, 1);
|
||||
} else {
|
||||
if (ReducedPrecisionMode) {
|
||||
switch (Op->StoreSize) {
|
||||
case OpSize::i32Bit:
|
||||
case OpSize::i64Bit: {
|
||||
if (Op->StoreSize == OpSize::i32Bit) {
|
||||
StackNode = IREmit->_Float_FToF(OpSize::i32Bit, OpSize::i64Bit, StackNode);
|
||||
}
|
||||
IREmit->_StoreMem(FPRClass, Op->StoreSize, StackNode, AddrNode, Offset, OpSize::iInvalid, MEM_OFFSET_SXTX, 1);
|
||||
break;
|
||||
}
|
||||
case OpSize::f80Bit: {
|
||||
StackNode = IREmit->_F80CVTTo(StackNode, OpSize::i64Bit);
|
||||
IREmit->_StoreMem(FPRClass, OpSize::i64Bit, StackNode, AddrNode, Offset, OpSize::iInvalid, MEM_OFFSET_SXTX, 1);
|
||||
auto Upper = IREmit->_VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, StackNode, 1);
|
||||
auto NewOffset = IREmit->_Add(OpSize::i64Bit, Offset, GetConstant(8));
|
||||
IREmit->_StoreMem(GPRClass, OpSize::i16Bit, Upper, AddrNode, NewOffset, OpSize::i64Bit, MEM_OFFSET_SXTX, 1);
|
||||
break;
|
||||
}
|
||||
default: ERROR_AND_DIE_FMT("Unsupported x87 size");
|
||||
}
|
||||
} else { // !ReducedPrecisionMode
|
||||
if (Op->StoreSize != OpSize::f80Bit) { // if it's not 80bits then convert
|
||||
StackNode = IREmit->_F80CVT(Op->StoreSize, StackNode);
|
||||
}
|
||||
if (Op->StoreSize == OpSize::f80Bit) {
|
||||
if (Features.SupportsSVE128 || Features.SupportsSVE256) {
|
||||
if (!IsZero(Offset)) {
|
||||
AddrNode = IREmit->_Add(OpSize::i64Bit, AddrNode, Offset);
|
||||
}
|
||||
IREmit->_StoreMemX87SVEOptPredicate(OpSize::i128Bit, OpSize::i16Bit, StackNode, AddrNode);
|
||||
} else {
|
||||
// For X87 extended doubles, split before storing
|
||||
IREmit->_StoreMem(FPRClass, OpSize::i64Bit, StackNode, AddrNode, Offset, OpSize::iInvalid, MEM_OFFSET_SXTX, 1);
|
||||
auto Upper = IREmit->_VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, StackNode, 1);
|
||||
auto NewOffset = IREmit->_Add(OpSize::i64Bit, Offset, GetConstant(8));
|
||||
IREmit->_StoreMem(GPRClass, OpSize::i16Bit, Upper, AddrNode, NewOffset, OpSize::i64Bit, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
} else {
|
||||
IREmit->_StoreMem(FPRClass, Op->StoreSize, StackNode, AddrNode, Offset, OpSize::iInvalid, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
}
|
||||
IREmit->_StoreMem(Value->InterpretAsFloat ? FPRClass : GPRClass, Op->StoreSize, Value->Source->second, AddrNode, Offset, Align,
|
||||
OffsetType, OffsetScale);
|
||||
break;
|
||||
}
|
||||
|
||||
if (ReducedPrecisionMode) {
|
||||
StoreStackMem_Reduced_Helper(Op, StackNode);
|
||||
break;
|
||||
}
|
||||
|
||||
StoreStackMem_Helper(Op, StackNode);
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -987,7 +1044,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
|
||||
case OP_INCSTACKTOP: {
|
||||
if (SlowPath) {
|
||||
UpdateTopForPush_Slow();
|
||||
UpdateTopForPop_Slow();
|
||||
} else {
|
||||
StackData.rotate(false);
|
||||
}
|
||||
@@ -996,7 +1053,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
|
||||
case OP_DECSTACKTOP: {
|
||||
if (SlowPath) {
|
||||
UpdateTopForPop_Slow();
|
||||
UpdateTopForPush_Slow();
|
||||
} else {
|
||||
StackData.rotate(true);
|
||||
}
|
||||
@@ -1045,7 +1102,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
return;
|
||||
}
|
||||
|
||||
fextl::unique_ptr<Pass> CreateX87StackOptimizationPass(const FEXCore::HostFeatures& Features) {
|
||||
return fextl::make_unique<X87StackOptimization>(Features);
|
||||
fextl::unique_ptr<Pass> CreateX87StackOptimizationPass(const FEXCore::HostFeatures& Features, OpSize GPROpSize) {
|
||||
return fextl::make_unique<X87StackOptimization>(Features, GPROpSize);
|
||||
}
|
||||
} // namespace FEXCore::IR
|
||||
@@ -42,7 +42,8 @@ constexpr uint32_t STLR_INST = 0x08'9F'FC'00;
|
||||
constexpr uint32_t STLXR_MASK = 0x3F'E0'FC'00;
|
||||
constexpr uint32_t STLXR_INST = 0x08'00'FC'00;
|
||||
|
||||
constexpr uint32_t LDSTREGISTER_MASK = 0b0011'1011'0010'0000'0000'1100'0000'0000;
|
||||
// Load/store register (register offset) (Rm encoded as xzr)
|
||||
constexpr uint32_t LDSTREGISTER_MASK = 0b0011'1111'1111'1111'1111'1100'0000'0000;
|
||||
constexpr uint32_t LDR_INST = 0b0011'1000'0111'1111'0110'1000'0000'0000;
|
||||
constexpr uint32_t STR_INST = 0b0011'1000'0011'1111'0110'1000'0000'0000;
|
||||
|
||||
@@ -118,14 +119,6 @@ inline uint32_t GetRmReg(uint32_t Instr) {
|
||||
return (Instr >> RM_OFFSET) & REGISTER_MASK;
|
||||
}
|
||||
|
||||
|
||||
FEXCORE_TELEMETRY_STATIC_INIT(SplitLock, TYPE_HAS_SPLIT_LOCKS);
|
||||
FEXCORE_TELEMETRY_STATIC_INIT(SplitLock16B, TYPE_16BYTE_SPLIT);
|
||||
FEXCORE_TELEMETRY_STATIC_INIT(Cas16Tear, TYPE_CAS_16BIT_TEAR);
|
||||
FEXCORE_TELEMETRY_STATIC_INIT(Cas32Tear, TYPE_CAS_32BIT_TEAR);
|
||||
FEXCORE_TELEMETRY_STATIC_INIT(Cas64Tear, TYPE_CAS_64BIT_TEAR);
|
||||
FEXCORE_TELEMETRY_STATIC_INIT(Cas128Tear, TYPE_CAS_128BIT_TEAR);
|
||||
|
||||
static void ClearICache(void* Begin, std::size_t Length) {
|
||||
__builtin___clear_cache(static_cast<char*>(Begin), static_cast<char*>(Begin) + Length);
|
||||
}
|
||||
@@ -366,7 +359,7 @@ static bool RunCASPAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg1, uint3
|
||||
|
||||
// Check for Split lock across a cacheline
|
||||
if ((Addr & 63) > 56) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock, 1);
|
||||
FEXCORE_TELEMETRY_SET(TYPE_HAS_SPLIT_LOCKS, 1);
|
||||
if (StrictSplitLockMutex && !Lock.has_value()) {
|
||||
Lock.emplace(StrictSplitLockMutex);
|
||||
}
|
||||
@@ -374,7 +367,7 @@ static bool RunCASPAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg1, uint3
|
||||
|
||||
uint64_t AlignmentMask = 0b1111;
|
||||
if ((Addr & AlignmentMask) > 8) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock16B, 1);
|
||||
FEXCORE_TELEMETRY_SET(TYPE_16BYTE_SPLIT, 1);
|
||||
if (StrictSplitLockMutex && !Lock.has_value()) {
|
||||
Lock.emplace(StrictSplitLockMutex);
|
||||
}
|
||||
@@ -430,7 +423,7 @@ static bool RunCASPAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg1, uint3
|
||||
} else {
|
||||
// CAS managed to tear, we can't really solve this
|
||||
// Continue down the path to let the guest know values weren't expected
|
||||
FEXCORE_TELEMETRY_SET(Cas128Tear, 1);
|
||||
FEXCORE_TELEMETRY_SET(TYPE_CAS_128BIT_TEAR, 1);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -666,7 +659,7 @@ static uint16_t DoCAS16(uint16_t DesiredSrc, uint16_t ExpectedSrc, uint64_t Addr
|
||||
std::optional<FEXCore::Utils::SpinWaitLock::UniqueSpinMutex<uint32_t>> Lock {};
|
||||
|
||||
if ((Addr & 63) == 63) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock, 1);
|
||||
FEXCORE_TELEMETRY_SET(TYPE_HAS_SPLIT_LOCKS, 1);
|
||||
if (StrictSplitLockMutex && !Lock.has_value()) {
|
||||
Lock.emplace(StrictSplitLockMutex);
|
||||
}
|
||||
@@ -675,7 +668,7 @@ static uint16_t DoCAS16(uint16_t DesiredSrc, uint16_t ExpectedSrc, uint64_t Addr
|
||||
// 16 bit
|
||||
uint64_t AlignmentMask = 0b1111;
|
||||
if ((Addr & AlignmentMask) == 15) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock16B, 1);
|
||||
FEXCORE_TELEMETRY_SET(TYPE_16BYTE_SPLIT, 1);
|
||||
if (StrictSplitLockMutex && !Lock.has_value()) {
|
||||
Lock.emplace(StrictSplitLockMutex);
|
||||
}
|
||||
@@ -713,12 +706,11 @@ static uint16_t DoCAS16(uint16_t DesiredSrc, uint16_t ExpectedSrc, uint64_t Addr
|
||||
// CAS managed to tear, we can't really solve this
|
||||
// Continue down the path to let the guest know values weren't expected
|
||||
Tear = true;
|
||||
FEXCORE_TELEMETRY_SET(Cas16Tear, 1);
|
||||
FEXCORE_TELEMETRY_SET(TYPE_CAS_16BIT_TEAR, 1);
|
||||
}
|
||||
}
|
||||
|
||||
ActualLower = ExpectedLower;
|
||||
ActualUpper = ExpectedUpper;
|
||||
}
|
||||
|
||||
// If the bits changed that we were wanting to change then we have failed and can return
|
||||
@@ -942,7 +934,7 @@ static uint32_t DoCAS32(uint32_t DesiredSrc, uint32_t ExpectedSrc, uint64_t Addr
|
||||
std::optional<FEXCore::Utils::SpinWaitLock::UniqueSpinMutex<uint32_t>> Lock {};
|
||||
|
||||
if ((Addr & 63) > 60) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock, 1);
|
||||
FEXCORE_TELEMETRY_SET(TYPE_HAS_SPLIT_LOCKS, 1);
|
||||
if (StrictSplitLockMutex && !Lock.has_value()) {
|
||||
Lock.emplace(StrictSplitLockMutex);
|
||||
}
|
||||
@@ -951,7 +943,7 @@ static uint32_t DoCAS32(uint32_t DesiredSrc, uint32_t ExpectedSrc, uint64_t Addr
|
||||
// 32 bit
|
||||
uint64_t AlignmentMask = 0b1111;
|
||||
if ((Addr & AlignmentMask) > 12) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock16B, 1);
|
||||
FEXCORE_TELEMETRY_SET(TYPE_16BYTE_SPLIT, 1);
|
||||
if (StrictSplitLockMutex && !Lock.has_value()) {
|
||||
Lock.emplace(StrictSplitLockMutex);
|
||||
}
|
||||
@@ -1000,7 +992,7 @@ static uint32_t DoCAS32(uint32_t DesiredSrc, uint32_t ExpectedSrc, uint64_t Addr
|
||||
// CAS managed to tear, we can't really solve this
|
||||
// Continue down the path to let the guest know values weren't expected
|
||||
Tear = true;
|
||||
FEXCORE_TELEMETRY_SET(Cas32Tear, 1);
|
||||
FEXCORE_TELEMETRY_SET(TYPE_CAS_32BIT_TEAR, 1);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1174,7 +1166,7 @@ static uint64_t DoCAS64(uint64_t DesiredSrc, uint64_t ExpectedSrc, uint64_t Addr
|
||||
std::optional<FEXCore::Utils::SpinWaitLock::UniqueSpinMutex<uint32_t>> Lock {};
|
||||
|
||||
if ((Addr & 63) > 56) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock, 1);
|
||||
FEXCORE_TELEMETRY_SET(TYPE_HAS_SPLIT_LOCKS, 1);
|
||||
if (StrictSplitLockMutex && !Lock.has_value()) {
|
||||
Lock.emplace(StrictSplitLockMutex);
|
||||
}
|
||||
@@ -1183,7 +1175,7 @@ static uint64_t DoCAS64(uint64_t DesiredSrc, uint64_t ExpectedSrc, uint64_t Addr
|
||||
// 64bit
|
||||
uint64_t AlignmentMask = 0b1111;
|
||||
if ((Addr & AlignmentMask) > 8) {
|
||||
FEXCORE_TELEMETRY_SET(SplitLock16B, 1);
|
||||
FEXCORE_TELEMETRY_SET(TYPE_16BYTE_SPLIT, 1);
|
||||
if (StrictSplitLockMutex && !Lock.has_value()) {
|
||||
Lock.emplace(StrictSplitLockMutex);
|
||||
}
|
||||
@@ -1234,7 +1226,7 @@ static uint64_t DoCAS64(uint64_t DesiredSrc, uint64_t ExpectedSrc, uint64_t Addr
|
||||
// CAS managed to tear, we can't really solve this
|
||||
// Continue down the path to let the guest know values weren't expected
|
||||
Tear = true;
|
||||
FEXCORE_TELEMETRY_SET(Cas64Tear, 1);
|
||||
FEXCORE_TELEMETRY_SET(TYPE_CAS_64BIT_TEAR, 1);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -1,13 +1,10 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <fcntl.h>
|
||||
#include <limits.h>
|
||||
#ifndef _WIN32
|
||||
#include <linux/magic.h>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/vfs.h>
|
||||
#include <time.h>
|
||||
#endif
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
@@ -15,10 +12,11 @@
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
#define BACKEND_OFF 0
|
||||
#define BACKEND_GPUVIS 1
|
||||
|
||||
#ifdef ENABLE_FEXCORE_PROFILER
|
||||
#if FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_GPUVIS
|
||||
#include <array>
|
||||
#include <limits.h>
|
||||
#include <time.h>
|
||||
#ifndef _WIN32
|
||||
static inline uint64_t GetTime() {
|
||||
// We want the time in the least amount of overhead possible
|
||||
@@ -49,7 +47,6 @@ static inline uint64_t GetTime() {
|
||||
|
||||
#endif
|
||||
|
||||
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
|
||||
namespace FEXCore::Profiler {
|
||||
ProfilerBlock::ProfilerBlock(std::string_view const Format)
|
||||
: DurationBegin {GetTime()}
|
||||
@@ -114,35 +111,122 @@ void TraceObject(std::string_view const Format) {
|
||||
}
|
||||
}
|
||||
} // namespace GPUVis
|
||||
#elif FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_TRACY
|
||||
#include "tracy/Tracy.hpp"
|
||||
namespace Tracy {
|
||||
static int EnableAfterFork = 0;
|
||||
static bool Enable = false;
|
||||
|
||||
void Init(std::string_view ProgramName, std::string_view ProgramPath) {
|
||||
const char* ProfileTargetName = getenv("FEX_PROFILE_TARGET_NAME"); // Match by application name
|
||||
const char* ProfileTargetPath = getenv("FEX_PROFILE_TARGET_PATH"); // Match by path suffix
|
||||
const char* WaitForFork = getenv("FEX_PROFILE_WAIT_FOR_FORK"); // Don't enable profiling until the process forks N times
|
||||
bool Matched = (ProfileTargetName && ProgramName == ProfileTargetName) || (ProfileTargetPath && ProgramPath.ends_with(ProfileTargetPath));
|
||||
if (Matched && WaitForFork) {
|
||||
EnableAfterFork = std::atoi(WaitForFork);
|
||||
}
|
||||
Enable = Matched && !EnableAfterFork;
|
||||
if (Enable) {
|
||||
tracy::StartupProfiler();
|
||||
LogMan::Msg::IFmt("Tracy profiling started");
|
||||
} else if (EnableAfterFork) {
|
||||
LogMan::Msg::IFmt("Tracy profiling will start after fork");
|
||||
}
|
||||
}
|
||||
|
||||
void PostForkAction(bool IsChild) {
|
||||
if (Enable) {
|
||||
// Tracy does not support multiprocess profiling
|
||||
LogMan::Msg::EFmt("Warning: Profiling a process with forks is not supported. Set the environment variable "
|
||||
"FEX_PROFILE_WAIT_FOR_FORK=<n> to start profiling after the n-th fork.");
|
||||
}
|
||||
|
||||
if (IsChild) {
|
||||
Enable = false;
|
||||
return;
|
||||
}
|
||||
|
||||
if (EnableAfterFork > 1) {
|
||||
--EnableAfterFork;
|
||||
LogMan::Msg::IFmt("Tracy profiling will start after {} forks", EnableAfterFork);
|
||||
} else if (EnableAfterFork == 1) {
|
||||
Enable = true;
|
||||
EnableAfterFork = 0;
|
||||
tracy::StartupProfiler();
|
||||
LogMan::Msg::IFmt("Tracy profiling started");
|
||||
}
|
||||
}
|
||||
|
||||
void Shutdown() {
|
||||
if (Tracy::Enable) {
|
||||
LogMan::Msg::IFmt("Stopping Tracy profiling");
|
||||
tracy::ShutdownProfiler();
|
||||
}
|
||||
}
|
||||
|
||||
void TraceObject(std::string_view const Format, uint64_t Duration) {}
|
||||
|
||||
void TraceObject(std::string_view const Format) {
|
||||
if (Tracy::Enable) {
|
||||
TracyMessage(Format.data(), Format.size());
|
||||
}
|
||||
}
|
||||
} // namespace Tracy
|
||||
#else
|
||||
#error Unknown profiler backend
|
||||
#endif
|
||||
#endif
|
||||
|
||||
namespace FEXCore::Profiler {
|
||||
|
||||
#ifdef ENABLE_FEXCORE_PROFILER
|
||||
void Init() {
|
||||
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
|
||||
void Init(std::string_view ProgramName, std::string_view ProgramPath) {
|
||||
#if FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_GPUVIS
|
||||
GPUVis::Init();
|
||||
#elif FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_TRACY
|
||||
Tracy::Init(ProgramName, ProgramPath);
|
||||
#endif
|
||||
}
|
||||
|
||||
void PostForkAction(bool IsChild) {
|
||||
#if FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_TRACY
|
||||
Tracy::PostForkAction(IsChild);
|
||||
#endif
|
||||
}
|
||||
|
||||
bool IsActive() {
|
||||
#if FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_GPUVIS
|
||||
// Always active
|
||||
return true;
|
||||
#elif FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_TRACY
|
||||
// Active if previously enabled
|
||||
return Tracy::Enable;
|
||||
#endif
|
||||
}
|
||||
|
||||
void Shutdown() {
|
||||
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
|
||||
#if FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_GPUVIS
|
||||
GPUVis::Shutdown();
|
||||
#elif FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_TRACY
|
||||
Tracy::Shutdown();
|
||||
#endif
|
||||
}
|
||||
|
||||
void TraceObject(std::string_view const Format, uint64_t Duration) {
|
||||
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
|
||||
#if FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_GPUVIS
|
||||
GPUVis::TraceObject(Format, Duration);
|
||||
#elif FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_TRACY
|
||||
Tracy::TraceObject(Format, Duration);
|
||||
#endif
|
||||
}
|
||||
|
||||
void TraceObject(std::string_view const Format) {
|
||||
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
|
||||
#if FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_GPUVIS
|
||||
GPUVis::TraceObject(Format);
|
||||
#elif FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_TRACY
|
||||
Tracy::TraceObject(Format);
|
||||
#endif
|
||||
}
|
||||
|
||||
#endif
|
||||
} // namespace FEXCore::Profiler
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <atomic>
|
||||
#include <chrono>
|
||||
#include <mutex>
|
||||
|
||||
@@ -45,7 +45,7 @@ void Initialize() {
|
||||
return;
|
||||
}
|
||||
|
||||
auto DataDirectory = Config::GetTelemetryDirectory();
|
||||
const auto& DataDirectory = Config::GetTelemetryDirectory();
|
||||
|
||||
// Ensure the folder structure is created for our configuration
|
||||
if (!FHU::Filesystem::Exists(DataDirectory) && !FHU::Filesystem::CreateDirectories(DataDirectory)) {
|
||||
@@ -73,7 +73,7 @@ void Shutdown(const fextl::string& ApplicationName) {
|
||||
for (size_t i = 0; i < TelemetryType::TYPE_LAST; ++i) {
|
||||
auto& Name = TelemetryNames.at(i);
|
||||
auto& Data = TelemetryValues.at(i);
|
||||
fextl::fmt::print(File, "{}: {}\n", Name, *Data);
|
||||
fextl::fmt::print(File, "{}: {}\n", Name, Data.load());
|
||||
}
|
||||
File.Flush();
|
||||
}
|
||||
|
||||
@@ -15,6 +15,7 @@
|
||||
#include <charconv>
|
||||
#include <optional>
|
||||
#include <stdint.h>
|
||||
#include <variant>
|
||||
|
||||
namespace FEXCore::Config {
|
||||
namespace Handler {
|
||||
@@ -114,9 +115,10 @@ namespace DefaultValues {
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
|
||||
namespace Type {
|
||||
using StringArrayType = fextl::list<fextl::string>;
|
||||
#define OPT_BASE(type, group, enum, json, default) using P(enum) = P(type);
|
||||
#define OPT_STR(group, enum, json, default) using P(enum) = fextl::string;
|
||||
#define OPT_STRARRAY(group, enum, json, default) OPT_STR(group, enum, json, default)
|
||||
#define OPT_STRARRAY(group, enum, json, default) using P(enum) = StringArrayType;
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
} // namespace Type
|
||||
#define FEX_CONFIG_OPT(name, enum) \
|
||||
@@ -137,7 +139,9 @@ FEX_DEFAULT_VISIBILITY const fextl::string& GetConfigDirectory(bool Global);
|
||||
FEX_DEFAULT_VISIBILITY const fextl::string& GetConfigFileLocation(bool Global = false);
|
||||
FEX_DEFAULT_VISIBILITY fextl::string GetApplicationConfig(const std::string_view Program, bool Global);
|
||||
|
||||
using LayerValue = fextl::list<fextl::string>;
|
||||
using LayerValue =
|
||||
std::variant< fextl::string, DefaultValues::Type::StringArrayType, uint8_t, int8_t, uint16_t, int16_t, uint32_t, int32_t, uint64_t, int64_t, bool >;
|
||||
|
||||
using LayerOptions = fextl::unordered_map<ConfigOption, LayerValue>;
|
||||
|
||||
class FEX_DEFAULT_VISIBILITY Layer {
|
||||
@@ -151,13 +155,16 @@ public:
|
||||
return OptionMap.find(Option) != OptionMap.end();
|
||||
}
|
||||
|
||||
std::optional<LayerValue*> All(ConfigOption Option) {
|
||||
std::optional<DefaultValues::Type::StringArrayType*> All(ConfigOption Option) {
|
||||
const auto it = OptionMap.find(Option);
|
||||
if (it == OptionMap.end()) {
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
return &it->second;
|
||||
auto& Value = it->second;
|
||||
LOGMAN_THROW_A_FMT(std::holds_alternative<DefaultValues::Type::StringArrayType>(Value), "Tried to get config of invalid type!");
|
||||
|
||||
return &std::get<DefaultValues::Type::StringArrayType>(Value);
|
||||
}
|
||||
|
||||
std::optional<fextl::string*> Get(ConfigOption Option) {
|
||||
@@ -166,31 +173,44 @@ public:
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
return &it->second.front();
|
||||
auto& Value = it->second;
|
||||
LOGMAN_THROW_A_FMT(std::holds_alternative<fextl::string>(Value), "Tried to get config of invalid type!");
|
||||
|
||||
return &std::get<fextl::string>(Value);
|
||||
}
|
||||
|
||||
// Set will overwrite the object with a fextl::string without tests.
|
||||
void Set(ConfigOption Option, const char* Data) {
|
||||
LOGMAN_THROW_A_FMT(Data != nullptr, "Data can't be null");
|
||||
OptionMap[Option].emplace_back(fextl::string(Data));
|
||||
OptionMap[Option].emplace<fextl::string>(fextl::string(Data));
|
||||
}
|
||||
|
||||
void Set(ConfigOption Option, std::string_view Data) {
|
||||
OptionMap[Option].emplace_back(fextl::string(Data));
|
||||
OptionMap[Option].emplace<fextl::string>(fextl::string(Data));
|
||||
}
|
||||
|
||||
void Set(ConfigOption Option, fextl::string Data) {
|
||||
OptionMap[Option].emplace_back(std::move(Data));
|
||||
OptionMap[Option].emplace<fextl::string>(std::move(Data));
|
||||
}
|
||||
|
||||
void Set(ConfigOption Option, std::optional<fextl::string> Data) {
|
||||
if (Data) {
|
||||
OptionMap[Option].emplace_back(std::move(*Data));
|
||||
OptionMap[Option].emplace<fextl::string>(std::move(*Data));
|
||||
}
|
||||
}
|
||||
|
||||
void EraseSet(ConfigOption Option, std::string_view Data) {
|
||||
Erase(Option);
|
||||
Set(Option, Data);
|
||||
// AppendStrArrayValue will append strings to its StringArrayType.
|
||||
// If the value was previously a different type, then throw an assert.
|
||||
void AppendStrArrayValue(ConfigOption Option, std::string_view Data) {
|
||||
auto it = OptionMap.find(Option);
|
||||
if (it == OptionMap.end()) {
|
||||
// If the option didn't exist as a StringArrayType yet, emplace it.
|
||||
it = OptionMap.emplace(Option, DefaultValues::Type::StringArrayType {}).first;
|
||||
}
|
||||
|
||||
auto& Value = it->second;
|
||||
LOGMAN_THROW_A_FMT(std::holds_alternative<DefaultValues::Type::StringArrayType>(Value), "Tried to get config of invalid type!");
|
||||
std::get<DefaultValues::Type::StringArrayType>(Value).emplace_back(Data);
|
||||
}
|
||||
|
||||
void Erase(ConfigOption Option) {
|
||||
@@ -220,60 +240,25 @@ FEX_DEFAULT_VISIBILITY fextl::string FindContainerPrefix();
|
||||
FEX_DEFAULT_VISIBILITY void AddLayer(fextl::unique_ptr<FEXCore::Config::Layer> _Layer);
|
||||
|
||||
FEX_DEFAULT_VISIBILITY bool Exists(ConfigOption Option);
|
||||
FEX_DEFAULT_VISIBILITY std::optional<LayerValue*> All(ConfigOption Option);
|
||||
FEX_DEFAULT_VISIBILITY std::optional<DefaultValues::Type::StringArrayType*> All(ConfigOption Option);
|
||||
FEX_DEFAULT_VISIBILITY std::optional<fextl::string*> Get(ConfigOption Option);
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void Set(ConfigOption Option, std::string_view Data);
|
||||
FEX_DEFAULT_VISIBILITY void Erase(ConfigOption Option);
|
||||
FEX_DEFAULT_VISIBILITY void EraseSet(ConfigOption Option, std::string_view Data);
|
||||
|
||||
template<typename T>
|
||||
class FEX_DEFAULT_VISIBILITY Value {
|
||||
public:
|
||||
// Single value type.
|
||||
template<typename TT = T>
|
||||
requires (!std::is_same_v<TT, fextl::string>)
|
||||
Value(FEXCore::Config::ConfigOption _Option, TT Default)
|
||||
: Option {_Option} {
|
||||
requires (std::is_fundamental_v<TT> || std::is_same_v<TT, fextl::string>)
|
||||
Value(FEXCore::Config::ConfigOption Option, TT Default) {
|
||||
ValueData = GetIfExists(Option, Default);
|
||||
}
|
||||
|
||||
template<typename TT = T>
|
||||
requires (std::is_same_v<TT, fextl::string>)
|
||||
Value(FEXCore::Config::ConfigOption _Option, TT Default)
|
||||
: Option {_Option} {
|
||||
requires (std::is_fundamental_v<TT> || std::is_same_v<TT, fextl::string>)
|
||||
Value(FEXCore::Config::ConfigOption Option, std::string_view Default) {
|
||||
ValueData = GetIfExists(Option, Default);
|
||||
GetListIfExists(Option, &AppendList);
|
||||
}
|
||||
|
||||
template<typename TT = T>
|
||||
requires (std::is_same_v<TT, fextl::string>)
|
||||
Value(FEXCore::Config::ConfigOption _Option, std::string_view Default)
|
||||
: Option {_Option} {
|
||||
ValueData = GetIfExists(Option, Default);
|
||||
GetListIfExists(Option, &AppendList);
|
||||
}
|
||||
|
||||
template<typename TT = T>
|
||||
requires (!std::is_same_v<TT, fextl::string>)
|
||||
Value(FEXCore::Config::ConfigOption _Option)
|
||||
: Option {_Option} {
|
||||
if (!FEXCore::Config::Exists(Option)) {
|
||||
ERROR_AND_DIE_FMT("FEXCore::Config::Value has no value");
|
||||
}
|
||||
|
||||
ValueData = Get(Option);
|
||||
}
|
||||
|
||||
template<typename TT = T>
|
||||
requires (std::is_same_v<TT, fextl::string>)
|
||||
Value(FEXCore::Config::ConfigOption _Option)
|
||||
: Option {_Option} {
|
||||
if (!FEXCore::Config::Exists(Option)) {
|
||||
ERROR_AND_DIE_FMT("FEXCore::Config::Value has no value");
|
||||
}
|
||||
|
||||
ValueData = GetIfExists(Option);
|
||||
GetListIfExists(Option, &AppendList);
|
||||
}
|
||||
|
||||
operator T() const {
|
||||
@@ -281,7 +266,7 @@ public:
|
||||
}
|
||||
|
||||
template<typename TT = T>
|
||||
requires (!std::is_same_v<TT, fextl::string>)
|
||||
requires (std::is_fundamental_v<TT>)
|
||||
T operator()() const {
|
||||
return ValueData;
|
||||
}
|
||||
@@ -292,22 +277,31 @@ public:
|
||||
return ValueData;
|
||||
}
|
||||
|
||||
template<typename TT = T>
|
||||
requires (!std::is_same_v<TT, DefaultValues::Type::StringArrayType>)
|
||||
Value<T>(T Value) {
|
||||
ValueData = std::move(Value);
|
||||
}
|
||||
fextl::list<T>& All() {
|
||||
return AppendList;
|
||||
|
||||
// Array value types.
|
||||
template<typename TT = T>
|
||||
requires (std::is_same_v<TT, DefaultValues::Type::StringArrayType>)
|
||||
Value(FEXCore::Config::ConfigOption Option, std::string_view) {
|
||||
GetListIfExists(Option, &ValueData);
|
||||
}
|
||||
|
||||
template<typename TT = T>
|
||||
requires (std::is_same_v<TT, DefaultValues::Type::StringArrayType>)
|
||||
DefaultValues::Type::StringArrayType& All() {
|
||||
return ValueData;
|
||||
}
|
||||
|
||||
private:
|
||||
FEXCore::Config::ConfigOption Option;
|
||||
T ValueData;
|
||||
fextl::list<T> AppendList;
|
||||
T ValueData {};
|
||||
|
||||
static T Get(FEXCore::Config::ConfigOption Option);
|
||||
static T GetIfExists(FEXCore::Config::ConfigOption Option, T Default);
|
||||
static T GetIfExists(FEXCore::Config::ConfigOption Option, std::string_view Default);
|
||||
|
||||
static void GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string>* List);
|
||||
static void GetListIfExists(FEXCore::Config::ConfigOption Option, DefaultValues::Type::StringArrayType* List);
|
||||
};
|
||||
} // namespace FEXCore::Config
|
||||
@@ -7,6 +7,7 @@
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/IntervalList.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/set.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
@@ -216,6 +217,15 @@ public:
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual void AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t GuestThunkEntrypoint) = 0;
|
||||
|
||||
/**
|
||||
* @brief Adds additional per-instruction granularity TSO enable/disable information for the given range.
|
||||
*
|
||||
* @param ValidRanges The set of address ranges covered by this information
|
||||
* @param Instructions The set of instruction addresses within the given ranges for which TSO should be enabled
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual void AddForceTSOInformation(const IntervalList<uint64_t>& ValidRanges, fextl::set<uint64_t>&& Instructions) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void RemoveForceTSOInformation(uint64_t Address, uint64_t Size) = 0;
|
||||
private:
|
||||
};
|
||||
|
||||
|
||||
@@ -92,12 +92,18 @@ struct CPUState {
|
||||
// Counts the nesting depth of program sections that cause signals to be deferred.
|
||||
NonAtomicRefCounter<uint64_t> DeferredSignalRefCount;
|
||||
|
||||
// PF/AF raw values. Really only a byte of each matters, but this layout
|
||||
// (32-bits and in the first 256 bytes) is necessary to use ldp/stp to
|
||||
// spill/fill these togethers efficiently.
|
||||
uint32_t pf_raw {};
|
||||
uint32_t af_raw {};
|
||||
|
||||
uint64_t rip {}; ///< Current core's RIP. May not be entirely accurate while JIT is active
|
||||
|
||||
// The high 128-bits of AVX registers when not being emulated by SVE256.
|
||||
uint64_t avx_high[16][2];
|
||||
|
||||
uint64_t rip {}; ///< Current core's RIP. May not be entirely accurate while JIT is active
|
||||
uint64_t gregs[16] {};
|
||||
uint64_t _pad {};
|
||||
XMMRegs xmm {};
|
||||
|
||||
// Raw segment register indexes
|
||||
@@ -110,8 +116,8 @@ struct CPUState {
|
||||
uint64_t gs_cached {};
|
||||
uint64_t fs_cached {};
|
||||
uint8_t flags[48] {};
|
||||
uint64_t pf_raw {};
|
||||
uint64_t af_raw {};
|
||||
uint64_t _pad1 {};
|
||||
uint64_t _pad2 {};
|
||||
uint64_t mm[8][2] {};
|
||||
|
||||
// 32bit x86 state
|
||||
|
||||
@@ -36,6 +36,7 @@ struct HostFeatures {
|
||||
bool SupportsAES256 {};
|
||||
bool SupportsSVEBitPerm {};
|
||||
bool SupportsCPUIndexInTPIDRRO {};
|
||||
bool SupportsFRINTTS {};
|
||||
|
||||
// Float exception behaviour
|
||||
bool SupportsAFP {};
|
||||
|
||||
@@ -36,6 +36,10 @@ class OpDispatchBuilder;
|
||||
class PassManager;
|
||||
} // namespace FEXCore::IR
|
||||
|
||||
namespace FEXCore::Profiler {
|
||||
struct ThreadStats;
|
||||
};
|
||||
|
||||
namespace FEXCore::Core {
|
||||
|
||||
// Special-purpose replacement for std::unique_ptr to allow InternalThreadState to be standard layout.
|
||||
@@ -95,6 +99,9 @@ struct InternalThreadState : public FEXCore::Allocator::FEXAllocOperators {
|
||||
|
||||
std::shared_mutex ObjectCacheRefCounter {};
|
||||
|
||||
// This pointer is owned by the frontend.
|
||||
FEXCore::Profiler::ThreadStats* ThreadStats {};
|
||||
|
||||
///< Data pointer for exclusive use by the frontend
|
||||
void* FrontendPtr;
|
||||
|
||||
|
||||
@@ -80,6 +80,10 @@ enum NamedVectorConstant : uint8_t {
|
||||
NAMED_VECTOR_CVTMAX_I32,
|
||||
NAMED_VECTOR_CVTMAX_I64,
|
||||
NAMED_VECTOR_F80_SIGN_MASK,
|
||||
NAMED_VECTOR_SHA1RNDS_K0,
|
||||
NAMED_VECTOR_SHA1RNDS_K1,
|
||||
NAMED_VECTOR_SHA1RNDS_K2,
|
||||
NAMED_VECTOR_SHA1RNDS_K3,
|
||||
|
||||
NAMED_VECTOR_CONST_POOL_MAX,
|
||||
// Beginning of named constants that don't have a constant pool backing.
|
||||
|
||||
@@ -197,7 +197,7 @@ private:
|
||||
bool ShouldClose {};
|
||||
bool IsValidHandle {};
|
||||
|
||||
FileHandleType Handle;
|
||||
FileHandleType Handle {};
|
||||
#ifndef _WIN32
|
||||
static constexpr int DEFAULT_USER_PERMS = S_IRWXU | S_IRWXG | S_IRWXO;
|
||||
|
||||
|
||||
+19
-2
@@ -6,6 +6,7 @@
|
||||
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
namespace FEXCore {
|
||||
template<typename SizeType>
|
||||
class IntervalList {
|
||||
public:
|
||||
@@ -66,6 +67,12 @@ public:
|
||||
FirstIt->End = End;
|
||||
}
|
||||
|
||||
void Insert(const IntervalList<SizeType>& Other) {
|
||||
for (const auto& Interval : Other.Intervals) {
|
||||
Insert(Interval);
|
||||
}
|
||||
}
|
||||
|
||||
void Remove(Interval Entry) {
|
||||
if (Entry.Offset == Entry.End) {
|
||||
return;
|
||||
@@ -121,7 +128,7 @@ public:
|
||||
Intervals.erase(EraseStartIt, EraseEndIt);
|
||||
}
|
||||
|
||||
QueryResult Query(SizeType Offset) {
|
||||
QueryResult Query(SizeType Offset) const {
|
||||
const auto It = std::upper_bound(Intervals.begin(), Intervals.end(), Offset, [](const auto& LHS, const auto& RHS) {
|
||||
return LHS < RHS.End;
|
||||
}); // Lowest offset interval that (maybe) overlaps with the query offset
|
||||
@@ -135,11 +142,21 @@ public:
|
||||
}
|
||||
}
|
||||
|
||||
bool Intersect(Interval Entry) {
|
||||
bool Intersect(Interval Entry) const {
|
||||
const auto It = std::upper_bound(Intervals.begin(), Intervals.end(), Entry, [](const auto& LHS, const auto& RHS) {
|
||||
return LHS.Offset < RHS.End;
|
||||
}); // Lowest offset interval that (maybe) overlaps with the query offset
|
||||
|
||||
return It != Intervals.end() && It->Offset < Entry.End;
|
||||
}
|
||||
|
||||
bool Contains(Interval Entry) const {
|
||||
const auto It = std::upper_bound(Intervals.begin(), Intervals.end(), Entry, [](const auto& LHS, const auto& RHS) {
|
||||
return LHS.Offset < RHS.End;
|
||||
}); // Lowest offset interval that (maybe) overlaps with the query offset
|
||||
|
||||
return It != Intervals.end() && It->Offset <= Entry.Offset && It->End >= Entry.End;
|
||||
}
|
||||
};
|
||||
|
||||
} // namespace FEXCore
|
||||
@@ -1,18 +1,99 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <atomic>
|
||||
#include <cstdint>
|
||||
#include <string_view>
|
||||
|
||||
#ifdef _M_X86_64
|
||||
#include <x86intrin.h>
|
||||
#endif
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
#define FEXCORE_PROFILER_BACKEND_OFF 0
|
||||
#define FEXCORE_PROFILER_BACKEND_GPUVIS 1
|
||||
#define FEXCORE_PROFILER_BACKEND_TRACY 2
|
||||
|
||||
#if defined(ENABLE_FEXCORE_PROFILER) && FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_TRACY
|
||||
#include "tracy/Tracy.hpp"
|
||||
#endif
|
||||
|
||||
namespace FEXCore::Profiler {
|
||||
// FEXCore live-stats
|
||||
constexpr uint8_t STATS_VERSION = 2;
|
||||
enum class AppType : uint8_t {
|
||||
LINUX_32,
|
||||
LINUX_64,
|
||||
WIN_ARM64EC,
|
||||
WIN_WOW64,
|
||||
};
|
||||
|
||||
struct ThreadStatsHeader {
|
||||
uint8_t Version;
|
||||
AppType app_type;
|
||||
uint8_t _pad[2];
|
||||
char fex_version[48];
|
||||
std::atomic<uint32_t> Head;
|
||||
std::atomic<uint32_t> Size;
|
||||
uint32_t Pad;
|
||||
};
|
||||
|
||||
struct ThreadStats {
|
||||
std::atomic<uint32_t> Next;
|
||||
std::atomic<uint32_t> TID;
|
||||
|
||||
// Accumulated time (In unscaled CPU cycles!)
|
||||
uint64_t AccumulatedJITTime;
|
||||
uint64_t AccumulatedSignalTime;
|
||||
|
||||
// Accumulated event counts
|
||||
uint64_t AccumulatedSIGBUSCount;
|
||||
uint64_t AccumulatedSMCCount;
|
||||
uint64_t AccumulatedFloatFallbackCount;
|
||||
};
|
||||
|
||||
#ifdef ENABLE_FEXCORE_PROFILER
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void Init();
|
||||
#ifdef _M_ARM_64
|
||||
/**
|
||||
* @brief Get the raw cycle counter with synchronizing isb.
|
||||
*
|
||||
* `CNTVCTSS_EL0` also does the same thing, but requires the FEAT_ECV feature.
|
||||
*/
|
||||
static inline uint64_t GetCycleCounter() {
|
||||
uint64_t Result {};
|
||||
__asm volatile(R"(
|
||||
isb;
|
||||
mrs %[Res], CNTVCT_EL0;
|
||||
)"
|
||||
: [Res] "=r"(Result));
|
||||
return Result;
|
||||
}
|
||||
#else
|
||||
static inline uint64_t GetCycleCounter() {
|
||||
unsigned dummy;
|
||||
uint64_t tsc = __rdtscp(&dummy);
|
||||
return tsc;
|
||||
}
|
||||
#endif
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void Init(std::string_view ProgramName, std::string_view ProgramPath);
|
||||
FEX_DEFAULT_VISIBILITY void PostForkAction(bool IsChild);
|
||||
FEX_DEFAULT_VISIBILITY bool IsActive();
|
||||
FEX_DEFAULT_VISIBILITY void Shutdown();
|
||||
FEX_DEFAULT_VISIBILITY void TraceObject(std::string_view const Format);
|
||||
FEX_DEFAULT_VISIBILITY void TraceObject(std::string_view const Format, uint64_t Duration);
|
||||
|
||||
#define UniqueScopeName2(name, line) name##line
|
||||
#define UniqueScopeName(name, line) UniqueScopeName2(name, line)
|
||||
|
||||
// Declare an instantaneous profiler event.
|
||||
#define FEXCORE_PROFILE_INSTANT(name) FEXCore::Profiler::TraceObject(name)
|
||||
|
||||
#if FEXCORE_PROFILER_BACKEND == FEXCORE_PROFILER_BACKEND_TRACY
|
||||
// Declare a scoped profile block variable with a fixed name.
|
||||
#define FEXCORE_PROFILE_SCOPED(name) ZoneNamedN(___tracy_scoped_zone, name, ::FEXCore::Profiler::IsActive())
|
||||
#else
|
||||
// A class that follows scoping rules to generate a profile duration block
|
||||
class ProfilerBlock final {
|
||||
public:
|
||||
@@ -25,18 +106,45 @@ private:
|
||||
std::string_view const Format;
|
||||
};
|
||||
|
||||
#define UniqueScopeName2(name, line) name##line
|
||||
#define UniqueScopeName(name, line) UniqueScopeName2(name, line)
|
||||
|
||||
// Declare an instantaneous profiler event.
|
||||
#define FEXCORE_PROFILE_INSTANT(name) FEXCore::Profiler::TraceObject(name)
|
||||
|
||||
// Declare a scoped profile block variable with a fixed name.
|
||||
#define FEXCORE_PROFILE_SCOPED(name) FEXCore::Profiler::ProfilerBlock UniqueScopeName(ScopedBlock_, __LINE__)(name)
|
||||
#endif
|
||||
|
||||
template<typename T, size_t FlatOffset = 0>
|
||||
class AccumulationBlock final {
|
||||
public:
|
||||
AccumulationBlock(T* Stat)
|
||||
: Begin {GetCycleCounter()}
|
||||
, Stat {Stat} {}
|
||||
|
||||
~AccumulationBlock() {
|
||||
const auto Duration = GetCycleCounter() - Begin + FlatOffset;
|
||||
if (Stat) {
|
||||
auto ref = std::atomic_ref<T>(*Stat);
|
||||
ref.fetch_add(Duration, std::memory_order_relaxed);
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
uint64_t Begin;
|
||||
T* Stat;
|
||||
};
|
||||
|
||||
#define FEXCORE_PROFILE_ACCUMULATION(ThreadState, Stat) \
|
||||
FEXCore::Profiler::AccumulationBlock<decltype(ThreadState->ThreadStats->Stat)> UniqueScopeName(ScopedAccumulation_, __LINE__)( \
|
||||
ThreadState->ThreadStats ? &ThreadState->ThreadStats->Stat : nullptr);
|
||||
#define FEXCORE_PROFILE_INSTANT_INCREMENT(ThreadState, Stat, value) \
|
||||
do { \
|
||||
if (ThreadState->ThreadStats) { \
|
||||
ThreadState->ThreadStats->Stat += value; \
|
||||
} \
|
||||
} while (0)
|
||||
|
||||
#else
|
||||
[[maybe_unused]]
|
||||
static void Init() {}
|
||||
static void Init(std::string_view ProgramName, std::string_view ProgramPath) {}
|
||||
[[maybe_unused]]
|
||||
static void PostForkAction(bool IsChild) {}
|
||||
[[maybe_unused]]
|
||||
static void Shutdown() {}
|
||||
[[maybe_unused]]
|
||||
@@ -50,5 +158,12 @@ static void TraceObject(std::string_view const, uint64_t) {}
|
||||
#define FEXCORE_PROFILE_SCOPED(...) \
|
||||
do { \
|
||||
} while (0)
|
||||
#define FEXCORE_PROFILE_ACCUMULATION(...) \
|
||||
do { \
|
||||
} while (0)
|
||||
#define FEXCORE_PROFILE_INSTANT_INCREMENT(...) \
|
||||
do { \
|
||||
} while (0)
|
||||
|
||||
#endif
|
||||
} // namespace FEXCore::Profiler
|
||||
@@ -33,36 +33,12 @@ enum TelemetryType {
|
||||
};
|
||||
|
||||
#ifndef FEX_DISABLE_TELEMETRY
|
||||
class Value;
|
||||
|
||||
class Value final {
|
||||
public:
|
||||
Value() = default;
|
||||
Value(uint64_t Default)
|
||||
: Data {Default} {}
|
||||
|
||||
uint64_t operator*() const {
|
||||
return Data;
|
||||
}
|
||||
void operator=(uint64_t Value) {
|
||||
Data = Value;
|
||||
}
|
||||
void operator|=(uint64_t Value) {
|
||||
Data |= Value;
|
||||
}
|
||||
void operator++(int) {
|
||||
Data++;
|
||||
}
|
||||
|
||||
std::atomic<uint64_t>* GetAddr() {
|
||||
return &Data;
|
||||
}
|
||||
|
||||
private:
|
||||
std::atomic<uint64_t> Data;
|
||||
};
|
||||
using Value = std::atomic<uint64_t>;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY extern std::array<Value, FEXCore::Telemetry::TelemetryType::TYPE_LAST> TelemetryValues;
|
||||
// This returns the internal structure to the telemetry data structures
|
||||
// One must be careful with placing these in the hot path of code execution
|
||||
// It can be fairly costly, especially in the static version where it puts barriers in the code
|
||||
inline Value& GetTelemetryValue(TelemetryType Type) {
|
||||
return FEXCore::Telemetry::TelemetryValues[Type];
|
||||
}
|
||||
@@ -71,26 +47,28 @@ FEX_DEFAULT_VISIBILITY void Initialize();
|
||||
FEX_DEFAULT_VISIBILITY void Shutdown(const fextl::string& ApplicationName);
|
||||
|
||||
// Telemetry object declaration
|
||||
// This returns the internal structure to the telemetry data structures
|
||||
// One must be careful with placing these in the hot path of code execution
|
||||
// It can be fairly costly, especially in the static version where it puts barriers in the code
|
||||
#define FEXCORE_TELEMETRY_STATIC_INIT(Name, Type) \
|
||||
static FEXCore::Telemetry::Value& Name = FEXCore::Telemetry::GetTelemetryValue(FEXCore::Telemetry::Type)
|
||||
#define FEXCORE_TELEMETRY_INIT(Name, Type) FEXCore::Telemetry::Value& Name = FEXCore::Telemetry::GetTelemetryValue(FEXCore::Telemetry::Type)
|
||||
// Telemetry ALU operations
|
||||
// These are typically 3-4 instructions depending on what you're doing
|
||||
#define FEXCORE_TELEMETRY_SET(Name, Value) Name = Value
|
||||
#define FEXCORE_TELEMETRY_OR(Name, Value) Name |= Value
|
||||
#define FEXCORE_TELEMETRY_INC(Name) Name++
|
||||
#define FEXCORE_TELEMETRY_SET(Type, Value) \
|
||||
do { \
|
||||
auto& Name = FEXCore::Telemetry::TelemetryValues[FEXCore::Telemetry::Type]; \
|
||||
Name = Value; \
|
||||
} while (0)
|
||||
#define FEXCORE_TELEMETRY_OR(Type, Value) \
|
||||
do { \
|
||||
auto& Name = FEXCore::Telemetry::TelemetryValues[FEXCore::Telemetry::Type]; \
|
||||
Name |= Value; \
|
||||
} while (0)
|
||||
#define FEXCORE_TELEMETRY_INC(Type, Value) \
|
||||
do { \
|
||||
auto& Name = FEXCore::Telemetry::TelemetryValues[FEXCore::Telemetry::Type]; \
|
||||
Name++; \
|
||||
} while (0)
|
||||
|
||||
// Returns a pointer to std::atomic<uint64_t>. Can be useful if you are attempting to JIT telemetry accesses for debug purposes
|
||||
// Not recommended to do telemetry inside JIT code in production code
|
||||
#define FEXCORE_TELEMETRY_Addr(Name) Name->GetAddr()
|
||||
#else
|
||||
static inline void Initialize() {}
|
||||
static inline void Shutdown(const fextl::string& ApplicationName) {}
|
||||
|
||||
#define FEXCORE_TELEMETRY_STATIC_INIT(Name, Type)
|
||||
#define FEXCORE_TELEMETRY_INIT(Name, Type)
|
||||
#define FEXCORE_TELEMETRY(Name, Value) \
|
||||
do { \
|
||||
@@ -104,6 +82,5 @@ static inline void Shutdown(const fextl::string& ApplicationName) {}
|
||||
#define FEXCORE_TELEMETRY_INC(Name) \
|
||||
do { \
|
||||
} while (0)
|
||||
#define FEXCORE_TELEMETRY_Addr(Name) reinterpret_cast<std::atomic<uint64_t>*>(nullptr)
|
||||
#endif
|
||||
} // namespace FEXCore::Telemetry
|
||||
@@ -0,0 +1,106 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/AllocatorHooks.h>
|
||||
|
||||
#include <functional>
|
||||
#include <type_traits>
|
||||
#include <utility>
|
||||
|
||||
namespace fextl {
|
||||
|
||||
/**
|
||||
* Equivalent to std::move_only_function but uses FEXCore::Allocator routines
|
||||
* for non-function pointers.
|
||||
*/
|
||||
template<typename F, void* (*Alloc)(size_t, size_t) = ::FEXCore::Allocator::aligned_alloc, void (*Dealloc)(void*) = ::FEXCore::Allocator::aligned_free>
|
||||
class move_only_function;
|
||||
|
||||
template<typename R, typename... Args, void* (*Alloc)(size_t, size_t), void (*Dealloc)(void*)>
|
||||
class move_only_function<R(Args...), Alloc, Dealloc> {
|
||||
public:
|
||||
template<typename F>
|
||||
requires std::is_invocable_r_v<R, F, Args...>
|
||||
move_only_function(F&& f) noexcept(std::is_nothrow_move_constructible_v<F>) {
|
||||
if constexpr (std::is_convertible_v<F, R (*)(Args...)>) {
|
||||
// Argument is a function pointer, a captureless lambda, or a stateless function object.
|
||||
// std::function can store these without allocation
|
||||
internal = std::move(f);
|
||||
} else if constexpr (std::is_nothrow_constructible_v<std::function<R(Args...)>, F>) {
|
||||
// If construction is guaranteed not to throw an exception, this implies
|
||||
// the std::function implementation won't allocate memory!
|
||||
internal = std::move(f);
|
||||
} else {
|
||||
// Other arguments require allocation, which is a problem since
|
||||
// std::function doesn't allow allocator customization. Implementations
|
||||
// are generally able to avoid allocation for lambdas with a single
|
||||
// pointer capture however. We can exploit this special case by wrapping
|
||||
// the actual argument in a lambda that points an external storage
|
||||
// location.
|
||||
|
||||
static_assert(!std::is_pointer_v<F>, "Pointer types must manually be dereferenced");
|
||||
|
||||
// First, relocate argument to a location returned from FEX's allocators
|
||||
using Fnoref = std::remove_reference_t<F>;
|
||||
storage = Alloc(std::alignment_of_v<Fnoref>, sizeof(Fnoref));
|
||||
auto moved_lambda = new (storage) Fnoref {std::move(f)};
|
||||
|
||||
// Second, wrap the relocated argument in a single-capture lambda
|
||||
auto wrapped_lambda = [moved_lambda](Args... args) {
|
||||
return (*moved_lambda)(std::forward<Args>(args)...);
|
||||
};
|
||||
|
||||
// Third, assign the result to std::function, ensuring it's indeed
|
||||
// allocation-free by checking for nothrow-constructibility
|
||||
static_assert(noexcept(internal = std::move(wrapped_lambda)), "This implementation of std::function "
|
||||
"does not support implementing "
|
||||
"fextl::move_only_function");
|
||||
internal = std::move(wrapped_lambda);
|
||||
|
||||
// Finally, if a destructor must be called, generate a pointer to its destructor
|
||||
if constexpr (!std::is_trivially_destructible_v<Fnoref>) {
|
||||
internal_destructor = [](move_only_function* self) {
|
||||
reinterpret_cast<Fnoref*>(self->storage)->~Fnoref();
|
||||
};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
move_only_function() noexcept {}
|
||||
move_only_function(std::nullptr_t) noexcept {}
|
||||
move_only_function(const move_only_function&) = delete;
|
||||
move_only_function(move_only_function&& other) noexcept {
|
||||
*this = std::move(other);
|
||||
}
|
||||
|
||||
move_only_function& operator=(move_only_function&& other) noexcept {
|
||||
if (!other && internal_destructor) {
|
||||
this->~move_only_function();
|
||||
}
|
||||
internal = std::exchange(other.internal, nullptr);
|
||||
internal_destructor = std::exchange(other.internal_destructor, nullptr);
|
||||
storage = std::exchange(other.storage, nullptr);
|
||||
return *this;
|
||||
}
|
||||
|
||||
~move_only_function() {
|
||||
if (internal_destructor) {
|
||||
internal_destructor(this);
|
||||
}
|
||||
Dealloc(storage);
|
||||
}
|
||||
|
||||
R operator()(Args... args) const {
|
||||
return internal(std::forward<Args>(args)...);
|
||||
}
|
||||
|
||||
explicit operator bool() const noexcept {
|
||||
return (bool)internal;
|
||||
}
|
||||
|
||||
private:
|
||||
std::function<R(Args...)> internal;
|
||||
void (*internal_destructor)(move_only_function*) = nullptr;
|
||||
void* storage = nullptr;
|
||||
};
|
||||
} // namespace fextl
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
#include <catch2/generators/catch_generators_range.hpp>
|
||||
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Utils/SpinWaitLock.h"
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
#include <chrono>
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
#include <catch2/generators/catch_generators_range.hpp>
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
#include <catch2/generators/catch_generators_range.hpp>
|
||||
#include <catch2/generators/catch_generators_random.hpp>
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "TestDisassembler.h"
|
||||
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "TestDisassembler.h"
|
||||
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
@@ -6,33 +7,24 @@
|
||||
using namespace ARMEmitter;
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ASIMD: Cryptographic AES") {
|
||||
if (false) {
|
||||
// vixl doesn't support these instructions.
|
||||
TEST_SINGLE(aese(VReg::v30, VReg::v29), "aese v30, v29");
|
||||
TEST_SINGLE(aesd(VReg::v30, VReg::v29), "aesd v30, v29");
|
||||
TEST_SINGLE(aesmc(VReg::v30, VReg::v29), "aesmc v30, v29");
|
||||
TEST_SINGLE(aesimc(VReg::v30, VReg::v29), "aesimc v30, v29");
|
||||
}
|
||||
TEST_SINGLE(aese(VReg::v30, VReg::v29), "aese v30.16b, v29.16b");
|
||||
TEST_SINGLE(aesd(VReg::v30, VReg::v29), "aesd v30.16b, v29.16b");
|
||||
TEST_SINGLE(aesmc(VReg::v30, VReg::v29), "aesmc v30.16b, v29.16b");
|
||||
TEST_SINGLE(aesimc(VReg::v30, VReg::v29), "aesimc v30.16b, v29.16b");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ASIMD: Cryptographic three-register SHA") {
|
||||
if (false) {
|
||||
// vixl doesn't support these instructions.
|
||||
TEST_SINGLE(sha1c(VReg::v30, SReg::s29, VReg::v28), "sha1c v30, s29, v28");
|
||||
TEST_SINGLE(sha1p(VReg::v30, SReg::s29, VReg::v28), "sha1p v30, s29, v28");
|
||||
TEST_SINGLE(sha1m(VReg::v30, SReg::s29, VReg::v28), "sha1m v30, s29, v28");
|
||||
TEST_SINGLE(sha1su0(VReg::v30, VReg::v29, VReg::v28), "sha1su0 v30, v29, v28");
|
||||
TEST_SINGLE(sha256h(VReg::v30, VReg::v29, VReg::v28), "sha256h v30, v29, v28");
|
||||
TEST_SINGLE(sha256h2(VReg::v30, VReg::v29, VReg::v28), "sha256h2 v30, v29, v28");
|
||||
TEST_SINGLE(sha256su1(VReg::v30, VReg::v29, VReg::v28), "sha256su1 v30, v29, v28");
|
||||
}
|
||||
TEST_SINGLE(sha1c(VReg::v30, SReg::s29, VReg::v28), "sha1c q30, s29, v28.4s");
|
||||
TEST_SINGLE(sha1p(VReg::v30, SReg::s29, VReg::v28), "sha1p q30, s29, v28.4s");
|
||||
TEST_SINGLE(sha1m(VReg::v30, SReg::s29, VReg::v28), "sha1m q30, s29, v28.4s");
|
||||
TEST_SINGLE(sha1su0(VReg::v30, VReg::v29, VReg::v28), "sha1su0 v30.4s, v29.4s, v28.4s");
|
||||
TEST_SINGLE(sha256h(VReg::v30, VReg::v29, VReg::v28), "sha256h q30, q29, v28.4s");
|
||||
TEST_SINGLE(sha256h2(VReg::v30, VReg::v29, VReg::v28), "sha256h2 q30, q29, v28.4s");
|
||||
TEST_SINGLE(sha256su1(VReg::v30, VReg::v29, VReg::v28), "sha256su1 v30.4s, v29.4s, v28.4s");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ASIMD: Cryptographic two-register SHA") {
|
||||
if (false) {
|
||||
// vixl doesn't support these instructions.
|
||||
TEST_SINGLE(sha1h(SReg::s30, SReg::s29), "sha1h s30, s29");
|
||||
TEST_SINGLE(sha1su1(VReg::v30, VReg::v29), "sha1su1 v30, v29");
|
||||
TEST_SINGLE(sha256su0(VReg::v30, VReg::v29), "sha256su0 v30, v29");
|
||||
}
|
||||
TEST_SINGLE(sha1h(SReg::s30, SReg::s29), "sha1h s30, s29");
|
||||
TEST_SINGLE(sha1su1(VReg::v30, VReg::v29), "sha1su1 v30.4s, v29.4s");
|
||||
TEST_SINGLE(sha256su0(VReg::v30, VReg::v29), "sha256su0 v30.4s, v29.4s");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ASIMD: Advanced SIMD table lookup") {
|
||||
TEST_SINGLE(tbl(QReg::q30, QReg::q26, QReg::q25), "tbl v30.16b, {v26.16b}, v25.16b");
|
||||
|
||||
Loaded 100 of 379 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user