mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-07 06:00:16 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3ba84ad06a | ||
|
|
c6aae9e05a | ||
|
|
95b4618833 | ||
|
|
1fe17d55d9 | ||
|
|
ead73371d9 | ||
|
|
6f089a4323 | ||
|
|
640f024551 | ||
|
|
99920f89dd | ||
|
|
8ea276267f | ||
|
|
6ebbd91245 | ||
|
|
fa0a54deb9 | ||
|
|
046043090f | ||
|
|
61150a18cc | ||
|
|
95ca20cfee | ||
|
|
5a536d47fd | ||
|
|
afbc7da027 | ||
|
|
c093c08c40 | ||
|
|
360d8c629e | ||
|
|
abb41d39e4 | ||
|
|
d4eb4ef594 | ||
|
|
02f45854e8 | ||
|
|
62de1004df | ||
|
|
7f216ca02f | ||
|
|
492b0fdda8 | ||
|
|
bb072c0112 | ||
|
|
afabe7cb47 | ||
|
|
38e0fc2434 | ||
|
|
16a70eafc6 | ||
|
|
de4becc26e | ||
|
|
af23f4325f | ||
|
|
94af96df8f | ||
|
|
e685ab818e | ||
|
|
5f2a72b65b | ||
|
|
c1842a6167 | ||
|
|
3f3907b5d1 | ||
|
|
1899465390 | ||
|
|
18360d4ccb | ||
|
|
a691c3cd99 | ||
|
|
70bc561bbf | ||
|
|
b9222d8431 | ||
|
|
a5fad89e57 | ||
|
|
a14360b89d | ||
|
|
c9aaedd217 | ||
|
|
cda15ce9ea | ||
|
|
3f3b6ad337 | ||
|
|
62410c4381 | ||
|
|
cf82b56dd8 | ||
|
|
4a74bea7ab | ||
|
|
c6d8e60ef8 | ||
|
|
1212cd526a | ||
|
|
79a8ed53b6 | ||
|
|
a6ce115d9c | ||
|
|
646a5a7f9e | ||
|
|
7f71b6f1b2 | ||
|
|
16d5ca447f | ||
|
|
3d0c20a263 | ||
|
|
1c38b8b046 | ||
|
|
e1124480be | ||
|
|
5243f50ed1 | ||
|
|
cde805147f | ||
|
|
7e39eb3df2 | ||
|
|
af366d4480 | ||
|
|
33ef98aae7 | ||
|
|
c16db2db4a | ||
|
|
059d980c33 | ||
|
|
581381fd86 | ||
|
|
0072b289bb | ||
|
|
e2bd79087e | ||
|
|
6cd78fc90d | ||
|
|
c346aca241 | ||
|
|
bf83569f0b | ||
|
|
1502f04a8a | ||
|
|
7bd9d0ae23 | ||
|
|
cf57afdf26 | ||
|
|
43e6aebc7a | ||
|
|
22780993e1 | ||
|
|
578dcee9af | ||
|
|
61d77e3f9b | ||
|
|
4ce0acba80 | ||
|
|
d137212222 | ||
|
|
9ad4e3a6a0 | ||
|
|
57627d4fcf | ||
|
|
23b69271eb | ||
|
|
9d2f557666 | ||
|
|
4f9e352ff0 | ||
|
|
3e85e60a30 | ||
|
|
febce21b21 | ||
|
|
755364e2df | ||
|
|
df63979773 | ||
|
|
992d86bbc1 | ||
|
|
534b338161 | ||
|
|
ef250f936c | ||
|
|
e57130e364 | ||
|
|
eedcb35270 | ||
|
|
1eb470083c | ||
|
|
1f15a4e35b | ||
|
|
7ed9bea16b | ||
|
|
8b1383d235 | ||
|
|
a73fab3bb5 | ||
|
|
7bd64d9c53 | ||
|
|
1786c2f157 | ||
|
|
958b671736 | ||
|
|
6549b66cf6 | ||
|
|
0c855a5ce3 | ||
|
|
6864d48dcf | ||
|
|
99816a23a8 | ||
|
|
2ff9546523 | ||
|
|
06541f21d6 | ||
|
|
45a37edd4a | ||
|
|
2a713a1f51 | ||
|
|
3e104de377 | ||
|
|
b22b316e70 | ||
|
|
df461546c5 | ||
|
|
1c1c43cd86 | ||
|
|
3eac9f937e | ||
|
|
f13a0d8e84 | ||
|
|
ca12dc9213 | ||
|
|
365ed2cd70 | ||
|
|
7efbfed0bf | ||
|
|
ed502738c6 | ||
|
|
ba162bb058 | ||
|
|
f194d35913 | ||
|
|
e81c84e00a | ||
|
|
68dc9030bc | ||
|
|
fedad275e7 | ||
|
|
f6b4c76d76 | ||
|
|
27854aa091 | ||
|
|
d966ae145e | ||
|
|
4cb37e6a1b | ||
|
|
d9da81e99b | ||
|
|
7d734740be | ||
|
|
06299cca4b | ||
|
|
af5aaab38b | ||
|
|
77bb01d384 | ||
|
|
a9eb1bd4e6 | ||
|
|
83ef2da95f | ||
|
|
ba13ccadb8 | ||
|
|
4a2dee873d | ||
|
|
589906e6e6 | ||
|
|
f0fcf6d9e6 | ||
|
|
1591ced5a7 | ||
|
|
0d235d63d0 | ||
|
|
820d5c1447 | ||
|
|
8093e8f5f1 | ||
|
|
802857e411 | ||
|
|
bf3a1839a2 | ||
|
|
ad7844d7da | ||
|
|
3ff9128a8a | ||
|
|
c865eb98ea | ||
|
|
217a9228f8 | ||
|
|
105d0a36ad | ||
|
|
263279d5dd | ||
|
|
861ecbec0f | ||
|
|
5ff9bb3669 | ||
|
|
e794584bb5 | ||
|
|
a792dd0703 | ||
|
|
a9a6a645bf | ||
|
|
7c93becd5f | ||
|
|
4bbaef58e9 | ||
|
|
95791a985a | ||
|
|
0dfefe9730 | ||
|
|
8481c797df | ||
|
|
a503b5e20b | ||
|
|
4078840ef1 | ||
|
|
ab51958b26 | ||
|
|
3376587b6a | ||
|
|
6681d7dcf9 | ||
|
|
34224481c6 | ||
|
|
1837aaabe4 | ||
|
|
5811914a78 | ||
|
|
a109a4efa1 | ||
|
|
a08a6ce5de | ||
|
|
3dc8a3ddc1 | ||
|
|
5e103365f7 | ||
|
|
cdea8d7f74 | ||
|
|
ef6dc3d802 | ||
|
|
e6edb349ba | ||
|
|
ad132267ec | ||
|
|
6f837281ef | ||
|
|
2a1d29d2df | ||
|
|
bdceb4ca89 | ||
|
|
656bb928cf | ||
|
|
0fbe69ebcf | ||
|
|
ece817c691 | ||
|
|
3fbc8204b7 | ||
|
|
44a5481254 | ||
|
|
cf2ff90f87 | ||
|
|
b8dd5d95b0 | ||
|
|
7cd52febc2 | ||
|
|
8e079c1965 | ||
|
|
4928af5a64 | ||
|
|
289df740cd | ||
|
|
6ad7392cd7 | ||
|
|
b15d5f299c | ||
|
|
5c7c959dd9 | ||
|
|
6927c7577a | ||
|
|
88682a457a | ||
|
|
61ae53cc03 | ||
|
|
0e67f30103 | ||
|
|
babd6e9a7b | ||
|
|
76caa2c6e3 | ||
|
|
dc9f8aa855 | ||
|
|
ded8b3284a | ||
|
|
2e24ee7a5f | ||
|
|
048ae597d2 | ||
|
|
0147f7aa19 | ||
|
|
23cda2c961 | ||
|
|
feb67658e1 | ||
|
|
65ee1fafa8 | ||
|
|
49f8332c5b | ||
|
|
afce108ed7 | ||
|
|
30f3b545af | ||
|
|
bf1597920c | ||
|
|
f66bf3811e | ||
|
|
a01e29ac99 | ||
|
|
3136a5e2f8 | ||
|
|
c4f00a05df | ||
|
|
e9a8f8a9ff | ||
|
|
261ae7a195 | ||
|
|
7eaf5ae9e0 | ||
|
|
10a02449b1 | ||
|
|
debc57e8c7 | ||
|
|
7a4fff8e5b | ||
|
|
ec0a2a8671 | ||
|
|
b8b516f7b6 | ||
|
|
3b1e91b1fc | ||
|
|
8532593d91 | ||
|
|
55bd16e2c8 | ||
|
|
c7797d56c9 | ||
|
|
9b573effd1 | ||
|
|
dfd1aedae5 | ||
|
|
221ae2d7b4 | ||
|
|
9127d206b5 | ||
|
|
ecc6fea54e | ||
|
|
99446da7c1 | ||
|
|
ad75563a26 | ||
|
|
197af972d9 | ||
|
|
fbd706c191 | ||
|
|
6b85fa5611 | ||
|
|
fae66a921f | ||
|
|
e6fc462e9d | ||
|
|
d11a265a53 | ||
|
|
395d870814 | ||
|
|
fec1ffaa6b | ||
|
|
d4fdb28e72 | ||
|
|
39a5c2021e | ||
|
|
f41501444d | ||
|
|
86b26b80ce | ||
|
|
ae9a5b1125 | ||
|
|
38579807c2 | ||
|
|
c19119bcd6 | ||
|
|
47f1ad693d | ||
|
|
2164d7bb96 | ||
|
|
c326e2d669 | ||
|
|
8eaf45414c | ||
|
|
b3297d106e | ||
|
|
2d0e19e6a7 | ||
|
|
fd2ee4dc46 | ||
|
|
b1bbc37c59 | ||
|
|
56409d4f2b | ||
|
|
ec0683b729 | ||
|
|
89e5041e70 | ||
|
|
8b3e7312e0 | ||
|
|
3c76a9176d | ||
|
|
a37def2c22 |
No files matched your search
@@ -78,7 +78,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_TESTS=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_TESTS=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
|
||||
+10
-11
@@ -35,10 +35,13 @@ set (FEXCORE_PROFILER_BACKEND "gpuvis" CACHE STRING "Set which backend to use fo
|
||||
option(ENABLE_GLIBC_ALLOCATOR_HOOK_FAULT "Enables glibc memory allocation hooking with fault for CI testing")
|
||||
option(USE_PDB_DEBUGINFO "Builds debug info in PDB format" FALSE)
|
||||
|
||||
set (X86_32_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_32.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting i686")
|
||||
set (X86_64_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_64.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting x86_64")
|
||||
set (X86_32_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/toolchain_x86_32.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting i686")
|
||||
set (X86_64_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/toolchain_x86_64.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting x86_64")
|
||||
set (X86_DEV_ROOTFS "/" CACHE FILEPATH "Path to the sysroot used for cross-compiling for i686 and x86_64")
|
||||
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu" CACHE PATH "global data directory")
|
||||
set (DATA_DIRECTORY "" CACHE PATH "Global data directory (override)")
|
||||
if (NOT DATA_DIRECTORY)
|
||||
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu")
|
||||
endif()
|
||||
|
||||
string(FIND ${CMAKE_BASE_NAME} mingw CONTAINS_MINGW)
|
||||
if (NOT CONTAINS_MINGW EQUAL -1)
|
||||
@@ -94,7 +97,7 @@ endif()
|
||||
# uninstall target
|
||||
if(NOT TARGET uninstall)
|
||||
configure_file(
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/CMakeFiles/cmake_uninstall.cmake.in"
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/cmake_uninstall.cmake.in"
|
||||
"${CMAKE_CURRENT_BINARY_DIR}/CMakeFiles/cmake_uninstall.cmake"
|
||||
IMMEDIATE @ONLY)
|
||||
|
||||
@@ -414,7 +417,7 @@ if (TUNE_CPU STREQUAL "native")
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-march=native")
|
||||
endif()
|
||||
endif()
|
||||
else()
|
||||
elseif (NOT TUNE_CPU STREQUAL "none")
|
||||
check_cxx_compiler_flag("-mcpu=${TUNE_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
|
||||
if(COMPILER_SUPPORTS_CPU_TYPE)
|
||||
list(APPEND FEX_TUNE_COMPILE_FLAGS "-mcpu=${TUNE_CPU}")
|
||||
@@ -433,10 +436,6 @@ endif()
|
||||
|
||||
add_compile_options(-Wall)
|
||||
|
||||
configure_file(
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/include/Config.h.in
|
||||
${CMAKE_BINARY_DIR}/generated/ConfigDefines.h)
|
||||
|
||||
include(CTest)
|
||||
if (BUILD_TESTS)
|
||||
message(STATUS "Unit tests are enabled")
|
||||
@@ -619,12 +618,12 @@ set (CPACK_PACKAGE_CONTACT "FEX-Emu Maintainers <team@fex-emu.com>")
|
||||
set (CPACK_PACKAGE_VERSION_MAJOR "${FEX_VERSION_MAJOR}")
|
||||
set (CPACK_PACKAGE_VERSION_MINOR "${FEX_VERSION_MINOR}")
|
||||
set (CPACK_PACKAGE_VERSION_PATCH "${FEX_VERSION_PATCH}")
|
||||
set (CPACK_PACKAGE_DESCRIPTION_FILE "${CMAKE_CURRENT_SOURCE_DIR}/CPack/Description.txt")
|
||||
set (CPACK_PACKAGE_DESCRIPTION_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/Description.txt")
|
||||
|
||||
# Debian defines
|
||||
set (CPACK_DEBIAN_PACKAGE_DEPENDS "libc6, libstdc++6, libepoxy0, libsdl2-2.0-0, libegl1, libx11-6, squashfuse")
|
||||
set (CPACK_DEBIAN_PACKAGE_CONTROL_EXTRA
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/CPack/postinst;${CMAKE_CURRENT_SOURCE_DIR}/CPack/prerm;${CMAKE_CURRENT_SOURCE_DIR}/CPack/triggers")
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/postinst;${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/prerm;${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/triggers")
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
|
||||
# binfmt_misc conflicts with qemu-user-static
|
||||
# We also only install binfmt_misc on aarch64 hosts
|
||||
|
||||
@@ -1,3 +0,0 @@
|
||||
x86 and x86-64 Linux emulator
|
||||
|
||||
FEX is very much work in progress, so expect things to change.
|
||||
@@ -820,7 +820,6 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i32Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i16Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 0, ConvertedSize, 0b00010, rd, rn);
|
||||
@@ -856,7 +855,6 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i32Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i16Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 0, ConvertedSize, 0b00110, rd, rn);
|
||||
@@ -1195,7 +1193,6 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i32Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i16Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 1, ConvertedSize, 0b00010, rd, rn);
|
||||
@@ -1225,7 +1222,6 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i32Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i16Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 1, ConvertedSize, 0b00110, rd, rn);
|
||||
@@ -1322,9 +1318,7 @@ public:
|
||||
void fcvtxn(ARMEmitter::SubRegSize size, ARMEmitter::VRegister rd, ARMEmitter::VRegister rn) {
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::SubRegSize::i32Bit, "Only 32-bit subregsize supported");
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i16Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc(Op, 1, ConvertedSize, 0b10110, rd.D(), rn.D());
|
||||
}
|
||||
@@ -1333,9 +1327,7 @@ public:
|
||||
void fcvtxn2(ARMEmitter::SubRegSize size, ARMEmitter::VRegister rd, ARMEmitter::VRegister rn) {
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::SubRegSize::i32Bit, "Only 32-bit subregsize supported");
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i16Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc(Op, 1, ConvertedSize, 0b10110, rd.Q(), rn.Q());
|
||||
}
|
||||
@@ -1344,9 +1336,7 @@ public:
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::SubRegSize::i64Bit || size == ARMEmitter::SubRegSize::i32Bit, "Only 32-bit & 64-bit subregsize "
|
||||
"supported");
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 1, ConvertedSize, 0b11000, rd, rn);
|
||||
}
|
||||
@@ -1355,9 +1345,7 @@ public:
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::SubRegSize::i64Bit || size == ARMEmitter::SubRegSize::i32Bit, "Only 32-bit & 64-bit subregsize "
|
||||
"supported");
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 1, ConvertedSize, 0b11001, rd, rn);
|
||||
}
|
||||
@@ -1367,9 +1355,7 @@ public:
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::SubRegSize::i64Bit || size == ARMEmitter::SubRegSize::i32Bit, "Only 32-bit & 64-bit subregsize "
|
||||
"supported");
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 1, ConvertedSize, 0b11010, rd, rn);
|
||||
}
|
||||
@@ -1378,9 +1364,7 @@ public:
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::SubRegSize::i64Bit || size == ARMEmitter::SubRegSize::i32Bit, "Only 32-bit & 64-bit subregsize "
|
||||
"supported");
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 1, ConvertedSize, 0b11011, rd, rn);
|
||||
}
|
||||
@@ -1389,9 +1373,7 @@ public:
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::SubRegSize::i64Bit || size == ARMEmitter::SubRegSize::i32Bit, "Only 32-bit & 64-bit subregsize "
|
||||
"supported");
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 1, ConvertedSize, 0b11100, rd, rn);
|
||||
}
|
||||
@@ -1400,9 +1382,7 @@ public:
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::SubRegSize::i64Bit || size == ARMEmitter::SubRegSize::i32Bit, "Only 32-bit & 64-bit subregsize "
|
||||
"supported");
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 1, ConvertedSize, 0b11101, rd, rn);
|
||||
}
|
||||
@@ -1411,9 +1391,7 @@ public:
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::SubRegSize::i64Bit || size == ARMEmitter::SubRegSize::i32Bit, "Only 32-bit & 64-bit subregsize "
|
||||
"supported");
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 1, ConvertedSize, 0b11110, rd, rn);
|
||||
}
|
||||
@@ -1422,9 +1400,7 @@ public:
|
||||
LOGMAN_THROW_A_FMT(size == ARMEmitter::SubRegSize::i64Bit || size == ARMEmitter::SubRegSize::i32Bit, "Only 32-bit & 64-bit subregsize "
|
||||
"supported");
|
||||
constexpr uint32_t Op = 0b0000'1110'0010'0000'0000'10 << 10;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMD2RegMisc<T>(Op, 1, ConvertedSize, 0b11111, rd, rn);
|
||||
}
|
||||
@@ -1553,7 +1529,6 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0011'0000'0000'10 << 10;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i32Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i16Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMDAcrossLanes<T>(Op, 0, ConvertedSize, 0b00011, rd, rn);
|
||||
@@ -1597,7 +1572,6 @@ public:
|
||||
constexpr uint32_t Op = 0b0000'1110'0011'0000'0000'10 << 10;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i32Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i16Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
ASIMDAcrossLanes<T>(Op, 1, ConvertedSize, 0b00011, rd, rn);
|
||||
@@ -1630,10 +1604,7 @@ public:
|
||||
LOGMAN_THROW_A_FMT(size != ARMEmitter::SubRegSize::i8Bit && size != ARMEmitter::SubRegSize::i64Bit, "Destination 8/64-bit subregsize "
|
||||
"unsupported");
|
||||
constexpr uint32_t Op = 0b0000'1110'0011'0000'0000'10 << 10;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
size == ARMEmitter::SubRegSize::i16Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
const auto U = size == ARMEmitter::SubRegSize::i16Bit ? 0 : 1;
|
||||
|
||||
@@ -1647,10 +1618,7 @@ public:
|
||||
LOGMAN_THROW_A_FMT(size != ARMEmitter::SubRegSize::i8Bit && size != ARMEmitter::SubRegSize::i64Bit, "Destination 8/64-bit subregsize "
|
||||
"unsupported");
|
||||
constexpr uint32_t Op = 0b0000'1110'0011'0000'0000'10 << 10;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
size == ARMEmitter::SubRegSize::i16Bit ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ARMEmitter::SubRegSize::i8Bit;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i16Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
const auto U = size == ARMEmitter::SubRegSize::i16Bit ? 0 : 1;
|
||||
|
||||
@@ -1664,10 +1632,7 @@ public:
|
||||
LOGMAN_THROW_A_FMT(size != ARMEmitter::SubRegSize::i8Bit && size != ARMEmitter::SubRegSize::i64Bit, "Destination 8/64-bit subregsize "
|
||||
"unsupported");
|
||||
constexpr uint32_t Op = 0b0000'1110'0011'0000'0000'10 << 10;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i64Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i32Bit :
|
||||
size == ARMEmitter::SubRegSize::i16Bit ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ARMEmitter::SubRegSize::i32Bit;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i32Bit;
|
||||
|
||||
const auto U = size == ARMEmitter::SubRegSize::i16Bit ? 0 : 1;
|
||||
|
||||
@@ -1681,10 +1646,7 @@ public:
|
||||
LOGMAN_THROW_A_FMT(size != ARMEmitter::SubRegSize::i8Bit && size != ARMEmitter::SubRegSize::i64Bit, "Destination 8/64-bit subregsize "
|
||||
"unsupported");
|
||||
constexpr uint32_t Op = 0b0000'1110'0011'0000'0000'10 << 10;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i64Bit :
|
||||
size == ARMEmitter::SubRegSize::i32Bit ? ARMEmitter::SubRegSize::i32Bit :
|
||||
size == ARMEmitter::SubRegSize::i16Bit ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ARMEmitter::SubRegSize::i32Bit;
|
||||
const auto ConvertedSize = size == ARMEmitter::SubRegSize::i64Bit ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i32Bit;
|
||||
|
||||
const auto U = size == ARMEmitter::SubRegSize::i16Bit ? 0 : 1;
|
||||
|
||||
@@ -3952,7 +3914,7 @@ public:
|
||||
L = (Index >> 0) & 1;
|
||||
M = 0;
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(std::is_same_v<ARMEmitter::QRegister, T>, "Can't encode DRegister with i64Bit");
|
||||
LOGMAN_THROW_A_FMT((std::is_same_v<ARMEmitter::QRegister, T>), "Can't encode DRegister with i64Bit");
|
||||
// Index encoded in H
|
||||
H = Index;
|
||||
L = 0;
|
||||
@@ -3983,7 +3945,7 @@ public:
|
||||
L = (Index >> 0) & 1;
|
||||
M = 0;
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(std::is_same_v<ARMEmitter::QRegister, T>, "Can't encode DRegister with i64Bit");
|
||||
LOGMAN_THROW_A_FMT((std::is_same_v<ARMEmitter::QRegister, T>), "Can't encode DRegister with i64Bit");
|
||||
// Index encoded in H
|
||||
H = Index;
|
||||
L = 0;
|
||||
@@ -4014,7 +3976,7 @@ public:
|
||||
L = (Index >> 0) & 1;
|
||||
M = 0;
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(std::is_same_v<ARMEmitter::QRegister, T>, "Can't encode DRegister with i64Bit");
|
||||
LOGMAN_THROW_A_FMT((std::is_same_v<ARMEmitter::QRegister, T>), "Can't encode DRegister with i64Bit");
|
||||
// Index encoded in H
|
||||
H = Index;
|
||||
L = 0;
|
||||
|
||||
@@ -1648,8 +1648,8 @@ public:
|
||||
template<typename T>
|
||||
void ASIMDLoadStoreSinglePost(uint32_t Op, uint32_t Q, uint32_t L, uint32_t R, uint32_t opcode, uint32_t S, uint32_t size,
|
||||
ARMEmitter::Register rm, ARMEmitter::Register rn, T rt) {
|
||||
LOGMAN_THROW_A_FMT(std::is_same_v<ARMEmitter::QRegister, T> || std::is_same_v<ARMEmitter::DRegister, T>, "Only supports 128-bit and "
|
||||
"64-bit vector registers.");
|
||||
LOGMAN_THROW_A_FMT((std::is_same_v<ARMEmitter::QRegister, T> || std::is_same_v<ARMEmitter::DRegister, T>), "Only supports 128-bit and "
|
||||
"64-bit vector registers.");
|
||||
uint32_t Instr = Op;
|
||||
|
||||
Instr |= Q << 30;
|
||||
|
||||
File renamed without changes.
File renamed without changes.
File renamed without changes.
@@ -0,0 +1,3 @@
|
||||
x86 and x86-64 Linux emulator
|
||||
|
||||
FEX allows you to run x86 applications on ARM64 Linux devices. It offers broad compatibility with both 32-bit and 64-bit binaries, and it can be used alongside Wine/Proton to play Windows games.
|
||||
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
@@ -1,6 +1,6 @@
|
||||
# This is a reference AArch64 cross compile script
|
||||
# Pass in to cmake when building:
|
||||
# eg: cmake -DCMAKE_TOOLCHAIN_FILE=../CMakeToolchains/AArch64.cmake ..
|
||||
# eg: cmake --toolchain ../Data/CMake/toolchain_aarch64.cmake ..
|
||||
if (NOT DEFINED ENV{SYSROOT})
|
||||
message(FATAL_ERROR "Need to have SYSROOT environment variable set")
|
||||
endif()
|
||||
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
@@ -0,0 +1,35 @@
|
||||
{ pkgs ? import <nixpkgs> { } }:
|
||||
|
||||
let
|
||||
pkgsCross32 = pkgs.pkgsCross.gnu32;
|
||||
pkgsCross64 = pkgs.pkgsCross.gnu64;
|
||||
|
||||
gcc32 = pkgs.writeText "toolchain_nix_gcc_x86_32.txt" ''
|
||||
set(CMAKE_SYSTEM_PROCESSOR i686)
|
||||
set(CMAKE_C_COMPILER ${pkgsCross32.buildPackages.gcc}/bin/i686-unknown-linux-gnu-gcc)
|
||||
set(CMAKE_CXX_COMPILER ${pkgsCross32.buildPackages.gcc}/bin/i686-unknown-linux-gnu-g++)
|
||||
'';
|
||||
|
||||
gcc64 = pkgs.writeText "toolchain_nix_gcc_x86_64.txt" ''
|
||||
set(CMAKE_SYSTEM_PROCESSOR x86_64)
|
||||
set(CMAKE_C_COMPILER ${pkgsCross64.buildPackages.gcc}/bin/x86_64-unknown-linux-gnu-gcc)
|
||||
set(CMAKE_CXX_COMPILER ${pkgsCross64.buildPackages.gcc}/bin/x86_64-unknown-linux-gnu-g++)
|
||||
'';
|
||||
in
|
||||
pkgs.mkShell {
|
||||
buildInputs = [
|
||||
pkgsCross64.buildPackages.clang
|
||||
pkgsCross32.buildPackages.clang
|
||||
];
|
||||
|
||||
shellHook = ''
|
||||
if [[ $- == *i* ]]; then
|
||||
echo "toolchain32: ${gcc32}"
|
||||
echo "toolchain64: ${gcc64}"
|
||||
echo ""
|
||||
echo "Use \$FEX_CMAKE_TOOLCHAINS to configure CMake."
|
||||
fi
|
||||
'';
|
||||
|
||||
FEX_CMAKE_TOOLCHAINS = "-DX86_32_TOOLCHAIN_FILE=${gcc32} -DX86_64_TOOLCHAIN_FILE=${gcc64}";
|
||||
}
|
||||
@@ -0,0 +1,83 @@
|
||||
{ pkgs ? import <nixpkgs> { } }:
|
||||
|
||||
let
|
||||
pkgsCross32 = pkgs.pkgsCross.gnu32;
|
||||
pkgsCross64 = pkgs.pkgsCross.gnu64;
|
||||
|
||||
devRootFS = pkgs.buildEnv {
|
||||
name = "fex-dev-rootfs";
|
||||
paths = [
|
||||
pkgsCross64.stdenv.cc.libc_dev
|
||||
pkgsCross32.stdenv.cc.libc_dev
|
||||
pkgsCross64.stdenv.cc.cc
|
||||
pkgsCross32.stdenv.cc.cc
|
||||
|
||||
pkgs.alsa-lib.dev
|
||||
pkgs.libdrm.dev
|
||||
pkgs.libGL.dev
|
||||
pkgs.wayland.dev
|
||||
pkgs.xorg.libX11.dev
|
||||
pkgs.xorg.libxcb.dev
|
||||
pkgs.xorg.libXrandr.dev
|
||||
pkgs.xorg.libXrender.dev
|
||||
pkgs.xorg.xorgproto
|
||||
];
|
||||
ignoreCollisions = true;
|
||||
pathsToLink = [
|
||||
"/include"
|
||||
"/lib"
|
||||
];
|
||||
|
||||
postBuild = ''
|
||||
mkdir -p $out/usr
|
||||
ln -s $out/include $out/usr/
|
||||
'';
|
||||
};
|
||||
|
||||
toolchain32 = pkgs.writeText "toolchain_nix_x86_32.txt" ''
|
||||
set(CMAKE_EXE_LINKER_FLAGS_INIT "-fuse-ld=lld")
|
||||
set(CMAKE_MODULE_LINKER_FLAGS_INIT "-fuse-ld=lld")
|
||||
set(CMAKE_SHARED_LINKER_FLAGS_INIT "-fuse-ld=lld")
|
||||
set(CMAKE_SYSTEM_PROCESSOR i686)
|
||||
set(CMAKE_C_COMPILER clang)
|
||||
set(CMAKE_CXX_COMPILER clang++)
|
||||
set(CMAKE_C_COMPILER ${pkgsCross32.buildPackages.clang}/bin/i686-unknown-linux-gnu-clang)
|
||||
set(CMAKE_CXX_COMPILER ${pkgsCross32.buildPackages.clang}/bin/i686-unknown-linux-gnu-clang++)
|
||||
set(CLANG_FLAGS "-nodefaultlibs -nostartfiles -lstdc++ -target i686-linux-gnu -msse2 -mfpmath=sse --sysroot=${devRootFS} -iwithsysroot/include")
|
||||
set(CMAKE_C_FLAGS "''${CMAKE_C_FLAGS} ''${CLANG_FLAGS}")
|
||||
set(CMAKE_CXX_FLAGS "''${CMAKE_CXX_FLAGS} ''${CLANG_FLAGS}")
|
||||
'';
|
||||
|
||||
toolchain64 = pkgs.writeText "toolchain_nix_x86_64.txt" ''
|
||||
set(CMAKE_EXE_LINKER_FLAGS_INIT "-fuse-ld=lld")
|
||||
set(CMAKE_MODULE_LINKER_FLAGS_INIT "-fuse-ld=lld")
|
||||
set(CMAKE_SHARED_LINKER_FLAGS_INIT "-fuse-ld=lld")
|
||||
set(CMAKE_SYSTEM_PROCESSOR x86_64)
|
||||
set(CMAKE_C_COMPILER clang)
|
||||
set(CMAKE_CXX_COMPILER clang++)
|
||||
set(CMAKE_C_COMPILER ${pkgsCross64.buildPackages.clang}/bin/x86_64-unknown-linux-gnu-clang)
|
||||
set(CMAKE_CXX_COMPILER ${pkgsCross64.buildPackages.clang}/bin/x86_64-unknown-linux-gnu-clang++)
|
||||
set(CLANG_FLAGS "-nodefaultlibs -nostartfiles -lstdc++ -target x86_64-linux-gnu --sysroot=${devRootFS} -iwithsysroot/usr/include")
|
||||
set(CMAKE_C_FLAGS "''${CMAKE_C_FLAGS} ''${CLANG_FLAGS}")
|
||||
set(CMAKE_CXX_FLAGS "''${CMAKE_CXX_FLAGS} ''${CLANG_FLAGS}")
|
||||
'';
|
||||
in
|
||||
pkgs.mkShell {
|
||||
buildInputs = [
|
||||
pkgsCross64.buildPackages.clang
|
||||
pkgsCross32.buildPackages.clang
|
||||
];
|
||||
|
||||
shellHook = ''
|
||||
if [[ $- == *i* ]]; then
|
||||
echo "Set up dev RootFS at ${devRootFS}"
|
||||
echo "toolchain32: ${toolchain32}"
|
||||
echo "toolchain64: ${toolchain64}"
|
||||
echo ""
|
||||
echo "Use \$FEX_CMAKE_TOOLCHAINS to configure CMake."
|
||||
fi
|
||||
'';
|
||||
|
||||
FEX_CMAKE_TOOLCHAINS = "-DX86_32_TOOLCHAIN_FILE=${toolchain32} -DX86_64_TOOLCHAIN_FILE=${toolchain64} -DX86_DEV_ROOTFS=${devRootFS}";
|
||||
ROOTFS = "${devRootFS}";
|
||||
}
|
||||
@@ -0,0 +1,52 @@
|
||||
{ pkgs ? import <nixpkgs> { } }:
|
||||
|
||||
let
|
||||
toolchain = pkgs.fetchzip {
|
||||
url = "https://github.com/bylaws/llvm-mingw/releases/download/20250305/llvm-mingw-20250305-ucrt-ubuntu-20.04-aarch64.tar.xz";
|
||||
sha256 = "sha256-cA03/ab9O61eO9+S2JzIXD4V0HzTXK5/AYyxW2d73Po=";
|
||||
};
|
||||
|
||||
cmakeToolchainFile = pkgs.substitute {
|
||||
# Use absolute paths that are discoverable outside of the nix shell
|
||||
src = ../../CMake/toolchain_mingw.cmake;
|
||||
substitutions = ["--replace-fail" "\${MINGW_TRIPLE}-" "${toolchain}/bin/\${MINGW_TRIPLE}-"];
|
||||
};
|
||||
|
||||
mesonCrossFile = pkgs.writeText "crossfile_llvm_mingw.txt" ''
|
||||
[binaries]
|
||||
ar = '${toolchain}/bin/arm64ec-w64-mingw32-ar'
|
||||
c = '${toolchain}/bin/arm64ec-w64-mingw32-gcc'
|
||||
cpp = '${toolchain}/bin/arm64ec-w64-mingw32-g++'
|
||||
ld = '${toolchain}/bin/arm64ec-w64-mingw32-ld'
|
||||
windres = '${toolchain}/bin/arm64ec-w64-mingw32-windres'
|
||||
strip = '${toolchain}/bin/strip'
|
||||
widl = '${toolchain}/bin/arm64ec-w64-mingw32-widl'
|
||||
pkgconfig = 'aarch64-linux-gnu-pkg-config'
|
||||
[host_machine]
|
||||
system = 'windows'
|
||||
cpu_family = 'aarch64'
|
||||
cpu = 'aarch64'
|
||||
endian = 'little'
|
||||
'';
|
||||
in
|
||||
pkgs.mkShell {
|
||||
buildInputs = [
|
||||
toolchain
|
||||
];
|
||||
|
||||
shellHook = ''
|
||||
if [[ $- == *i* ]]; then
|
||||
echo "llvm-mingw set up at ${toolchain}."
|
||||
echo ""
|
||||
echo "To configure DXVK/vkd3d-proton: meson setup \$FEX_MESON_CROSSFILE"
|
||||
echo ""
|
||||
echo "To configure 32-bit FEX build: cmake \$FEX_CMAKE_TOOLCHAIN_WOW64"
|
||||
echo "To configure 64-bit FEX build: cmake \$FEX_CMAKE_TOOLCHAIN_ARM64EC"
|
||||
fi
|
||||
'';
|
||||
|
||||
# E.g. cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTS=False
|
||||
FEX_CMAKE_TOOLCHAIN_ARM64EC = "--toolchain ${cmakeToolchainFile} -DMINGW_TRIPLE=arm64ec-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows";
|
||||
FEX_CMAKE_TOOLCHAIN_WOW64 = "--toolchain ${cmakeToolchainFile} -DMINGW_TRIPLE=aarch64-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows";
|
||||
FEX_MESON_CROSSFILE = "--cross-file ${mesonCrossFile}";
|
||||
}
|
||||
Executable
+21
@@ -0,0 +1,21 @@
|
||||
#! /usr/bin/env nix-shell
|
||||
#! nix-shell -i bash WineOnArm/shell.nix
|
||||
|
||||
# Helper script to configure CMake for building FEX as library for emulation
|
||||
# of 32-bit applications in Wine/Proton.
|
||||
# The required cross-toolchains will be set up and managed by nix.
|
||||
|
||||
if [ $# -eq 0 ]
|
||||
then
|
||||
echo "Expected CMake argument list"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if [ -f CMakeCache.txt ]
|
||||
then
|
||||
echo "Expected empty build folder"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
set -o xtrace
|
||||
cmake $FEX_CMAKE_TOOLCHAIN_WOW64 -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTS=False $@
|
||||
Executable
+21
@@ -0,0 +1,21 @@
|
||||
#! /usr/bin/env nix-shell
|
||||
#! nix-shell -i bash WineOnArm/shell.nix
|
||||
|
||||
# Helper script to configure CMake for building FEX as library for emulation
|
||||
# of 64-bit applications in Wine/Proton
|
||||
# Nix is used to install and manage the required cross-toolchains.
|
||||
|
||||
if [ $# -eq 0 ]
|
||||
then
|
||||
echo "Expected CMake argument list"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if [ -f CMakeCache.txt ]
|
||||
then
|
||||
echo "Expected empty build folder"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
set -o xtrace
|
||||
cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTS=False $@
|
||||
Executable
+17
@@ -0,0 +1,17 @@
|
||||
#! /usr/bin/env nix-shell
|
||||
#! nix-shell -i bash FEXLinuxTests/shell.nix
|
||||
|
||||
# Helper script to configure CMake for building FEXLinuxTests.
|
||||
# Nix is used to install and manage the required cross-toolchains.
|
||||
|
||||
if [ ! -f CMakeCache.txt ]
|
||||
then
|
||||
echo "Must be run from a pre-configured CMake build folder"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Remove previous build to ensure the new toolchain is applied
|
||||
rm -rf unittests/FEXLinuxTests
|
||||
|
||||
set -o xtrace
|
||||
cmake . $FEX_CMAKE_TOOLCHAINS -DBUILD_TESTS=ON -DBUILD_FEX_LINUX_TESTS=ON
|
||||
Executable
+22
@@ -0,0 +1,22 @@
|
||||
# Helper script to configure CMake for library forwarding in FEX.
|
||||
# Nix is used to install and manage the required cross-toolchains.
|
||||
|
||||
if [ ! -f CMakeCache.txt ]
|
||||
then
|
||||
echo "Must be run from a pre-configured CMake build folder"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Remove previous build to ensure the new toolchain is applied
|
||||
rm -rf guest-libs guest-libs-32 Guest Guest_32
|
||||
|
||||
# Set clang executable path manually since the one from the nix store
|
||||
# will be picked up otherwise
|
||||
CLANG_EXEC_PATH=""
|
||||
if ! grep -q CLANG_EXEC_PATH CMakeCache.txt
|
||||
then
|
||||
CLANG_EXEC_PATH="-DCLANG_EXEC_PATH=`which clang`"
|
||||
fi
|
||||
|
||||
nix-shell `dirname -- "$0"`/LibraryForwarding/shell.nix \
|
||||
--run "set -o xtrace; cmake . \$FEX_CMAKE_TOOLCHAINS -DBUILD_THUNKS=ON $CLANG_EXEC_PATH; set +o xtrace"
|
||||
Vendored
+1
-1
Submodule External/tracy updated: 5d542dc09f...650c98ece7.
@@ -575,7 +575,7 @@ def print_ir_arg_printer():
|
||||
|
||||
if arg.IsSSA:
|
||||
# SSA value
|
||||
output_file.write("\tPrintArg(out, IR, Op->Header.Args[{}], RAData);\n".format(SSAArgNum))
|
||||
output_file.write("\tPrintArg(out, IR, Op->Header.Args[{}]);\n".format(SSAArgNum))
|
||||
SSAArgNum = SSAArgNum + 1
|
||||
else:
|
||||
# User defined op that is stored
|
||||
@@ -587,6 +587,15 @@ def print_ir_arg_printer():
|
||||
output_file.write("#undef IROP_ARGPRINTER_HELPER\n")
|
||||
output_file.write("#endif\n")
|
||||
|
||||
def print_validation(op):
|
||||
if op.EmitValidation != None:
|
||||
output_file.write("\t\t#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED\n")
|
||||
|
||||
for Validation in op.EmitValidation:
|
||||
Sanitized = Validation.replace("\"", "\\\"")
|
||||
output_file.write("\tLOGMAN_THROW_A_FMT({}, \"{}\");\n".format(Validation, Sanitized))
|
||||
output_file.write("\t\t#endif\n")
|
||||
|
||||
# Print out IR allocator helpers
|
||||
def print_ir_allocator_helpers():
|
||||
output_file.write("#ifdef IROP_ALLOCATE_HELPERS\n")
|
||||
@@ -678,7 +687,7 @@ def print_ir_allocator_helpers():
|
||||
output_file.write("{} {}".format(CType, arg.Name));
|
||||
elif arg.IsSSA:
|
||||
# SSA value
|
||||
output_file.write("OrderedNode *{}".format(arg.Name))
|
||||
output_file.write("OrderedNodeWrapper {}".format(arg.Name))
|
||||
else:
|
||||
# User defined op that is stored
|
||||
CType = IRTypesToCXX[arg.Type].CXXName
|
||||
@@ -708,35 +717,16 @@ def print_ir_allocator_helpers():
|
||||
output_file.write("\t\tauto _Op = AllocateOp<IROp_{}, IROps::OP_{}>();\n".format(op.Name, op.Name.upper()))
|
||||
|
||||
if op.SSAArgNum != 0:
|
||||
output_file.write("\t\tauto ListDataBegin = DualListData.ListBegin();\n")
|
||||
for arg in op.Arguments:
|
||||
if arg.IsSSA:
|
||||
output_file.write("\t\t_Op.first->{} = {}->Wrapped(ListDataBegin);\n".format(arg.Name, arg.Name))
|
||||
|
||||
if op.SSAArgNum != 0:
|
||||
for arg in op.Arguments:
|
||||
if arg.IsSSA:
|
||||
output_file.write("\t\t{}->AddUse();\n".format(arg.Name))
|
||||
output_file.write("\t\t_Op.first->{} = {};\n".format(arg.Name, arg.Name))
|
||||
|
||||
if len(op.Arguments) != 0:
|
||||
for arg in op.Arguments:
|
||||
if not arg.Temporary and not arg.IsSSA:
|
||||
output_file.write("\t\t_Op.first->{} = {};\n".format(arg.Name, arg.Name))
|
||||
|
||||
if (op.HasDest):
|
||||
# We can only infer a size if we have arguments
|
||||
if op.DestSize == None:
|
||||
# We need to infer destination size
|
||||
output_file.write("\t\tIR::OpSize InferSize = OpSize::iUnsized;\n")
|
||||
if len(op.Arguments) != 0:
|
||||
for arg in op.Arguments:
|
||||
if arg.IsSSA:
|
||||
output_file.write("\t\tauto Size{} = GetOpSize({});\n".format(arg.Name, arg.Name))
|
||||
for arg in op.Arguments:
|
||||
if arg.IsSSA:
|
||||
output_file.write("\t\tInferSize = std::max(InferSize, Size{});\n".format(arg.Name))
|
||||
|
||||
output_file.write("\t\t_Op.first->Header.Size = InferSize;\n")
|
||||
assert not (op.HasDest and op.DestSize is None)
|
||||
|
||||
# Some ops without a destination still need an operating size
|
||||
# Effectively reusing the destination size value for operation size
|
||||
@@ -748,18 +738,64 @@ def print_ir_allocator_helpers():
|
||||
else:
|
||||
output_file.write("\t\t_Op.first->Header.ElementSize = {};\n".format(op.ElementSize))
|
||||
|
||||
# Insert validation here
|
||||
if op.EmitValidation != None:
|
||||
output_file.write("\t\t#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED\n")
|
||||
|
||||
for Validation in op.EmitValidation:
|
||||
Sanitized = Validation.replace("\"", "\\\"")
|
||||
output_file.write("\tLOGMAN_THROW_A_FMT({}, \"{}\");\n".format(Validation, Sanitized))
|
||||
output_file.write("\t\t#endif\n")
|
||||
# Only validate here if there's no OrderedNode * version. Else
|
||||
# validation is in that version, see the comment below.
|
||||
if op.SSAArgNum == 0:
|
||||
print_validation(op)
|
||||
|
||||
output_file.write("\t\treturn _Op;\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
# Now do the OrderedNode * version if necessary
|
||||
if op.SSAArgNum:
|
||||
output_file.write("\tIRPair<IROp_{}> _{}(" .format(op.Name, op.Name))
|
||||
|
||||
for i in range(0, len(op.Arguments)):
|
||||
arg = op.Arguments[i]
|
||||
LastArg = len(op.Arguments) - i - 1 == 0
|
||||
|
||||
if arg.Temporary:
|
||||
CType = IRTypesToCXX[arg.Type].CXXName
|
||||
output_file.write("{} {}".format(CType, arg.Name));
|
||||
elif arg.IsSSA:
|
||||
output_file.write("OrderedNode *{}".format(arg.Name))
|
||||
else:
|
||||
CType = IRTypesToCXX[arg.Type].CXXName
|
||||
output_file.write("{} {}".format(CType, arg.Name));
|
||||
|
||||
if arg.DefaultInitializer != None:
|
||||
output_file.write(" = {}".format(arg.DefaultInitializer))
|
||||
|
||||
if not LastArg:
|
||||
output_file.write(", ")
|
||||
|
||||
output_file.write(") {\n")
|
||||
output_file.write("\t\tauto ListDataBegin = DualListData.ListBegin();\n")
|
||||
|
||||
for arg in op.Arguments:
|
||||
if arg.IsSSA:
|
||||
output_file.write("\t\t{}->AddUse();\n".format(arg.Name))
|
||||
|
||||
# Insert validation here. This is skipped for the
|
||||
# OrderedNodeWrapper version because validation can depend on
|
||||
# the OrderedNode, but that's ok in practice. Everything pre-RA
|
||||
# uses the OrderedNode version, and anything RA-onwards is
|
||||
# dubious to validate.
|
||||
print_validation(op)
|
||||
|
||||
output_file.write(f"\t\treturn _{op.Name}(")
|
||||
for i in range(0, len(op.Arguments)):
|
||||
arg = op.Arguments[i]
|
||||
LastArg = len(op.Arguments) - i - 1 == 0
|
||||
output_file.write(arg.Name)
|
||||
if arg.IsSSA:
|
||||
output_file.write("->Wrapped(ListDataBegin)")
|
||||
if not LastArg:
|
||||
output_file.write(", ")
|
||||
output_file.write(");\n");
|
||||
output_file.write("\t}\n\n");
|
||||
|
||||
output_file.write("#undef IROP_ALLOCATE_HELPERS\n")
|
||||
output_file.write("#endif\n")
|
||||
|
||||
|
||||
@@ -69,7 +69,6 @@ set (SRCS
|
||||
Interface/IR/Passes/ConstProp.cpp
|
||||
Interface/IR/Passes/IRDumperPass.cpp
|
||||
Interface/IR/Passes/IRValidation.cpp
|
||||
Interface/IR/Passes/RAValidation.cpp
|
||||
Interface/IR/Passes/RedundantFlagCalculationElimination.cpp
|
||||
Interface/IR/Passes/RegisterAllocationPass.cpp
|
||||
Interface/IR/Passes/x87StackOptimizationPass.cpp
|
||||
|
||||
@@ -3,14 +3,32 @@
|
||||
|
||||
#ifdef _M_X86_64
|
||||
#include <xmmintrin.h>
|
||||
#include <immintrin.h>
|
||||
#endif
|
||||
|
||||
namespace FEXCore {
|
||||
struct VectorScalarF64Pair {
|
||||
double val[2];
|
||||
};
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
// Can't use uint8x16_t directly from arm_neon.h here.
|
||||
// Overrides softfloat-3e's defines which causes problems.
|
||||
using VectorRegType = __attribute__((neon_vector_type(16))) uint8_t;
|
||||
struct VectorRegPairType {
|
||||
VectorRegType val[2];
|
||||
};
|
||||
|
||||
static inline VectorRegPairType MakeVectorRegPair(VectorRegType low, VectorRegType high) {
|
||||
return VectorRegPairType {low, high};
|
||||
}
|
||||
|
||||
#elif defined(_M_X86_64)
|
||||
using VectorRegType = __m128i;
|
||||
using VectorRegPairType = __m256i;
|
||||
|
||||
static inline VectorRegPairType MakeVectorRegPair(VectorRegType low, VectorRegType high) {
|
||||
return _mm256_set_m128i(high, low);
|
||||
}
|
||||
#endif
|
||||
} // namespace FEXCore
|
||||
@@ -118,7 +118,7 @@
|
||||
},
|
||||
"ThunkHostLibs": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/@CMAKE_INSTALL_LIBDIR@/fex-emu/HostThunks/",
|
||||
"Default": "@CMAKE_INSTALL_FULL_LIBDIR@/fex-emu/HostThunks",
|
||||
"ShortArg": "t",
|
||||
"Desc": [
|
||||
"Folder to find the host-side thunking libraries."
|
||||
@@ -126,26 +126,12 @@
|
||||
},
|
||||
"ThunkGuestLibs": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/share/fex-emu/GuestThunks/",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/share/fex-emu/GuestThunks",
|
||||
"ShortArg": "j",
|
||||
"Desc": [
|
||||
"Folder to find the guest-side thunking libraries."
|
||||
]
|
||||
},
|
||||
"ThunkHostLibs32": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/@CMAKE_INSTALL_LIBDIR@/fex-emu/HostThunks_32/",
|
||||
"Desc": [
|
||||
"Folder to find the 32-bit host-side thunking libraries."
|
||||
]
|
||||
},
|
||||
"ThunkGuestLibs32": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/share/fex-emu/GuestThunks_32/",
|
||||
"Desc": [
|
||||
"Folder to find the 32-bit guest-side thunking libraries."
|
||||
]
|
||||
},
|
||||
"ThunkConfig": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
|
||||
@@ -50,7 +50,6 @@ namespace HLE {
|
||||
} // namespace FEXCore
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class RegisterAllocationData;
|
||||
struct IRListCopy;
|
||||
class IRListView;
|
||||
namespace Validation {
|
||||
@@ -76,7 +75,7 @@ struct CustomIRResult {
|
||||
using BlockDelinkerFunc = void (*)(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record);
|
||||
constexpr uint32_t TSC_SCALE_MAXIMUM = 1'000'000'000; ///< 1Ghz
|
||||
|
||||
class ContextImpl final : public FEXCore::Context::Context {
|
||||
class ContextImpl final : public FEXCore::Context::Context, CPU::CodeBufferManager {
|
||||
public:
|
||||
// Context base class implementation.
|
||||
bool InitCore() override;
|
||||
@@ -165,7 +164,8 @@ public:
|
||||
IRCaptureCache.WriteFilesWithCode(Writer);
|
||||
}
|
||||
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
void OnCodeBufferAllocated(CPU::CodeBuffer&) override;
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer = true) override;
|
||||
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) override;
|
||||
FEXCore::ForkableSharedMutex& GetCodeInvalidationMutex() override {
|
||||
return CodeInvalidationMutex;
|
||||
@@ -264,6 +264,10 @@ public:
|
||||
auto Thread = Frame->Thread;
|
||||
auto lk = GuardSignalDeferringSection(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
// NOTE: Other threads sharing the same CodeBuffer may reference
|
||||
// invalidated data ranges through their L1/L2 caches. This is
|
||||
// not currently a problem since FEX does not repurpose the
|
||||
// invalidated CodeBuffer memory range currently.
|
||||
ThreadRemoveCodeEntry(Thread, GuestRIP);
|
||||
}
|
||||
|
||||
@@ -271,7 +275,6 @@ public:
|
||||
|
||||
struct GenerateIRResult {
|
||||
std::optional<IR::IRListView> IRView;
|
||||
IR::RegisterAllocationData* RAData;
|
||||
uint64_t TotalInstructions;
|
||||
uint64_t TotalInstructionsLength;
|
||||
uint64_t StartAddr;
|
||||
@@ -299,6 +302,7 @@ public:
|
||||
|
||||
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator;
|
||||
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator;
|
||||
FEXCore::Utils::PooledAllocatorVirtual CPUBackendAllocator;
|
||||
|
||||
// If Atomic-based TSO emulation is enabled or not.
|
||||
bool IsAtomicTSOEnabled() const {
|
||||
|
||||
@@ -34,7 +34,15 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, b
|
||||
//
|
||||
// If the AddrSize is not the GPRSize then we need to clear the upper bits.
|
||||
if ((A.AddrSize < GPRSize) && !AllowUpperGarbage && Tmp) {
|
||||
Tmp = IREmit->_Bfe(GPRSize, IR::OpSizeAsBits(A.AddrSize), 0, Tmp);
|
||||
uint32_t Bits = IR::OpSizeAsBits(A.AddrSize);
|
||||
|
||||
if (A.Base || A.Index) {
|
||||
Tmp = IREmit->_Bfe(GPRSize, Bits, 0, Tmp);
|
||||
} else if (A.Offset) {
|
||||
uint64_t X = A.Offset;
|
||||
X &= (1ull << Bits) - 1;
|
||||
Tmp = IREmit->_Constant(X);
|
||||
}
|
||||
}
|
||||
|
||||
if (A.Segment && AddSegmentBase) {
|
||||
@@ -155,4 +163,4 @@ AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSi
|
||||
}
|
||||
|
||||
|
||||
}; // namespace FEXCore::IR
|
||||
}; // namespace FEXCore::IR
|
||||
@@ -501,7 +501,10 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
|
||||
|
||||
// If the aligned offset is within the 4GB window then we can use ADRP+ADD
|
||||
// and the number of move segments more than 1
|
||||
if (RequiredMoveSegments > 1 && ARMEmitter::Emitter::IsInt32(AlignedOffset)) {
|
||||
// NOTE: JIT output is moved to a different buffer after compilation, so the
|
||||
// current cursor address doesn't match the runtime instruction address.
|
||||
// Hence this optimization is disabled until we enable code relocation patches.
|
||||
if (RequiredMoveSegments > 1 && ARMEmitter::Emitter::IsInt32(AlignedOffset) && false) {
|
||||
// If this is 4k page aligned then we only need ADRP
|
||||
if ((AlignedOffset & 0xFFF) == 0) {
|
||||
adrp(Reg, AlignedOffset >> 12);
|
||||
@@ -590,8 +593,8 @@ void Arm64Emitter::PushCalleeSavedRegisters() {
|
||||
|
||||
void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
constexpr static std::array< std::tuple<ARMEmitter::DRegister, ARMEmitter::DRegister, ARMEmitter::DRegister, ARMEmitter::DRegister>, 2> FPRs = {{
|
||||
{ARMEmitter::DReg::d12, ARMEmitter::DReg::d13, ARMEmitter::DReg::d14, ARMEmitter::DReg::d15},
|
||||
{ARMEmitter::DReg::d8, ARMEmitter::DReg::d9, ARMEmitter::DReg::d10, ARMEmitter::DReg::d11},
|
||||
{ARMEmitter::DReg::d12, ARMEmitter::DReg::d13, ARMEmitter::DReg::d14, ARMEmitter::DReg::d15},
|
||||
}};
|
||||
|
||||
for (auto& RegQuad : FPRs) {
|
||||
|
||||
@@ -6,6 +6,8 @@
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include <cstdint>
|
||||
|
||||
#include "LookupCache.h"
|
||||
|
||||
#ifndef _WIN32
|
||||
#include <sys/prctl.h>
|
||||
#endif
|
||||
@@ -13,6 +15,10 @@
|
||||
namespace FEXCore {
|
||||
namespace CPU {
|
||||
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
// We don't want to move above 128MB atm because that means we will have to encode longer jumps
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
|
||||
|
||||
constexpr static uint64_t NamedVectorConstants[FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_CONST_POOL_MAX][2] = {
|
||||
{0x0003'0002'0001'0000ULL, 0x0007'0006'0005'0004ULL}, // NAMED_VECTOR_INCREMENTAL_U16_INDEX
|
||||
{0x000B'000A'0009'0008ULL, 0x000F'000E'000D'000CULL}, // NAMED_VECTOR_INCREMENTAL_U16_INDEX_UPPER
|
||||
@@ -264,10 +270,9 @@ namespace CPU {
|
||||
return TotalLUT;
|
||||
}()};
|
||||
|
||||
CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState* ThreadState, size_t InitialCodeSize, size_t MaxCodeSize)
|
||||
CPUBackend::CPUBackend(CodeBufferManager& CodeBuffers, FEXCore::Core::InternalThreadState* ThreadState)
|
||||
: ThreadState(ThreadState)
|
||||
, InitialCodeSize(InitialCodeSize)
|
||||
, MaxCodeSize(MaxCodeSize) {
|
||||
, CodeBuffers(CodeBuffers) {
|
||||
|
||||
auto& Common = ThreadState->CurrentFrame->Pointers.Common;
|
||||
|
||||
@@ -304,52 +309,63 @@ namespace CPU {
|
||||
#endif
|
||||
}
|
||||
|
||||
CPUBackend::~CPUBackend() {
|
||||
for (auto CodeBuffer : CodeBuffers) {
|
||||
FreeCodeBuffer(CodeBuffer);
|
||||
}
|
||||
CodeBuffers.clear();
|
||||
}
|
||||
CPUBackend::~CPUBackend() = default;
|
||||
|
||||
auto CPUBackend::GetEmptyCodeBuffer() -> CodeBuffer* {
|
||||
if (ThreadState->CurrentFrame->SignalHandlerRefCounter == 0) {
|
||||
if (CodeBuffers.empty()) {
|
||||
auto NewCodeBuffer = AllocateNewCodeBuffer(InitialCodeSize);
|
||||
EmplaceNewCodeBuffer(NewCodeBuffer);
|
||||
} else {
|
||||
if (CodeBuffers.size() > 1) {
|
||||
// If we have more than one code buffer we are tracking then walk them and delete
|
||||
// This is a cleanup step
|
||||
for (size_t i = 1; i < CodeBuffers.size(); i++) {
|
||||
FreeCodeBuffer(CodeBuffers[i]);
|
||||
}
|
||||
CodeBuffers.resize(1);
|
||||
}
|
||||
// Set the current code buffer to the initial
|
||||
CurrentCodeBuffer = CodeBuffers.data();
|
||||
auto PrevCodeBuffer = CurrentCodeBuffer;
|
||||
|
||||
if (CurrentCodeBuffer->Size != MaxCodeSize) {
|
||||
FreeCodeBuffer(*CurrentCodeBuffer);
|
||||
// Resize the code buffer and reallocate our code size
|
||||
CurrentCodeBuffer = CodeBuffers.StartLargerCodeBuffer();
|
||||
|
||||
// Resize the code buffer and reallocate our code size
|
||||
CurrentCodeBuffer->Size *= 1.5;
|
||||
CurrentCodeBuffer->Size = std::min(CurrentCodeBuffer->Size, MaxCodeSize);
|
||||
|
||||
*CurrentCodeBuffer = AllocateNewCodeBuffer(CurrentCodeBuffer->Size);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// We have signal handlers that have generated code
|
||||
// This means that we can not safely clear the code at this point in time
|
||||
// Allocate some new code buffers that we can switch over to instead
|
||||
auto NewCodeBuffer = AllocateNewCodeBuffer(InitialCodeSize);
|
||||
EmplaceNewCodeBuffer(NewCodeBuffer);
|
||||
}
|
||||
|
||||
return CurrentCodeBuffer;
|
||||
RegisterForSignalHandler(PrevCodeBuffer);
|
||||
return CurrentCodeBuffer.get();
|
||||
}
|
||||
|
||||
auto CPUBackend::AllocateNewCodeBuffer(size_t Size) -> CodeBuffer {
|
||||
void CPUBackend::RegisterForSignalHandler(fextl::shared_ptr<CodeBuffer> CodeBuffer) {
|
||||
if (ThreadState->CurrentFrame->SignalHandlerRefCounter != 0) {
|
||||
// We have signal handlers that have generated code
|
||||
// This means that we can not safely clear the code at this point in time
|
||||
// Keep a reference to the old code buffer to delay deallocation
|
||||
SignalHandlerCodeBuffers.push_back(CodeBuffer);
|
||||
} else {
|
||||
SignalHandlerCodeBuffers.clear();
|
||||
}
|
||||
}
|
||||
|
||||
fextl::shared_ptr<CodeBuffer> CPUBackend::CheckCodeBufferUpdate() {
|
||||
fextl::shared_ptr<CodeBuffer> OldCodeBuffer;
|
||||
auto NewCodeBuffer = CodeBuffers.GetLatest();
|
||||
if (CurrentCodeBuffer != NewCodeBuffer) {
|
||||
RegisterForSignalHandler(CurrentCodeBuffer);
|
||||
return std::exchange(CurrentCodeBuffer, NewCodeBuffer);
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
GuestToHostMap& GetLookupCache(const CodeBuffer& Buffer) {
|
||||
return *Buffer.LookupCache;
|
||||
}
|
||||
|
||||
CodeBuffer::CodeBuffer(size_t Size)
|
||||
: Size(Size) {
|
||||
Ptr = static_cast<uint8_t*>(FEXCore::Allocator::VirtualAlloc(Size, true));
|
||||
LOGMAN_THROW_A_FMT(!!Ptr, "Couldn't allocate code buffer");
|
||||
|
||||
// Protect the last page of the allocated buffer to trigger SIGSEGV on write access
|
||||
uintptr_t LastPageAddr = AlignDown(reinterpret_cast<uintptr_t>(Ptr) + Size - 1, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
if (!FEXCore::Allocator::VirtualProtect(reinterpret_cast<void*>(LastPageAddr), FEXCore::Utils::FEX_PAGE_SIZE,
|
||||
FEXCore::Allocator::ProtectOptions::None)) {
|
||||
LogMan::Msg::EFmt("Failed to mprotect last page of code buffer.");
|
||||
}
|
||||
|
||||
LookupCache = fextl::make_unique<GuestToHostMap>();
|
||||
}
|
||||
|
||||
CodeBuffer::~CodeBuffer() {
|
||||
FEXCore::Allocator::VirtualFree(Ptr, Size);
|
||||
}
|
||||
|
||||
auto CodeBufferManager::AllocateNew(size_t Size) -> fextl::shared_ptr<CodeBuffer> {
|
||||
#ifndef _WIN32
|
||||
// MDWE (Memory-Deny-Write-Execute) is a new Linux 6.3 feature.
|
||||
// It's equivalent to systemd's `MemoryDenyWriteExecute` but implemented entirely in the kernel.
|
||||
@@ -375,39 +391,51 @@ namespace CPU {
|
||||
}
|
||||
#endif
|
||||
|
||||
CodeBuffer Buffer;
|
||||
Buffer.Size = Size;
|
||||
Buffer.Ptr = static_cast<uint8_t*>(FEXCore::Allocator::VirtualAlloc(Buffer.Size, true));
|
||||
LOGMAN_THROW_A_FMT(!!Buffer.Ptr, "Couldn't allocate code buffer");
|
||||
auto Buffer = fextl::make_shared<CodeBuffer>(Size);
|
||||
|
||||
if (static_cast<Context::ContextImpl*>(ThreadState->CTX)->Config.GlobalJITNaming()) {
|
||||
static_cast<Context::ContextImpl*>(ThreadState->CTX)->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
Latest = Buffer;
|
||||
LatestOffset = 0;
|
||||
|
||||
// Protect the last page of the allocated buffer to trigger SIGSEGV on write access
|
||||
uintptr_t LastPageAddr = AlignDown(reinterpret_cast<uintptr_t>(Buffer.Ptr) + Buffer.Size - 1, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
if (!FEXCore::Allocator::VirtualProtect(reinterpret_cast<void*>(LastPageAddr), FEXCore::Utils::FEX_PAGE_SIZE,
|
||||
FEXCore::Allocator::ProtectOptions::None)) {
|
||||
LogMan::Msg::EFmt("Failed to mprotect last page of code buffer.");
|
||||
}
|
||||
OnCodeBufferAllocated(*Buffer);
|
||||
|
||||
return Buffer;
|
||||
}
|
||||
|
||||
void CPUBackend::FreeCodeBuffer(CodeBuffer Buffer) {
|
||||
FEXCore::Allocator::VirtualFree(Buffer.Ptr, Buffer.Size);
|
||||
fextl::shared_ptr<CodeBuffer> CodeBufferManager::GetLatest() {
|
||||
if (!Latest) {
|
||||
AllocateNew(INITIAL_CODE_SIZE);
|
||||
}
|
||||
return Latest;
|
||||
}
|
||||
|
||||
fextl::shared_ptr<CodeBuffer> CodeBufferManager::StartLargerCodeBuffer() {
|
||||
if (!Latest) {
|
||||
// Allocate initial CodeBuffer and return it
|
||||
return GetLatest();
|
||||
}
|
||||
|
||||
auto NewCodeBufferSize = GetLatest()->Size;
|
||||
NewCodeBufferSize = std::min<size_t>(NewCodeBufferSize * 2, MAX_CODE_SIZE);
|
||||
return AllocateNew(NewCodeBufferSize);
|
||||
}
|
||||
|
||||
|
||||
bool CPUBackend::IsAddressInCodeBuffer(uintptr_t Address) const {
|
||||
// The last page of the code buffer is protected, so we need to exclude it from the valid range
|
||||
// when checking if the address is in the code buffer.
|
||||
for (auto& Buffer : CodeBuffers) {
|
||||
auto CheckCodeBuffer = [](CodeBuffer& Buffer, uintptr_t Address) {
|
||||
// The last page of the code buffer is protected, so we need to exclude it from the valid range
|
||||
// when checking if the address is in the code buffer.
|
||||
uintptr_t LastPageAddr = AlignDown(reinterpret_cast<uintptr_t>(Buffer.Ptr) + Buffer.Size - 1, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
if (Address >= reinterpret_cast<uintptr_t>(Buffer.Ptr) && Address < LastPageAddr) {
|
||||
return (Address >= reinterpret_cast<uintptr_t>(Buffer.Ptr) && Address < LastPageAddr);
|
||||
};
|
||||
|
||||
if (CheckCodeBuffer(*CurrentCodeBuffer, Address)) {
|
||||
return true;
|
||||
}
|
||||
for (auto& Buffer : SignalHandlerCodeBuffers) {
|
||||
if (CheckCodeBuffer(*Buffer, Address)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
@@ -9,6 +9,8 @@ $end_info$
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
@@ -18,7 +20,6 @@ namespace FEXCore {
|
||||
|
||||
namespace IR {
|
||||
class IRListView;
|
||||
class RegisterAllocationData;
|
||||
} // namespace IR
|
||||
|
||||
namespace Core {
|
||||
@@ -32,19 +33,61 @@ namespace CodeSerialize {
|
||||
struct CodeObjectFileSection;
|
||||
}
|
||||
|
||||
struct GuestToHostMap;
|
||||
|
||||
namespace CPU {
|
||||
struct CodeBuffer {
|
||||
uint8_t* Ptr;
|
||||
size_t Size;
|
||||
|
||||
fextl::unique_ptr<GuestToHostMap> LookupCache;
|
||||
|
||||
CodeBuffer(size_t Size);
|
||||
CodeBuffer(const CodeBuffer&) = delete;
|
||||
CodeBuffer& operator=(const CodeBuffer&) = delete;
|
||||
CodeBuffer(CodeBuffer&& oth) = delete;
|
||||
CodeBuffer& operator=(CodeBuffer&&) = delete;
|
||||
|
||||
~CodeBuffer();
|
||||
};
|
||||
|
||||
/**
|
||||
* A manager that coordinates access to the CodeBuffer used for compiling new code across threads.
|
||||
*
|
||||
* The CodeBuffer is managed as a partially persistent data structure:
|
||||
* - Exactly one CodeBuffer is now designated as "active", which means data can be appended to it
|
||||
* - Lossy modifications to the active CodeBuffer will not invalidate any data in use by other threads (which is what enables save CodeBuffer sharing across threads)
|
||||
* - Instead, such lossy modifications trigger a new "version" of the data in the modifying thread. Old versions of the CodeBuffer persist as read-only data for use by the other threads.
|
||||
* - The other threads can update their version of the CodeBuffer. This will decrease the reference count and eventually trigger deallocation of the old version
|
||||
*/
|
||||
class CodeBufferManager {
|
||||
public:
|
||||
// Get the CodeBuffer that was most recently allocated.
|
||||
// This is the only CodeBuffer that data may be written to.
|
||||
fextl::shared_ptr<CodeBuffer> GetLatest();
|
||||
|
||||
// Allocate a new CodeBuffer with geometric growth up to an internal maximum.
|
||||
// Subsequent calls to GetLatest will point to the returned buffer.
|
||||
fextl::shared_ptr<CodeBuffer> StartLargerCodeBuffer();
|
||||
|
||||
// Write offset into the latest CodeBuffer
|
||||
std::size_t LatestOffset {};
|
||||
|
||||
// Protects writes to the latest CodeBuffer and changes to LatestOffset
|
||||
FEXCore::ForkableUniqueMutex CodeBufferWriteMutex;
|
||||
|
||||
virtual void OnCodeBufferAllocated(CodeBuffer&) {};
|
||||
|
||||
private:
|
||||
fextl::shared_ptr<CodeBuffer> Latest;
|
||||
|
||||
fextl::shared_ptr<CodeBuffer> AllocateNew(size_t Size);
|
||||
};
|
||||
|
||||
class CPUBackend {
|
||||
public:
|
||||
struct CodeBuffer {
|
||||
uint8_t* Ptr;
|
||||
size_t Size;
|
||||
};
|
||||
|
||||
/**
|
||||
* @param InitialCodeSize - Initial size for the code buffers
|
||||
* @param MaxCodeSize - Max size for the code buffers
|
||||
*/
|
||||
CPUBackend(FEXCore::Core::InternalThreadState* ThreadState, size_t InitialCodeSize, size_t MaxCodeSize);
|
||||
CPUBackend(CodeBufferManager&, FEXCore::Core::InternalThreadState*);
|
||||
|
||||
virtual ~CPUBackend();
|
||||
|
||||
@@ -119,7 +162,7 @@ namespace CPU {
|
||||
*/
|
||||
[[nodiscard]]
|
||||
virtual CompiledCode CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, const FEXCore::IR::RegisterAllocationData* RAData, bool CheckTF) = 0;
|
||||
FEXCore::Core::DebugData* DebugData, bool CheckTF) = 0;
|
||||
|
||||
/**
|
||||
* @brief Relocates a block of code from the JIT code object cache
|
||||
@@ -143,6 +186,11 @@ namespace CPU {
|
||||
|
||||
bool IsAddressInCodeBuffer(uintptr_t Address) const;
|
||||
|
||||
// Updates the CodeBuffer if needed and returns a reference to the old one.
|
||||
// The returned reference should be kept alive carefully to avoid early deletion of resources.
|
||||
[[nodiscard]]
|
||||
fextl::shared_ptr<CodeBuffer> CheckCodeBufferUpdate();
|
||||
|
||||
protected:
|
||||
// Max spill slot size in bytes. We need at most 32 bytes
|
||||
// to be able to handle a 256-bit vector store to a slot.
|
||||
@@ -150,24 +198,20 @@ namespace CPU {
|
||||
|
||||
FEXCore::Core::InternalThreadState* ThreadState;
|
||||
|
||||
size_t InitialCodeSize, MaxCodeSize;
|
||||
[[nodiscard]]
|
||||
CodeBuffer* GetEmptyCodeBuffer();
|
||||
|
||||
// This is the current code buffer that we are tracking
|
||||
CodeBuffer* CurrentCodeBuffer {};
|
||||
// This is the code buffer containing the main code under execution by this thread.
|
||||
// CheckCodeBufferUpdate must be used before compiling new code.
|
||||
fextl::shared_ptr<CodeBuffer> CurrentCodeBuffer;
|
||||
|
||||
// Old CodeBuffer generations required to be valid until returning from signal handlers
|
||||
fextl::vector<fextl::shared_ptr<CodeBuffer>> SignalHandlerCodeBuffers;
|
||||
|
||||
CodeBufferManager& CodeBuffers;
|
||||
|
||||
private:
|
||||
CodeBuffer AllocateNewCodeBuffer(size_t Size);
|
||||
void FreeCodeBuffer(CodeBuffer Buffer);
|
||||
|
||||
void EmplaceNewCodeBuffer(CodeBuffer Buffer) {
|
||||
CurrentCodeBuffer = &CodeBuffers.emplace_back(Buffer);
|
||||
}
|
||||
|
||||
// This is the array of code buffers. Unless signals force us to keep more than
|
||||
// buffer, there will be only one entry here
|
||||
fextl::vector<CodeBuffer> CodeBuffers {};
|
||||
void RegisterForSignalHandler(fextl::shared_ptr<CodeBuffer>);
|
||||
};
|
||||
|
||||
} // namespace CPU
|
||||
|
||||
@@ -115,7 +115,7 @@ public:
|
||||
|
||||
private:
|
||||
const FEXCore::Context::ContextImpl* CTX;
|
||||
bool SupportsCPUIndexInTPIDRRO {};
|
||||
[[maybe_unused]] bool SupportsCPUIndexInTPIDRRO {};
|
||||
bool Hybrid {};
|
||||
uint32_t Cores {};
|
||||
FEX_CONFIG_OPT(HideHypervisorBit, HIDEHYPERVISORBIT);
|
||||
|
||||
@@ -426,7 +426,7 @@ void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread)
|
||||
Thread->PassManager->RegisterSyscallHandler(SyscallHandler);
|
||||
|
||||
// Create CPU backend
|
||||
Thread->PassManager->InsertRegisterAllocationPass();
|
||||
Thread->PassManager->InsertRegisterAllocationPass(this);
|
||||
Thread->CPUBackend = FEXCore::CPU::CreateArm64JITCore(this, Thread);
|
||||
|
||||
Thread->PassManager->Finalize();
|
||||
@@ -495,7 +495,13 @@ void ContextImpl::LockBeforeFork(FEXCore::Core::InternalThreadState* Thread) {
|
||||
}
|
||||
#endif
|
||||
|
||||
void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread) {
|
||||
void ContextImpl::OnCodeBufferAllocated(CPU::CodeBuffer& Buffer) {
|
||||
if (Config.GlobalJITNaming()) {
|
||||
Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer) {
|
||||
FEXCORE_PROFILE_INSTANT("ClearCodeCache");
|
||||
|
||||
if (CodeObjectCacheService) {
|
||||
@@ -503,18 +509,22 @@ void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread) {
|
||||
// Use the thread's object cache ref counter for this
|
||||
CodeSerialize::CodeObjectSerializeService::WaitForEmptyJobQueue(&Thread->ObjectCacheRefCounter);
|
||||
}
|
||||
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
|
||||
|
||||
Thread->LookupCache->ClearCache();
|
||||
Thread->CPUBackend->ClearCache();
|
||||
if (NewCodeBuffer) {
|
||||
// Allocate new CodeBuffer + L3 LookupCache and clear L1+L2 caches
|
||||
Thread->CPUBackend->ClearCache();
|
||||
} else {
|
||||
// Clear L1+L2 cache of this thread, and clear L3 cache across any threads using it
|
||||
Thread->LookupCache->ClearCache();
|
||||
}
|
||||
}
|
||||
|
||||
static void IRDumper(FEXCore::Core::InternalThreadState* Thread, IR::IREmitter* IREmitter, uint64_t GuestRIP, IR::RegisterAllocationData* RA) {
|
||||
static void IRDumper(FEXCore::Core::InternalThreadState* Thread, IR::IREmitter* IREmitter, uint64_t GuestRIP) {
|
||||
FEXCore::File::File FD = FEXCore::File::File::GetStdERR();
|
||||
fextl::stringstream out;
|
||||
auto NewIR = IREmitter->ViewIR();
|
||||
FEXCore::IR::Dump(&out, &NewIR, RA);
|
||||
fextl::fmt::print(FD, "IR-ShouldDump-{} 0x{:x}:\n{}\n@@@@@\n", RA ? "post" : "pre", GuestRIP, out.str());
|
||||
FEXCore::IR::Dump(&out, &NewIR);
|
||||
fextl::fmt::print(FD, "IR-ShouldDump-{} 0x{:x}:\n{}\n@@@@@\n", NewIR.PostRA() ? "post" : "pre", GuestRIP, out.str());
|
||||
};
|
||||
|
||||
ContextImpl::GenerateIRResult
|
||||
@@ -686,7 +696,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
if (HadDispatchError && TotalInstructions == 0) {
|
||||
// Couldn't handle any instruction in op dispatcher
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
return {{}, nullptr, 0, 0, 0, 0};
|
||||
return {{}, 0, 0, 0, 0};
|
||||
}
|
||||
|
||||
if (NeedsBlockEnd) {
|
||||
@@ -712,22 +722,19 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
auto ShouldDump = Thread->OpDispatcher->ShouldDumpIR();
|
||||
// Debug
|
||||
if (ShouldDump) {
|
||||
IRDumper(Thread, IREmitter, GuestRIP, nullptr);
|
||||
IRDumper(Thread, IREmitter, GuestRIP);
|
||||
}
|
||||
|
||||
// Run the passmanager over the IR from the dispatcher
|
||||
Thread->PassManager->Run(IREmitter);
|
||||
|
||||
auto RAData = Thread->PassManager->HasPass("RA") ? Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA")->GetAllocationData() : nullptr;
|
||||
|
||||
// Debug
|
||||
if (ShouldDump) {
|
||||
IRDumper(Thread, IREmitter, GuestRIP, RAData);
|
||||
IRDumper(Thread, IREmitter, GuestRIP);
|
||||
}
|
||||
|
||||
return {
|
||||
.IRView = IREmitter->ViewIR(),
|
||||
.RAData = RAData,
|
||||
.TotalInstructions = TotalInstructions,
|
||||
.TotalInstructionsLength = TotalInstructionsLength,
|
||||
.StartAddr = Thread->FrontendDecoder->DecodedMinAddress,
|
||||
@@ -760,19 +767,29 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
}
|
||||
|
||||
// Generate IR + Meta Info
|
||||
auto [IRView, RAData, TotalInstructions, TotalInstructionsLength, StartAddr, Length] =
|
||||
GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst);
|
||||
auto [IRView, TotalInstructions, TotalInstructionsLength, StartAddr, Length] = GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst);
|
||||
if (!IRView) {
|
||||
return {nullptr, nullptr, 0, 0};
|
||||
}
|
||||
|
||||
// Attempt to get the CPU backend to compile this code
|
||||
// Re-check if another thread raced us in compiling this block.
|
||||
// We could lock CodeBufferWriteMutex earlier to prevent this from happening,
|
||||
// but this would increase lock contention. Redundant frontend runs aren't
|
||||
// as expensive and are easily reverted.
|
||||
if (MaxInst != 1) {
|
||||
if (auto Block = Thread->LookupCache->FindBlock(GuestRIP)) {
|
||||
Thread->OpDispatcher->DelayedDisownBuffer();
|
||||
return {.CompiledCode = reinterpret_cast<uint8_t*>(Block), .DebugData = nullptr, .StartAddr = 0, .Length = 0};
|
||||
}
|
||||
}
|
||||
|
||||
auto DebugData = fextl::make_unique<FEXCore::Core::DebugData>();
|
||||
|
||||
// If the trap flag is set we generate single instruction blocks that each check to generate a single step exception.
|
||||
bool TFSet = Thread->CurrentFrame->State.flags[X86State::RFLAG_TF_RAW_LOC];
|
||||
|
||||
// Attempt to get the CPU backend to compile this code
|
||||
|
||||
auto CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, Length, TotalInstructions == 1, &*IRView, DebugData.get(), RAData, TFSet);
|
||||
auto CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, Length, TotalInstructions == 1, &*IRView, DebugData.get(), TFSet);
|
||||
|
||||
// Release the IR
|
||||
Thread->OpDispatcher->DelayedDisownBuffer();
|
||||
@@ -807,6 +824,9 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
auto [CodePtr, DebugData, StartAddr, Length] = CompileCode(Thread, GuestRIP, MaxInst);
|
||||
if (CodePtr == nullptr) {
|
||||
return 0;
|
||||
} else if (!DebugData) {
|
||||
// DebugData wasn't populated, indicating another thread raced us for compiling this block
|
||||
return reinterpret_cast<uintptr_t>(CodePtr);
|
||||
}
|
||||
|
||||
// The core managed to compile the code.
|
||||
@@ -888,7 +908,7 @@ uintptr_t ContextImpl::CompileSingleStep(FEXCore::Core::CpuStateFrame* Frame, ui
|
||||
}
|
||||
|
||||
static void InvalidateGuestThreadCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) {
|
||||
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
|
||||
auto lk = Thread->LookupCache->AcquireLock();
|
||||
|
||||
auto lower = Thread->LookupCache->CodePages.lower_bound(Start >> 12);
|
||||
auto upper = Thread->LookupCache->CodePages.upper_bound((Start + Length - 1) >> 12);
|
||||
@@ -915,8 +935,8 @@ void ContextImpl::MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) {
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
|
||||
if (Config.TSOAutoMigration) {
|
||||
// Only the lookup cache is cleared here, so that old code can keep running until next compilation
|
||||
std::lock_guard<std::recursive_mutex> lkLookupCache(Thread->LookupCache->WriteLock);
|
||||
// Only the lookup cache is cleared here, so that old code can keep running until next compilation.
|
||||
// This will leak previously compiled blocks until the CodeBuffer is cleared for some other reason.
|
||||
Thread->LookupCache->ClearCache();
|
||||
}
|
||||
}
|
||||
@@ -959,7 +979,7 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu
|
||||
auto Result = AddCustomIREntrypoint(
|
||||
Entrypoint,
|
||||
[this, GuestThunkEntrypoint](uintptr_t Entrypoint, FEXCore::IR::IREmitter* emit) {
|
||||
auto IRHeader = emit->_IRHeader(emit->Invalid(), Entrypoint, 0, 0);
|
||||
auto IRHeader = emit->_IRHeader(emit->Invalid(), Entrypoint, 0, 0, 0, 0);
|
||||
auto Block = emit->CreateCodeNode();
|
||||
IRHeader.first->Blocks = emit->WrapNode(Block);
|
||||
emit->SetCurrentCodeBlock(Block);
|
||||
@@ -967,12 +987,13 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
|
||||
if (GPRSize == IR::OpSize::i64Bit) {
|
||||
emit->_StoreRegister(emit->_Constant(Entrypoint), X86State::REG_R11, IR::GPRClass, GPRSize);
|
||||
IR::Ref R = emit->_StoreRegister(emit->_Constant(Entrypoint), GPRSize);
|
||||
R->Reg = IR::PhysicalRegister(IR::GPRFixedClass, X86State::REG_R11).Raw;
|
||||
} else {
|
||||
emit->_StoreContext(GPRSize, IR::FPRClass, emit->_VCastFromGPR(IR::OpSize::i64Bit, IR::OpSize::i64Bit, emit->_Constant(Entrypoint)),
|
||||
offsetof(Core::CPUState, mm[0][0]));
|
||||
}
|
||||
emit->_ExitFunction(emit->_Constant(GuestThunkEntrypoint));
|
||||
emit->_ExitFunction(IR::OpSize::i64Bit, emit->_Constant(GuestThunkEntrypoint));
|
||||
},
|
||||
ThunkHandler, (void*)GuestThunkEntrypoint);
|
||||
|
||||
|
||||
@@ -517,14 +517,15 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, R, Offset);
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
}
|
||||
// Result is now in x0
|
||||
|
||||
// Result is now in x0, x1
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(TMP1, ARMEmitter::XReg::x0);
|
||||
mov(TMP2, ARMEmitter::XReg::x1);
|
||||
}
|
||||
|
||||
FillStaticRegs();
|
||||
@@ -539,8 +540,6 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
LUDIVHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.AArch64.LUDIV));
|
||||
LDIVHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.AArch64.LDIV));
|
||||
LUREMHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.AArch64.LUREM));
|
||||
LREMHandlerAddress = EmitLongALUOpHandler(STATE_PTR(CpuStateFrame, Pointers.AArch64.LREM));
|
||||
|
||||
// Interpreter fallbacks
|
||||
{
|
||||
@@ -559,6 +558,8 @@ void Dispatcher::EmitDispatcher() {
|
||||
FABI_I64_I16_F80_F80_PTR,
|
||||
FABI_F80_I16_F80_PTR,
|
||||
FABI_F80_I16_F80_F80_PTR,
|
||||
FABI_F80x2_I16_F80_PTR,
|
||||
FABI_F64x2_I16_F64_PTR,
|
||||
FABI_I32_I64_I64_V128_V128_I16,
|
||||
FABI_I32_V128_V128_I16,
|
||||
}};
|
||||
@@ -627,6 +628,22 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
|
||||
constexpr static auto VABI1 = ARMEmitter::VReg::v0;
|
||||
constexpr static auto VABI2 = ARMEmitter::VReg::v1;
|
||||
|
||||
auto FillF80x2Result = [&]() {
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(VTMP1.Q(), VABI1.Q());
|
||||
mov(VTMP2.Q(), VABI2.Q());
|
||||
}
|
||||
FillForABICall(CTX->HostFeatures.SupportsPreserveAllABI, true);
|
||||
};
|
||||
|
||||
auto FillF64x2Result = [&]() {
|
||||
if (!TMP_ABIARGS) {
|
||||
fmov(VTMP1.D(), VABI1.D());
|
||||
fmov(VTMP2.D(), VABI2.D());
|
||||
}
|
||||
FillForABICall(CTX->HostFeatures.SupportsPreserveAllABI, true);
|
||||
};
|
||||
|
||||
auto FillF80Result = [&]() {
|
||||
if (VTMP1 != VABI1) {
|
||||
mov(VTMP1.Q(), VABI1.Q());
|
||||
@@ -947,6 +964,52 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
|
||||
|
||||
FillF80Result();
|
||||
} break;
|
||||
case FABI_F80x2_I16_F80_PTR: {
|
||||
// Linux Reg/Win32 Reg:
|
||||
// tmp4 (x4/x13): FallbackHandler
|
||||
// x30: return
|
||||
// vtmp1 (v0/v16): vector source 1
|
||||
// vtmp2 (v1/v16): vector source 2
|
||||
|
||||
SpillForABICall(CTX->HostFeatures.SupportsPreserveAllABI, TMP3, true);
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(VABI1.Q(), VTMP1.Q());
|
||||
}
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
// GenerateIndirectRuntimeCall<FEXCore::VectorRegPairType, uint16_t, FEXCore::VectorRegType, uint64_t>(FallbackPointerReg);
|
||||
} else {
|
||||
blr(FallbackPointerReg);
|
||||
}
|
||||
|
||||
FillF80x2Result();
|
||||
} break;
|
||||
case FABI_F64x2_I16_F64_PTR: {
|
||||
// Linux Reg/Win32 Reg:
|
||||
// tmp4 (x4/x13): FallbackHandler
|
||||
// x30: return
|
||||
// vtmp1 (v0/v16): vector source 1
|
||||
// vtmp2 (v1/v16): vector source 2
|
||||
|
||||
SpillForABICall(CTX->HostFeatures.SupportsPreserveAllABI, TMP3, true);
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
if (!TMP_ABIARGS) {
|
||||
fmov(VABI1.D(), VTMP1.D());
|
||||
}
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
// GenerateIndirectRuntimeCall<FEXCore::VectorScalarF64Pair, uint16_t, FEXCore::VectorRegType, uint64_t>(FallbackPointerReg);
|
||||
} else {
|
||||
blr(FallbackPointerReg);
|
||||
}
|
||||
|
||||
FillF64x2Result();
|
||||
} break;
|
||||
case FABI_I32_I64_I64_V128_V128_I16: {
|
||||
// Linux Reg/Win32 Reg:
|
||||
// stack: FallbackHandler
|
||||
@@ -1036,8 +1099,6 @@ void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState* Thread)
|
||||
auto& AArch64 = Thread->CurrentFrame->Pointers.AArch64;
|
||||
AArch64.LUDIVHandler = LUDIVHandlerAddress;
|
||||
AArch64.LDIVHandler = LDIVHandlerAddress;
|
||||
AArch64.LUREMHandler = LUREMHandlerAddress;
|
||||
AArch64.LREMHandler = LREMHandlerAddress;
|
||||
|
||||
// Fill in the fallback handlers
|
||||
InterpreterOps::FillFallbackIndexPointers(Common.FallbackHandlerPointers, &ABIPointers[0]);
|
||||
|
||||
@@ -118,8 +118,6 @@ private:
|
||||
// Long division helpers
|
||||
uint64_t LUDIVHandlerAddress {};
|
||||
uint64_t LDIVHandlerAddress {};
|
||||
uint64_t LUREMHandlerAddress {};
|
||||
uint64_t LREMHandlerAddress {};
|
||||
|
||||
void EmitDispatcher();
|
||||
uint64_t GenerateABICall(FallbackABI ABI);
|
||||
|
||||
@@ -87,7 +87,7 @@ private:
|
||||
|
||||
static constexpr size_t DefaultDecodedBufferSize = 0x10000;
|
||||
FEXCore::X86Tables::DecodedInst* DecodedBuffer {};
|
||||
Utils::FixedSizePooledAllocation<FEXCore::X86Tables::DecodedInst*, 5000, 500> PoolObject;
|
||||
Utils::PoolBufferWithTimedRetirement<FEXCore::X86Tables::DecodedInst*, 5000, 500> PoolObject;
|
||||
size_t DecodedSize {};
|
||||
|
||||
const uint8_t* InstStream {};
|
||||
|
||||
@@ -202,6 +202,15 @@ struct OpHandlers<IR::OP_F80COS> {
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SINCOS> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegPairType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
softfloat_state State = SoftFloatStateFromFCW(FCW, true);
|
||||
return FEXCore::MakeVectorRegPair(X80SoftFloat::FSIN(&State, Src1), X80SoftFloat::FCOS(&State, Src1));
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80XTRACT_EXP> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
@@ -315,6 +324,21 @@ struct OpHandlers<IR::OP_F64COS> {
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SINCOS> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorScalarF64Pair handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
double sin, cos;
|
||||
#ifdef _WIN32
|
||||
sin = ::sin(src);
|
||||
cos = ::cos(src);
|
||||
#else
|
||||
sincos(src, &sin, &cos);
|
||||
#endif
|
||||
return VectorScalarF64Pair {sin, cos};
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64TAN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
|
||||
@@ -50,6 +50,8 @@ void InterpreterOps::FillFallbackIndexPointers(Core::FallbackABIInfo* Info, uint
|
||||
Info[Core::OPINDEX_F80SQRT] = {ABIHandlers[FABI_F80_I16_F80_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80SQRT>::handle)};
|
||||
Info[Core::OPINDEX_F80SIN] = {ABIHandlers[FABI_F80_I16_F80_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80SIN>::handle)};
|
||||
Info[Core::OPINDEX_F80COS] = {ABIHandlers[FABI_F80_I16_F80_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80COS>::handle)};
|
||||
Info[Core::OPINDEX_F80SINCOS] = {ABIHandlers[FABI_F80x2_I16_F80_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80SINCOS>::handle)};
|
||||
Info[Core::OPINDEX_F80XTRACT_EXP] = {ABIHandlers[FABI_F80_I16_F80_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80XTRACT_EXP>::handle)};
|
||||
Info[Core::OPINDEX_F80XTRACT_SIG] = {ABIHandlers[FABI_F80_I16_F80_PTR],
|
||||
@@ -82,6 +84,8 @@ void InterpreterOps::FillFallbackIndexPointers(Core::FallbackABIInfo* Info, uint
|
||||
// Double Precision Unary
|
||||
Info[Core::OPINDEX_F64SIN] = {ABIHandlers[FABI_F64_I16_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64SIN>::handle)};
|
||||
Info[Core::OPINDEX_F64COS] = {ABIHandlers[FABI_F64_I16_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64COS>::handle)};
|
||||
Info[Core::OPINDEX_F64SINCOS] = {ABIHandlers[FABI_F64x2_I16_F64_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64SINCOS>::handle)};
|
||||
Info[Core::OPINDEX_F64TAN] = {ABIHandlers[FABI_F64_I16_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64TAN>::handle)};
|
||||
Info[Core::OPINDEX_F64F2XM1] = {ABIHandlers[FABI_F64_I16_F64_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64F2XM1>::handle)};
|
||||
@@ -198,6 +202,12 @@ bool InterpreterOps::GetFallbackHandler(const IR::IROp_Header* IROp, FallbackInf
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_UNARYPAIR_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = {FABI_F80x2_I16_F80_PTR, Core::OPINDEX_F80##OP}; \
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_BINARY_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = {FABI_F80_I16_F80_F80_PTR, Core::OPINDEX_F80##OP}; \
|
||||
@@ -215,6 +225,12 @@ bool InterpreterOps::GetFallbackHandler(const IR::IROp_Header* IROp, FallbackInf
|
||||
*Info = {FABI_F64_I16_F64_PTR, Core::OPINDEX_F64##OP}; \
|
||||
return true; \
|
||||
}
|
||||
#define COMMON_UNARYPAIR_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = {FABI_F64x2_I16_F64_PTR, Core::OPINDEX_F64##OP}; \
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_BINARY_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = {FABI_F64_I16_F64_F64_PTR, Core::OPINDEX_F64##OP}; \
|
||||
@@ -228,6 +244,7 @@ bool InterpreterOps::GetFallbackHandler(const IR::IROp_Header* IROp, FallbackInf
|
||||
COMMON_UNARY_X87_OP(SQRT)
|
||||
COMMON_UNARY_X87_OP(SIN)
|
||||
COMMON_UNARY_X87_OP(COS)
|
||||
COMMON_UNARYPAIR_X87_OP(SINCOS)
|
||||
COMMON_UNARY_X87_OP(XTRACT_EXP)
|
||||
COMMON_UNARY_X87_OP(XTRACT_SIG)
|
||||
COMMON_UNARY_X87_OP(BCDSTORE)
|
||||
@@ -249,6 +266,7 @@ bool InterpreterOps::GetFallbackHandler(const IR::IROp_Header* IROp, FallbackInf
|
||||
COMMON_UNARY_F64_OP(TAN)
|
||||
COMMON_UNARY_F64_OP(SIN)
|
||||
COMMON_UNARY_F64_OP(COS)
|
||||
COMMON_UNARYPAIR_F64_OP(SINCOS)
|
||||
|
||||
// Double Precision Binary
|
||||
COMMON_BINARY_F64_OP(FYL2X)
|
||||
|
||||
@@ -27,6 +27,8 @@ enum FallbackABI {
|
||||
FABI_I64_I16_F80_F80_PTR,
|
||||
FABI_F80_I16_F80_PTR,
|
||||
FABI_F80_I16_F80_F80_PTR,
|
||||
FABI_F80x2_I16_F80_PTR,
|
||||
FABI_F64x2_I16_F64_PTR,
|
||||
FABI_I32_I64_I64_V128_V128_I16,
|
||||
FABI_I32_V128_V128_I16,
|
||||
FABI_UNKNOWN,
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -10,18 +10,17 @@ $end_info$
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
DEF_OP(CASPair) {
|
||||
auto Op = IROp->C<IR::IROp_CASPair>();
|
||||
LOGMAN_THROW_A_FMT(IROp->ElementSize == IR::OpSize::i32Bit || IROp->ElementSize == IR::OpSize::i64Bit, "Wrong element size");
|
||||
// Size is the size of each pair element
|
||||
auto Dst0 = GetReg(Op->OutLo.ID());
|
||||
auto Dst1 = GetReg(Op->OutHi.ID());
|
||||
auto Expected0 = GetReg(Op->ExpectedLo.ID());
|
||||
auto Expected1 = GetReg(Op->ExpectedHi.ID());
|
||||
auto Desired0 = GetReg(Op->DesiredLo.ID());
|
||||
auto Desired1 = GetReg(Op->DesiredHi.ID());
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Dst0 = GetReg(Op->OutLo);
|
||||
auto Dst1 = GetReg(Op->OutHi);
|
||||
auto Expected0 = GetReg(Op->ExpectedLo);
|
||||
auto Expected1 = GetReg(Op->ExpectedHi);
|
||||
auto Desired0 = GetReg(Op->DesiredLo);
|
||||
auto Desired1 = GetReg(Op->DesiredHi);
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
|
||||
const auto EmitSize = IROp->ElementSize == IR::OpSize::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
@@ -98,9 +97,9 @@ DEF_OP(CAS) {
|
||||
// if (DataSrc == Src3) { *Src1 == Src2; } Src2 = DataSrc
|
||||
// This will write to memory! Careful!
|
||||
|
||||
auto Expected = GetReg(Op->Expected.ID());
|
||||
auto Desired = GetReg(Op->Desired.ID());
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Expected = GetReg(Op->Expected);
|
||||
auto Desired = GetReg(Op->Desired);
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Dst = GetReg(Node);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
@@ -144,8 +143,8 @@ DEF_OP(AtomicAdd) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
staddl(SubEmitSize, Src, MemSrc);
|
||||
@@ -164,8 +163,8 @@ DEF_OP(AtomicSub) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
neg(EmitSize, TMP2, Src);
|
||||
@@ -185,8 +184,8 @@ DEF_OP(AtomicAnd) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mvn(EmitSize, TMP2, Src);
|
||||
@@ -206,8 +205,8 @@ DEF_OP(AtomicCLR) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
stclrl(SubEmitSize, Src, MemSrc);
|
||||
@@ -226,8 +225,8 @@ DEF_OP(AtomicOr) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
stsetl(SubEmitSize, Src, MemSrc);
|
||||
@@ -246,8 +245,8 @@ DEF_OP(AtomicXor) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
steorl(SubEmitSize, Src, MemSrc);
|
||||
@@ -266,7 +265,7 @@ DEF_OP(AtomicNeg) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
@@ -284,8 +283,8 @@ DEF_OP(AtomicSwap) {
|
||||
"d CAS "
|
||||
"size");
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = OpSize == IR::OpSize::i64Bit ? ARMEmitter::SubRegSize::i64Bit :
|
||||
@@ -310,8 +309,8 @@ DEF_OP(AtomicFetchAdd) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
ldaddal(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
@@ -331,8 +330,8 @@ DEF_OP(AtomicFetchSub) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
neg(EmitSize, TMP2, Src);
|
||||
@@ -353,8 +352,8 @@ DEF_OP(AtomicFetchAnd) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mvn(EmitSize, TMP2, Src);
|
||||
@@ -375,8 +374,8 @@ DEF_OP(AtomicFetchCLR) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
ldclral(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
@@ -396,8 +395,8 @@ DEF_OP(AtomicFetchOr) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
ldsetal(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
@@ -417,8 +416,8 @@ DEF_OP(AtomicFetchXor) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
ldeoral(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
@@ -438,7 +437,7 @@ DEF_OP(AtomicFetchNeg) {
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr.ID());
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
@@ -452,7 +451,7 @@ DEF_OP(AtomicFetchNeg) {
|
||||
DEF_OP(TelemetrySetValue) {
|
||||
#ifndef FEX_DISABLE_TELEMETRY
|
||||
auto Op = IROp->C<IR::IROp_TelemetrySetValue>();
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.TelemetryValueAddresses[Op->TelemetryValueIndex]));
|
||||
|
||||
@@ -473,5 +472,4 @@ DEF_OP(TelemetrySetValue) {
|
||||
#endif
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -18,7 +18,6 @@ $end_info$
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(CallbackReturn) {
|
||||
// spill back to CTX
|
||||
@@ -82,7 +81,7 @@ DEF_OP(ExitFunction) {
|
||||
} else {
|
||||
|
||||
ARMEmitter::ForwardLabel FullLookup;
|
||||
auto RipReg = GetReg(Op->NewRIP.ID());
|
||||
auto RipReg = GetReg(Op->NewRIP);
|
||||
|
||||
// L1 Cache
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
@@ -109,9 +108,9 @@ DEF_OP(ExitFunction) {
|
||||
|
||||
DEF_OP(Jump) {
|
||||
const auto Op = IROp->C<IR::IROp_Jump>();
|
||||
const auto Target = Op->TargetBlock.ID();
|
||||
const auto Target = Op->TargetBlock;
|
||||
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(Target).first->second;
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(Target.ID()).first->second;
|
||||
}
|
||||
|
||||
DEF_OP(CondJump) {
|
||||
@@ -125,10 +124,10 @@ DEF_OP(CondJump) {
|
||||
[[maybe_unused]] uint64_t Const;
|
||||
[[maybe_unused]] const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
|
||||
|
||||
auto Reg = GetReg(Op->Cmp1.ID());
|
||||
auto Reg = GetReg(Op->Cmp1);
|
||||
const auto Size = Op->CompareSize == IR::OpSize::i32Bit ? ARMEmitter::Size::i32Bit : ARMEmitter::Size::i64Bit;
|
||||
|
||||
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1.ID()), "CondJump: Expected GPR");
|
||||
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1), "CondJump: Expected GPR");
|
||||
LOGMAN_THROW_A_FMT(isConst, "CondJump: Expected constant source");
|
||||
|
||||
if (Op->Cond.Val == FEXCore::IR::COND_EQ) {
|
||||
@@ -184,7 +183,7 @@ DEF_OP(Syscall) {
|
||||
if (Op->Header.Args[i].IsInvalid()) {
|
||||
continue;
|
||||
}
|
||||
str(GetReg(Op->Header.Args[i].ID()).X(), ARMEmitter::Reg::rsp, i * 8);
|
||||
str(GetReg(Op->Header.Args[i]).X(), ARMEmitter::Reg::rsp, i * 8);
|
||||
}
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SyscallHandlerObj));
|
||||
@@ -244,7 +243,7 @@ DEF_OP(InlineSyscall) {
|
||||
break;
|
||||
}
|
||||
|
||||
auto Reg = GetReg(Op->Header.Args[i].ID());
|
||||
auto Reg = GetReg(Op->Header.Args[i]);
|
||||
if (Reg == ARMEmitter::Reg::r8 || Reg == ARMEmitter::Reg::r4 || Reg == ARMEmitter::Reg::r5) {
|
||||
|
||||
SpillMask |= (1U << Reg.Idx());
|
||||
@@ -274,7 +273,7 @@ DEF_OP(InlineSyscall) {
|
||||
break;
|
||||
}
|
||||
|
||||
auto Reg = GetReg(Op->Header.Args[i].ID());
|
||||
auto Reg = GetReg(Op->Header.Args[i]);
|
||||
if (SpillMask & (1U << Reg.Idx())) {
|
||||
// In the case of intersection with x4, x5, or x8 then these are currently SRA
|
||||
// for registers RAX, RDX, and RSP. Which have just been spilled
|
||||
@@ -292,7 +291,7 @@ DEF_OP(InlineSyscall) {
|
||||
break;
|
||||
}
|
||||
|
||||
mov(EmitSize, RegArgs[i].R(), GetReg(Op->Header.Args[i].ID()));
|
||||
mov(EmitSize, RegArgs[i].R(), GetReg(Op->Header.Args[i]));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -325,7 +324,7 @@ DEF_OP(Thunk) {
|
||||
|
||||
PushDynamicRegs(TMP1);
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetReg(Op->ArgPtr.ID()));
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetReg(Op->ArgPtr));
|
||||
|
||||
auto thunkFn = static_cast<Context::ContextImpl*>(ThreadState->CTX)->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, (uintptr_t)thunkFn);
|
||||
@@ -412,8 +411,9 @@ DEF_OP(ThreadRemoveCodeEntry) {
|
||||
DEF_OP(CPUID) {
|
||||
auto Op = IROp->C<IR::IROp_CPUID>();
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, TMP2, GetReg(Op->Function.ID()));
|
||||
mov(ARMEmitter::Size::i64Bit, TMP3, GetReg(Op->Leaf.ID()));
|
||||
isb();
|
||||
mov(ARMEmitter::Size::i64Bit, TMP2, GetReg(Op->Function));
|
||||
mov(ARMEmitter::Size::i64Bit, TMP3, GetReg(Op->Leaf));
|
||||
|
||||
PushDynamicRegs(TMP4);
|
||||
SpillStaticRegs(TMP4);
|
||||
@@ -446,10 +446,10 @@ DEF_OP(CPUID) {
|
||||
|
||||
// Results are in x0, x1
|
||||
// Results want to be 4xi32 scalars
|
||||
mov(ARMEmitter::Size::i32Bit, GetReg(Op->OutEAX.ID()), TMP1);
|
||||
mov(ARMEmitter::Size::i32Bit, GetReg(Op->OutECX.ID()), TMP2);
|
||||
ubfx(ARMEmitter::Size::i64Bit, GetReg(Op->OutEBX.ID()), TMP1, 32, 32);
|
||||
ubfx(ARMEmitter::Size::i64Bit, GetReg(Op->OutEDX.ID()), TMP2, 32, 32);
|
||||
mov(ARMEmitter::Size::i32Bit, GetReg(Op->OutEAX), TMP1);
|
||||
mov(ARMEmitter::Size::i32Bit, GetReg(Op->OutECX), TMP2);
|
||||
ubfx(ARMEmitter::Size::i64Bit, GetReg(Op->OutEBX), TMP1, 32, 32);
|
||||
ubfx(ARMEmitter::Size::i64Bit, GetReg(Op->OutEDX), TMP2, 32, 32);
|
||||
}
|
||||
|
||||
DEF_OP(XGetBV) {
|
||||
@@ -458,7 +458,7 @@ DEF_OP(XGetBV) {
|
||||
PushDynamicRegs(TMP4);
|
||||
SpillStaticRegs(TMP4);
|
||||
|
||||
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, GetReg(Op->Function.ID()));
|
||||
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, GetReg(Op->Function));
|
||||
|
||||
// x0 = CPUID Handler
|
||||
// x1 = XCR Function
|
||||
@@ -479,9 +479,8 @@ DEF_OP(XGetBV) {
|
||||
PopDynamicRegs();
|
||||
|
||||
// Results are in x0, need to split into i32 parts
|
||||
mov(ARMEmitter::Size::i32Bit, GetReg(Op->OutEAX.ID()), TMP1);
|
||||
ubfx(ARMEmitter::Size::i64Bit, GetReg(Op->OutEDX.ID()), TMP1, 32, 32);
|
||||
mov(ARMEmitter::Size::i32Bit, GetReg(Op->OutEAX), TMP1);
|
||||
ubfx(ARMEmitter::Size::i64Bit, GetReg(Op->OutEDX), TMP1, 32, 32);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -9,7 +9,6 @@ $end_info$
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
DEF_OP(VInsGPR) {
|
||||
const auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -23,8 +22,8 @@ DEF_OP(VInsGPR) {
|
||||
const auto ElementsPer128Bit = IR::NumElements(IR::OpSize::i128Bit, ElementSize);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto DestVector = GetVReg(Op->DestVector.ID());
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
const auto DestVector = GetVReg(Op->DestVector);
|
||||
const auto Src = GetReg(Op->Src);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto ElementSizeBits = IR::OpSizeAsBits(ElementSize);
|
||||
@@ -88,7 +87,7 @@ DEF_OP(VInsGPR) {
|
||||
DEF_OP(VCastFromGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VCastFromGPR>();
|
||||
auto Dst = GetVReg(Node);
|
||||
auto Src = GetReg(Op->Src.ID());
|
||||
auto Src = GetReg(Op->Src);
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
case IR::OpSize::i8Bit:
|
||||
@@ -109,8 +108,8 @@ DEF_OP(VLoadTwoGPRs) {
|
||||
const auto Op = IROp->C<IR::IROp_VLoadTwoGPRs>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto SrcLower = GetReg(Op->Lower.ID());
|
||||
const auto SrcUpper = GetReg(Op->Upper.ID());
|
||||
const auto SrcLower = GetReg(Op->Lower);
|
||||
const auto SrcUpper = GetReg(Op->Upper);
|
||||
fmov(ARMEmitter::Size::i64Bit, Dst.D(), SrcLower);
|
||||
fmov(ARMEmitter::Size::i64Bit, Dst.D(), SrcUpper, true);
|
||||
}
|
||||
@@ -120,7 +119,7 @@ DEF_OP(VDupFromGPR) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
const auto Src = GetReg(Op->Src);
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
@@ -141,7 +140,7 @@ DEF_OP(Float_FromGPR_S) {
|
||||
const uint16_t Conv = (ElementSize << 8) | IR::OpSizeToSize(Op->SrcElementSize);
|
||||
|
||||
auto Dst = GetVReg(Node);
|
||||
auto Src = GetReg(Op->Src.ID());
|
||||
auto Src = GetReg(Op->Src);
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0204: { // Half <- int32_t
|
||||
@@ -179,7 +178,7 @@ DEF_OP(Float_FToF) {
|
||||
const uint16_t Conv = (IR::OpSizeToSize(Op->Header.ElementSize) << 8) | IR::OpSizeToSize(Op->SrcElementSize);
|
||||
|
||||
auto Dst = GetVReg(Node);
|
||||
auto Src = GetVReg(Op->Scalar.ID());
|
||||
auto Src = GetVReg(Op->Scalar);
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0204: { // Half <- Float
|
||||
@@ -220,7 +219,7 @@ DEF_OP(Vector_SToF) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B;
|
||||
scvtf(Dst.Z(), SubEmitSize, Mask.Merging(), Vector.Z(), SubEmitSize);
|
||||
@@ -253,7 +252,7 @@ DEF_OP(Vector_FToZS) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B;
|
||||
fcvtzs(Dst.Z(), SubEmitSize, Mask.Merging(), Vector.Z(), SubEmitSize);
|
||||
@@ -286,7 +285,7 @@ DEF_OP(Vector_FToS) {
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B;
|
||||
@@ -294,7 +293,7 @@ DEF_OP(Vector_FToS) {
|
||||
fcvtzs(Dst.Z(), SubEmitSize, Mask.Merging(), Dst.Z(), SubEmitSize);
|
||||
} else {
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
if (OpSize == IR::OpSize::i64Bit) {
|
||||
frinti(SubEmitSize, Dst.D(), Vector.D());
|
||||
fcvtzs(SubEmitSize, Dst.D(), Dst.D());
|
||||
@@ -317,7 +316,7 @@ DEF_OP(Vector_FToF) {
|
||||
const auto Conv = (IR::OpSizeToSize(ElementSize) << 8) | IR::OpSizeToSize(Op->SrcElementSize);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// Curiously, FCVTLT and FCVTNT have no bottom variants,
|
||||
@@ -381,7 +380,7 @@ DEF_OP(VFCVTL2) {
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
fcvtl2(SubEmitSize, Dst.D(), Vector.D());
|
||||
}
|
||||
@@ -392,8 +391,8 @@ DEF_OP(VFCVTN2) {
|
||||
const auto SubEmitSize = ConvertSubRegSize248(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorLower = GetVReg(Op->VectorLower.ID());
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper.ID());
|
||||
const auto VectorLower = GetVReg(Op->VectorLower);
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper);
|
||||
|
||||
auto Lower = VectorLower;
|
||||
if (Dst != VectorLower) {
|
||||
@@ -418,7 +417,7 @@ DEF_OP(Vector_FToI) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -479,7 +478,7 @@ DEF_OP(Vector_FToISized) {
|
||||
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFRINTTS, "Need FRINTTS for Vector_FToISized");
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (ElementSize == IROp->Size) {
|
||||
// See above
|
||||
@@ -533,7 +532,7 @@ DEF_OP(Vector_F64ToI32) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
if (HostSupportsSVE128 || HostSupportsSVE256) {
|
||||
const auto Mask = Is256Bit ? PRED_TMP_32B.Merging() : PRED_TMP_16B.Merging();
|
||||
// First step is to round the f64 values to integrals (frint*)
|
||||
@@ -583,5 +582,4 @@ DEF_OP(Vector_F64ToI32) {
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -8,11 +8,10 @@ $end_info$
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(VAESImc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESImc>();
|
||||
aesimc(GetVReg(Node), GetVReg(Op->Vector.ID()));
|
||||
aesimc(GetVReg(Node), GetVReg(Op->Vector));
|
||||
}
|
||||
|
||||
DEF_OP(VAESEnc) {
|
||||
@@ -20,9 +19,9 @@ DEF_OP(VAESEnc) {
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Key = GetVReg(Op->Key.ID());
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
const auto Key = GetVReg(Op->Key);
|
||||
const auto State = GetVReg(Op->State);
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg);
|
||||
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
@@ -45,9 +44,9 @@ DEF_OP(VAESEncLast) {
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Key = GetVReg(Op->Key.ID());
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
const auto Key = GetVReg(Op->Key);
|
||||
const auto State = GetVReg(Op->State);
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg);
|
||||
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
@@ -68,9 +67,9 @@ DEF_OP(VAESDec) {
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Key = GetVReg(Op->Key.ID());
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
const auto Key = GetVReg(Op->Key);
|
||||
const auto State = GetVReg(Op->State);
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg);
|
||||
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
@@ -93,9 +92,9 @@ DEF_OP(VAESDecLast) {
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Key = GetVReg(Op->Key.ID());
|
||||
const auto State = GetVReg(Op->State.ID());
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
const auto Key = GetVReg(Op->Key);
|
||||
const auto State = GetVReg(Op->State);
|
||||
const auto ZeroReg = GetVReg(Op->ZeroReg);
|
||||
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
@@ -114,9 +113,9 @@ DEF_OP(VAESDecLast) {
|
||||
DEF_OP(VAESKeyGenAssist) {
|
||||
auto Op = IROp->C<IR::IROp_VAESKeyGenAssist>();
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src = GetVReg(Op->Src.ID());
|
||||
const auto Swizzle = GetVReg(Op->KeyGenTBLSwizzle.ID());
|
||||
auto ZeroReg = GetVReg(Op->ZeroReg.ID());
|
||||
const auto Src = GetVReg(Op->Src);
|
||||
const auto Swizzle = GetVReg(Op->KeyGenTBLSwizzle);
|
||||
auto ZeroReg = GetVReg(Op->ZeroReg);
|
||||
|
||||
if (Dst == ZeroReg) {
|
||||
// Seriously? ZeroReg ended up being the destination register?
|
||||
@@ -148,8 +147,8 @@ DEF_OP(CRC32) {
|
||||
auto Op = IROp->C<IR::IROp_CRC32>();
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
const auto Src1 = GetReg(Op->Src1);
|
||||
const auto Src2 = GetReg(Op->Src2);
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case IR::OpSize::i8Bit: crc32cb(Dst.W(), Src1.W(), Src2.W()); break;
|
||||
@@ -164,7 +163,7 @@ DEF_OP(VSha1H) {
|
||||
auto Op = IROp->C<IR::IROp_VSha1H>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src = GetVReg(Op->Src.ID());
|
||||
const auto Src = GetVReg(Op->Src);
|
||||
|
||||
sha1h(Dst.S(), Src.S());
|
||||
}
|
||||
@@ -173,9 +172,9 @@ DEF_OP(VSha1C) {
|
||||
auto Op = IROp->C<IR::IROp_VSha1C>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src3 = GetVReg(Op->Src3.ID());
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
const auto Src3 = GetVReg(Op->Src3);
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha1c(Dst, Src2.S(), Src3);
|
||||
@@ -193,9 +192,9 @@ DEF_OP(VSha1M) {
|
||||
auto Op = IROp->C<IR::IROp_VSha1M>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src3 = GetVReg(Op->Src3.ID());
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
const auto Src3 = GetVReg(Op->Src3);
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha1m(Dst, Src2.S(), Src3);
|
||||
@@ -213,9 +212,9 @@ DEF_OP(VSha1P) {
|
||||
auto Op = IROp->C<IR::IROp_VSha1P>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src3 = GetVReg(Op->Src3.ID());
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
const auto Src3 = GetVReg(Op->Src3);
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha1p(Dst, Src2.S(), Src3);
|
||||
@@ -233,8 +232,8 @@ DEF_OP(VSha1SU1) {
|
||||
auto Op = IROp->C<IR::IROp_VSha1SU1>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha1su1(Dst, Src2);
|
||||
@@ -252,9 +251,9 @@ DEF_OP(VSha256H) {
|
||||
auto Op = IROp->C<IR::IROp_VSha256H>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src3 = GetVReg(Op->Src3.ID());
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
const auto Src3 = GetVReg(Op->Src3);
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha256h(Dst, Src2, Src3);
|
||||
@@ -272,9 +271,9 @@ DEF_OP(VSha256H2) {
|
||||
auto Op = IROp->C<IR::IROp_VSha256H2>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src3 = GetVReg(Op->Src3.ID());
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
const auto Src3 = GetVReg(Op->Src3);
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha256h2(Dst, Src2, Src3);
|
||||
@@ -292,8 +291,8 @@ DEF_OP(VSha256U0) {
|
||||
auto Op = IROp->C<IR::IROp_VSha256U0>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
|
||||
if (Dst == Src1) {
|
||||
sha256su0(Dst, Src2);
|
||||
@@ -308,8 +307,8 @@ DEF_OP(VSha256U1) {
|
||||
auto Op = IROp->C<IR::IROp_VSha256U1>();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
|
||||
if (Dst != Src1 && Dst != Src2) {
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), 0);
|
||||
@@ -326,8 +325,8 @@ DEF_OP(PCLMUL) {
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1.ID());
|
||||
const auto Src2 = GetVReg(Op->Src2.ID());
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i128Bit, "Currently only supports 128-bit operations.");
|
||||
|
||||
@@ -346,5 +345,4 @@ DEF_OP(PCLMUL) {
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -40,36 +40,33 @@ $end_info$
|
||||
#include <string.h>
|
||||
#include <limits>
|
||||
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
// We don't want to move above 128MB atm because that means we will have to encode longer jumps
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
|
||||
|
||||
namespace {
|
||||
static uint64_t LUDIV(uint64_t SrcHigh, uint64_t SrcLow, uint64_t Divisor) {
|
||||
struct DivRem {
|
||||
uint64_t Quotient;
|
||||
uint64_t Remainder;
|
||||
};
|
||||
|
||||
static struct DivRem LUDIV(uint64_t SrcHigh, uint64_t SrcLow, uint64_t Divisor) {
|
||||
__uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
|
||||
__uint128_t Res = Source / Divisor;
|
||||
return Res;
|
||||
|
||||
return {
|
||||
.Quotient = (uint64_t)(Source / Divisor),
|
||||
.Remainder = (uint64_t)(Source % Divisor),
|
||||
};
|
||||
}
|
||||
|
||||
static int64_t LDIV(uint64_t SrcHigh, uint64_t SrcLow, int64_t Divisor) {
|
||||
static struct DivRem
|
||||
LDIV(uint64_t SrcHigh, uint64_t SrcLow, int64_t Divisor) {
|
||||
__int128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
|
||||
__int128_t Res = Source / Divisor;
|
||||
return Res;
|
||||
|
||||
return {
|
||||
.Quotient = (uint64_t)(Source / Divisor),
|
||||
.Remainder = (uint64_t)(Source % Divisor),
|
||||
};
|
||||
}
|
||||
|
||||
static uint64_t LUREM(uint64_t SrcHigh, uint64_t SrcLow, uint64_t Divisor) {
|
||||
__uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
|
||||
__uint128_t Res = Source % Divisor;
|
||||
return Res;
|
||||
}
|
||||
|
||||
static int64_t LREM(uint64_t SrcHigh, uint64_t SrcLow, int64_t Divisor) {
|
||||
__int128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
|
||||
__int128_t Res = Source % Divisor;
|
||||
return Res;
|
||||
}
|
||||
|
||||
static void PrintValue(uint64_t Value) {
|
||||
static void
|
||||
PrintValue(uint64_t Value) {
|
||||
LogMan::Msg::DFmt("Value: 0x{:x}", Value);
|
||||
}
|
||||
|
||||
@@ -80,13 +77,23 @@ static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) {
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
FallbackInfo Info;
|
||||
if (!InterpreterOps::GetFallbackHandler(IROp, &Info)) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
LOGMAN_MSG_A_FMT("Unhandled IR Op: {}", FEXCore::IR::GetName(IROp->Op));
|
||||
#endif
|
||||
} else {
|
||||
auto FillF80x2Result = [&](auto DstLo, auto DstHi) {
|
||||
mov(DstLo.Q(), VTMP1.Q());
|
||||
mov(DstHi.Q(), VTMP2.Q());
|
||||
};
|
||||
|
||||
auto FillF64x2Result = [&](auto DstLo, auto DstHi) {
|
||||
fmov(DstLo.D(), VTMP1.D());
|
||||
fmov(DstHi.D(), VTMP2.D());
|
||||
};
|
||||
|
||||
auto FillF80Result = [&]() {
|
||||
const auto Dst = GetVReg(Node);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
@@ -125,7 +132,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
// vtmp1 (v0/v16): source
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
fmov(VTMP1.S(), Src1.S());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
@@ -143,7 +150,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
// vtmp1 (v0/v16): source
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
fmov(VTMP1.D(), Src1.D());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
@@ -162,7 +169,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
// tmp2 (x1/x11): source
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetReg(IROp->Args[0].ID());
|
||||
const auto Src1 = GetReg(IROp->Args[0]);
|
||||
|
||||
// Need to sign or zero extend this for the dispatcher handler.
|
||||
if (Info.ABI == FABI_F80_I16_I16_PTR) {
|
||||
@@ -186,7 +193,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
// vtmp1 (v0/v16): source
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
@@ -204,7 +211,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
// vtmp1 (v0/v16): source
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
@@ -222,7 +229,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
// vtmp1 (v0/v16): vector source
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
fmov(VTMP1.D(), Src1.D());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
@@ -232,6 +239,30 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
FillF64Result();
|
||||
} break;
|
||||
case FABI_F64x2_I16_F64_PTR: {
|
||||
// Linux Reg/Win32 Reg:
|
||||
// tmp4 (x4/x13): FallbackHandler
|
||||
// x30: return
|
||||
// vtmp1 (v0/v16): vector source
|
||||
// vtmp2 (v1/v16): vector source
|
||||
#ifdef VIXL_SIMULATOR
|
||||
LOGMAN_THROW_A_FMT(CTX->Config.DisableVixlIndirectCalls, "Vector register pairs unsupported by simulator currently");
|
||||
#endif
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
const auto DstLo = GetVReg(IROp->Args[1]);
|
||||
const auto DstHi = GetVReg(IROp->Args[2]);
|
||||
|
||||
fmov(VTMP1.D(), Src1.D());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
FillF64x2Result(DstLo, DstHi);
|
||||
} break;
|
||||
|
||||
case FABI_F64_I16_F64_F64_PTR: {
|
||||
// Linux Reg/Win32 Reg:
|
||||
@@ -241,8 +272,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
// vtmp2 (v1/v17): vector source 2
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src2 = GetVReg(IROp->Args[1].ID());
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
const auto Src2 = GetVReg(IROp->Args[1]);
|
||||
|
||||
fmov(VTMP1.D(), Src1.D());
|
||||
fmov(VTMP2.D(), Src2.D());
|
||||
@@ -262,7 +293,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
// vtmp1 (v0/v16): source
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
@@ -280,7 +311,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
// vtmp1 (v0/v16): source
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
@@ -298,7 +329,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
// vtmp1 (v0/v16): source
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
@@ -317,8 +348,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
// vtmp2 (v1/v17): vector source 2
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src2 = GetVReg(IROp->Args[1].ID());
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
const auto Src2 = GetVReg(IROp->Args[1]);
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
mov(VTMP2.Q(), Src2.Q());
|
||||
|
||||
@@ -337,7 +368,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
// vtmp1 (v0/v16): vector source 1
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
@@ -348,6 +379,31 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
FillF80Result();
|
||||
} break;
|
||||
|
||||
case FABI_F80x2_I16_F80_PTR: {
|
||||
// Linux Reg/Win32 Reg:
|
||||
// tmp4 (x4/x13): FallbackHandler
|
||||
// x30: return
|
||||
// vtmp1 (v0/v16): vector source 1
|
||||
// vtmp2 (v1/v16): vector source 2
|
||||
#ifdef VIXL_SIMULATOR
|
||||
LOGMAN_THROW_A_FMT(CTX->Config.DisableVixlIndirectCalls, "Vector register pairs unsupported by simulator currently");
|
||||
#endif
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
const auto DstLo = GetVReg(IROp->Args[1]);
|
||||
const auto DstHi = GetVReg(IROp->Args[2]);
|
||||
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex].Func));
|
||||
blr(TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
FillF80x2Result(DstLo, DstHi);
|
||||
} break;
|
||||
|
||||
case FABI_F80_I16_F80_F80_PTR: {
|
||||
// Linux Reg/Win32 Reg:
|
||||
// tmp4 (x4/x13): FallbackHandler
|
||||
@@ -356,8 +412,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
// vtmp2 (v1/v17): vector source 2
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src2 = GetVReg(IROp->Args[1].ID());
|
||||
const auto Src1 = GetVReg(IROp->Args[0]);
|
||||
const auto Src2 = GetVReg(IROp->Args[1]);
|
||||
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
mov(VTMP2.Q(), Src2.Q());
|
||||
@@ -385,16 +441,16 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
|
||||
stp<ARMEmitter::IndexType::PRE>(TMP1, ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto SrcRAX = GetReg(Op->RAX.ID());
|
||||
const auto SrcRDX = GetReg(Op->RDX.ID());
|
||||
const auto SrcRAX = GetReg(Op->RAX);
|
||||
const auto SrcRDX = GetReg(Op->RDX);
|
||||
const auto Control = Op->Control;
|
||||
|
||||
mov(TMP1, SrcRAX.X());
|
||||
mov(TMP2, SrcRDX.X());
|
||||
movz(ARMEmitter::Size::i32Bit, TMP3, Control);
|
||||
|
||||
const auto Src1 = GetVReg(Op->LHS.ID());
|
||||
const auto Src2 = GetVReg(Op->RHS.ID());
|
||||
const auto Src1 = GetVReg(Op->LHS);
|
||||
const auto Src2 = GetVReg(Op->RHS);
|
||||
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
mov(VTMP2.Q(), Src2.Q());
|
||||
@@ -414,8 +470,8 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) {
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPISTRX>();
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
const auto Src1 = GetVReg(Op->LHS.ID());
|
||||
const auto Src2 = GetVReg(Op->RHS.ID());
|
||||
const auto Src1 = GetVReg(Op->LHS);
|
||||
const auto Src2 = GetVReg(Op->RHS);
|
||||
const auto Control = Op->Control;
|
||||
|
||||
mov(VTMP1.Q(), Src1.Q());
|
||||
@@ -454,6 +510,8 @@ static void DirectBlockDelinker(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Co
|
||||
|
||||
static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
|
||||
auto Thread = Frame->Thread;
|
||||
auto Lock = Thread->LookupCache->AcquireLock();
|
||||
|
||||
bool TFSet = Thread->CurrentFrame->State.flags[X86State::RFLAG_TF_RAW_LOC];
|
||||
uintptr_t HostCode {};
|
||||
auto GuestRip = Record->GuestRIP;
|
||||
@@ -493,17 +551,18 @@ static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame* Fram
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
void Arm64JITCore::Op_NoOp(const IR::IROp_Header* IROp, IR::NodeID Node) {}
|
||||
void Arm64JITCore::Op_NoOp(const IR::IROp_Header* IROp, IR::Ref Node) {}
|
||||
|
||||
Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread)
|
||||
: CPUBackend(Thread, INITIAL_CODE_SIZE, MAX_CODE_SIZE)
|
||||
: CPUBackend(*ctx, Thread)
|
||||
, Arm64Emitter(ctx)
|
||||
, HostSupportsSVE128 {ctx->HostFeatures.SupportsSVE128}
|
||||
, HostSupportsSVE256 {ctx->HostFeatures.SupportsSVE256}
|
||||
, HostSupportsAVX256 {ctx->HostFeatures.SupportsAVX && ctx->HostFeatures.SupportsSVE256}
|
||||
, HostSupportsRPRES {ctx->HostFeatures.SupportsRPRES}
|
||||
, HostSupportsAFP {ctx->HostFeatures.SupportsAFP}
|
||||
, CTX {ctx} {
|
||||
, CTX {ctx}
|
||||
, TempAllocator(ctx->CPUBackendAllocator, 0) {
|
||||
|
||||
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
|
||||
@@ -546,12 +605,10 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
|
||||
|
||||
AArch64.LUDIV = reinterpret_cast<uint64_t>(LUDIV);
|
||||
AArch64.LDIV = reinterpret_cast<uint64_t>(LDIV);
|
||||
AArch64.LUREM = reinterpret_cast<uint64_t>(LUREM);
|
||||
AArch64.LREM = reinterpret_cast<uint64_t>(LREM);
|
||||
}
|
||||
|
||||
// Must be done after Dispatcher init
|
||||
ClearCache();
|
||||
CurrentCodeBuffer = CodeBuffers.GetLatest();
|
||||
ThreadState->LookupCache->Shared = CurrentCodeBuffer->LookupCache.get();
|
||||
|
||||
// Setup dynamic dispatch.
|
||||
if (ParanoidTSO()) {
|
||||
@@ -570,16 +627,24 @@ void Arm64JITCore::EmitDetectionString() {
|
||||
}
|
||||
|
||||
void Arm64JITCore::ClearCache() {
|
||||
// Get the backing code buffer
|
||||
// NOTE: Holding on to the reference here is required to ensure validity of the WriteLock mutex
|
||||
auto PrevCodeBuffer = CurrentCodeBuffer;
|
||||
std::lock_guard lk(PrevCodeBuffer->LookupCache->WriteLock);
|
||||
|
||||
auto CodeBuffer = GetEmptyCodeBuffer();
|
||||
SetBuffer(CodeBuffer->Ptr, CodeBuffer->Size);
|
||||
EmitDetectionString();
|
||||
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*PrevCodeBuffer, *CurrentCodeBuffer->LookupCache);
|
||||
}
|
||||
|
||||
Arm64JITCore::~Arm64JITCore() {}
|
||||
|
||||
bool Arm64JITCore::IsInlineConstant(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const {
|
||||
if (WNode.IsImmediate()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
auto OpHeader = IR->GetOp<IR::IROp_Header>(WNode);
|
||||
|
||||
if (OpHeader->Op == IR::IROps::OP_INLINECONSTANT) {
|
||||
@@ -594,6 +659,10 @@ bool Arm64JITCore::IsInlineConstant(const IR::OrderedNodeWrapper& WNode, uint64_
|
||||
}
|
||||
|
||||
bool Arm64JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const {
|
||||
if (WNode.IsImmediate()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
auto OpHeader = IR->GetOp<IR::IROp_Header>(WNode);
|
||||
|
||||
if (OpHeader->Op == IR::IROps::OP_INLINEENTRYPOINTOFFSET) {
|
||||
@@ -612,22 +681,6 @@ bool Arm64JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode,
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::IR::RegisterClassType Arm64JITCore::GetRegClass(IR::NodeID Node) const {
|
||||
return FEXCore::IR::RegisterClassType {GetPhys(Node).Class};
|
||||
}
|
||||
|
||||
bool Arm64JITCore::IsFPR(IR::NodeID Node) const {
|
||||
auto Class = GetRegClass(Node);
|
||||
|
||||
return Class == IR::FPRClass || Class == IR::FPRFixedClass;
|
||||
}
|
||||
|
||||
bool Arm64JITCore::IsGPR(IR::NodeID Node) const {
|
||||
auto Class = GetRegClass(Node);
|
||||
|
||||
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
|
||||
}
|
||||
|
||||
void Arm64JITCore::EmitInterruptChecks(bool CheckTF) {
|
||||
if (CheckTF) {
|
||||
ARMEmitter::ForwardLabel l_TFUnset;
|
||||
@@ -689,23 +742,23 @@ void Arm64JITCore::EmitInterruptChecks(bool CheckTF) {
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, const FEXCore::IR::RegisterAllocationData* RAData,
|
||||
bool CheckTF) {
|
||||
FEXCore::Core::DebugData* DebugData, bool CheckTF) {
|
||||
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
|
||||
|
||||
JumpTargets.clear();
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
|
||||
this->Entry = Entry;
|
||||
this->RAData = RAData;
|
||||
this->DebugData = DebugData;
|
||||
this->IR = IR;
|
||||
|
||||
// Fairly excessive buffer range to make sure we don't overflow
|
||||
uint32_t BufferRange = SSACount * 16;
|
||||
if ((GetCursorOffset() + BufferRange) > (CurrentCodeBuffer->Size - Utils::FEX_PAGE_SIZE)) {
|
||||
CTX->ClearCodeCache(ThreadState);
|
||||
}
|
||||
uint32_t BufferRange = 0x100 + SSACount * 24;
|
||||
|
||||
// JIT output is first written to a temporary buffer and later relocated to the CodeBuffer.
|
||||
// This minimizes lock contention of CodeBufferWriteMutex.
|
||||
auto TempCodeBuffer = TempAllocator.ReownOrClaimBuffer(BufferRange);
|
||||
SetBuffer(TempCodeBuffer, BufferRange);
|
||||
|
||||
CodeData.BlockBegin = GetCursorAddress<uint8_t*>();
|
||||
|
||||
@@ -748,7 +801,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
|
||||
EmitInterruptChecks(CheckTF);
|
||||
|
||||
SpillSlots = RAData->SpillSlots();
|
||||
SpillSlots = IR->SpillSlots();
|
||||
|
||||
if (SpillSlots) {
|
||||
const auto TotalSpillSlotsSize = SpillSlots * MaxSpillSlotSize;
|
||||
@@ -785,25 +838,22 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
}
|
||||
|
||||
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
|
||||
const auto ID = IR->GetID(CodeNode);
|
||||
switch (IROp->Op) {
|
||||
#define REGISTER_OP_RT(op, x) \
|
||||
case FEXCore::IR::IROps::OP_##op: std::invoke(RT_##x, this, IROp, ID); break
|
||||
case FEXCore::IR::IROps::OP_##op: std::invoke(RT_##x, this, IROp, CodeNode); break
|
||||
#define REGISTER_OP(op, x) \
|
||||
case FEXCore::IR::IROps::OP_##op: Op_##x(IROp, ID); break
|
||||
case FEXCore::IR::IROps::OP_##op: Op_##x(IROp, CodeNode); break
|
||||
|
||||
#define IROP_DISPATCH_DISPATCH
|
||||
#include <FEXCore/IR/IRDefines_Dispatch.inc>
|
||||
#undef REGISTER_OP
|
||||
|
||||
default: Op_Unhandled(IROp, ID); break;
|
||||
default: Op_Unhandled(IROp, CodeNode); break;
|
||||
}
|
||||
}
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->Subblocks.push_back({static_cast<uint32_t>(BlockStartHostCode - CodeData.BlockEntry),
|
||||
static_cast<uint32_t>(GetCursorAddress<uint8_t*>() - BlockStartHostCode)});
|
||||
}
|
||||
DebugData->Subblocks.push_back({static_cast<uint32_t>(BlockStartHostCode - CodeData.BlockEntry),
|
||||
static_cast<uint32_t>(GetCursorAddress<uint8_t*>() - BlockStartHostCode)});
|
||||
}
|
||||
|
||||
// Make sure last branch is generated. It certainly can't be eliminated here.
|
||||
@@ -878,6 +928,49 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
|
||||
JITBlockTail->Size = CodeData.Size;
|
||||
|
||||
// Migrate the compile output from temporary storage to the actual CodeBuffer.
|
||||
// This can block progress in other compiling threads, so the duration of the lock should be as small as possible.
|
||||
{
|
||||
auto CodeBufferLock = std::unique_lock {CodeBuffers.CodeBufferWriteMutex};
|
||||
|
||||
// Query size of generated code
|
||||
const auto TempSize = GetCursorOffset();
|
||||
LOGMAN_THROW_A_FMT(TempSize <= BufferRange, "Exceeded bounds of temporary buffer ({:#x} vs {:#x})", TempSize, BufferRange);
|
||||
|
||||
// Bring CodeBuffer up to date
|
||||
{
|
||||
LOGMAN_THROW_A_FMT(CurrentCodeBuffer->LookupCache.get() == ThreadState->LookupCache->Shared, "INVARIANT VIOLATED: SharedLookupCache "
|
||||
"doesn't match up!\n");
|
||||
if (auto Prev = CheckCodeBufferUpdate()) {
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*Prev, *CurrentCodeBuffer->LookupCache);
|
||||
}
|
||||
|
||||
// NOTE: 16-byte alignment of the new cursor offset must be preserved for block linking records
|
||||
SetBuffer(CurrentCodeBuffer->Ptr, CurrentCodeBuffer->Size);
|
||||
SetCursorOffset(AlignUp(CodeBuffers.LatestOffset, 16));
|
||||
if ((GetCursorOffset() + TempSize) > (CurrentCodeBuffer->Size - Utils::FEX_PAGE_SIZE)) {
|
||||
CTX->ClearCodeCache(ThreadState);
|
||||
}
|
||||
|
||||
Align16B();
|
||||
|
||||
CodeBuffers.LatestOffset = GetCursorOffset();
|
||||
}
|
||||
|
||||
// Adjust host addresses
|
||||
const auto Delta = GetCursorAddress<uint8_t*>() - CodeData.BlockBegin;
|
||||
CodeData.BlockBegin += Delta;
|
||||
CodeData.BlockEntry += Delta;
|
||||
|
||||
// Copy over CodeBuffer contents
|
||||
memcpy(GetCursorAddress<uint8_t*>(), TempCodeBuffer, TempSize);
|
||||
SetCursorOffset(CodeBuffers.LatestOffset + TempSize);
|
||||
|
||||
CodeBuffers.LatestOffset = GetCursorOffset();
|
||||
}
|
||||
|
||||
TempAllocator.DelayedDisownBuffer();
|
||||
|
||||
ClearICache(CodeData.BlockBegin, CodeOnlySize);
|
||||
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
@@ -903,10 +996,8 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
}
|
||||
#endif
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->HostCodeSize = CodeData.Size;
|
||||
DebugData->Relocations = &Relocations;
|
||||
}
|
||||
DebugData->HostCodeSize = CodeData.Size;
|
||||
DebugData->Relocations = &Relocations;
|
||||
|
||||
this->IR = nullptr;
|
||||
|
||||
|
||||
@@ -38,9 +38,8 @@ public:
|
||||
~Arm64JITCore() override;
|
||||
|
||||
[[nodiscard]]
|
||||
CPUBackend::CompiledCode
|
||||
CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData,
|
||||
const FEXCore::IR::RegisterAllocationData* RAData, bool CheckTF) override;
|
||||
CPUBackend::CompiledCode CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, bool CheckTF) override;
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
@@ -65,10 +64,10 @@ private:
|
||||
|
||||
fextl::map<IR::NodeID, ARMEmitter::BiDirectionalLabel> JumpTargets;
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Register GetReg(IR::NodeID Node) const {
|
||||
const auto Reg = GetPhys(Node);
|
||||
Utils::PoolBufferWithTimedRetirement<uint8_t*, 5000, 500> TempAllocator;
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Register GetReg(IR::PhysicalRegister Reg) const {
|
||||
LOGMAN_THROW_A_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
@@ -81,9 +80,17 @@ private:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::VRegister GetVReg(IR::NodeID Node) const {
|
||||
const auto Reg = GetPhys(Node);
|
||||
ARMEmitter::Register GetReg(IR::Ref Node) const {
|
||||
return GetReg(IR::PhysicalRegister(Node));
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Register GetReg(IR::OrderedNodeWrapper Wrap) const {
|
||||
return GetReg(IR::PhysicalRegister(Wrap));
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::VRegister GetVReg(IR::PhysicalRegister Reg) const {
|
||||
LOGMAN_THROW_A_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
@@ -96,15 +103,18 @@ private:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
FEXCore::IR::RegisterClassType GetRegClass(IR::NodeID Node) const;
|
||||
ARMEmitter::VRegister GetVReg(IR::Ref Node) const {
|
||||
return GetVReg(IR::PhysicalRegister(Node));
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
IR::PhysicalRegister GetPhys(IR::NodeID Node) const {
|
||||
auto PhyReg = RAData->GetNodeRegister(Node);
|
||||
ARMEmitter::VRegister GetVReg(IR::OrderedNodeWrapper Wrap) const {
|
||||
return GetVReg(IR::PhysicalRegister(Wrap));
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(!PhyReg.IsInvalid(), "Couldn't Allocate register for node: ssa{}. Class: {}", Node, PhyReg.Class);
|
||||
|
||||
return PhyReg;
|
||||
[[nodiscard]]
|
||||
FEXCore::IR::RegisterClassType GetRegClass(IR::Ref Node) const {
|
||||
return FEXCore::IR::RegisterClassType {IR::PhysicalRegister(Node).Class};
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
@@ -114,7 +124,7 @@ private:
|
||||
LOGMAN_THROW_A_FMT(Const == 0, "Only valid constant");
|
||||
return ARMEmitter::Reg::zr;
|
||||
} else {
|
||||
return GetReg(Src.ID());
|
||||
return GetReg(Src);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -224,9 +234,34 @@ private:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsFPR(IR::NodeID Node) const;
|
||||
bool IsFPR(IR::RegisterClassType Class) const {
|
||||
return Class == IR::FPRClass || Class == IR::FPRFixedClass;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsGPR(IR::NodeID Node) const;
|
||||
bool IsGPR(IR::RegisterClassType Class) const {
|
||||
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsGPR(IR::Ref Node) {
|
||||
return IsGPR(GetRegClass(Node));
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsFPR(IR::Ref Node) {
|
||||
return IsFPR(GetRegClass(Node));
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsGPR(IR::OrderedNodeWrapper Wrap) {
|
||||
return IsGPR(IR::RegisterClassType {IR::PhysicalRegister(Wrap).Class});
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsFPR(IR::OrderedNodeWrapper Wrap) {
|
||||
return IsFPR(IR::RegisterClassType {IR::PhysicalRegister(Wrap).Class});
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::ExtendedMemOperand GenerateMemOperand(IR::OpSize AccessSize, ARMEmitter::Register Base, IR::OrderedNodeWrapper Offset,
|
||||
@@ -258,7 +293,6 @@ private:
|
||||
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
|
||||
void EmitDetectionString();
|
||||
IR::RegisterAllocationPass* RAPass {};
|
||||
const IR::RegisterAllocationData* RAData {};
|
||||
FEXCore::Core::DebugData* DebugData {};
|
||||
|
||||
void ResetStack();
|
||||
@@ -319,7 +353,7 @@ private:
|
||||
/** @} */
|
||||
|
||||
uint32_t SpillSlots {};
|
||||
using OpType = void (Arm64JITCore::*)(const IR::IROp_Header* IROp, IR::NodeID Node);
|
||||
using OpType = void (Arm64JITCore::*)(const IR::IROp_Header* IROp, IR::Ref Node);
|
||||
|
||||
using ScalarFMAOpCaller =
|
||||
std::function<void(ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2, ARMEmitter::VRegister Src3)>;
|
||||
@@ -346,7 +380,7 @@ private:
|
||||
OpType RT_LoadMemTSO;
|
||||
OpType RT_StoreMemTSO;
|
||||
|
||||
#define DEF_OP(x) void Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
#define DEF_OP(x) void Op_##x(IR::IROp_Header const* IROp, IR::Ref Node)
|
||||
|
||||
// Dynamic Dispatcher supporting operations
|
||||
DEF_OP(ParanoidLoadMemTSO);
|
||||
@@ -363,6 +397,8 @@ private:
|
||||
#undef DEF_OP
|
||||
};
|
||||
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::Ref Node)
|
||||
|
||||
[[nodiscard]]
|
||||
fextl::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
|
||||
@@ -11,11 +11,11 @@ $end_info$
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
#include "Interface/IR/RegisterAllocationData.h"
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(LoadContext) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
@@ -53,8 +53,8 @@ DEF_OP(LoadContextPair) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadContextPair>();
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst1 = GetReg(Op->OutValue1.ID());
|
||||
const auto Dst2 = GetReg(Op->OutValue2.ID());
|
||||
const auto Dst1 = GetReg(Op->OutValue1);
|
||||
const auto Dst2 = GetReg(Op->OutValue2);
|
||||
|
||||
switch (IROp->Size) {
|
||||
case IR::OpSize::i32Bit: ldp<ARMEmitter::IndexType::OFFSET>(Dst1.W(), Dst2.W(), STATE, Op->Offset); break;
|
||||
@@ -62,8 +62,8 @@ DEF_OP(LoadContextPair) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemPair size: {}", IROp->Size); break;
|
||||
}
|
||||
} else {
|
||||
const auto Dst1 = GetVReg(Op->OutValue1.ID());
|
||||
const auto Dst2 = GetVReg(Op->OutValue2.ID());
|
||||
const auto Dst1 = GetVReg(Op->OutValue1);
|
||||
const auto Dst2 = GetVReg(Op->OutValue2);
|
||||
|
||||
switch (IROp->Size) {
|
||||
case IR::OpSize::i32Bit: ldp<ARMEmitter::IndexType::OFFSET>(Dst1.S(), Dst2.S(), STATE, Op->Offset); break;
|
||||
@@ -89,7 +89,7 @@ DEF_OP(StoreContext) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
const auto Src = GetVReg(Op->Value.ID());
|
||||
const auto Src = GetVReg(Op->Value);
|
||||
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: strb(Src, STATE, Op->Offset); break;
|
||||
@@ -120,8 +120,8 @@ DEF_OP(StoreContextPair) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
const auto Src1 = GetVReg(Op->Value1.ID());
|
||||
const auto Src2 = GetVReg(Op->Value2.ID());
|
||||
const auto Src1 = GetVReg(Op->Value1);
|
||||
const auto Src2 = GetVReg(Op->Value2);
|
||||
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i32Bit: stp<ARMEmitter::IndexType::OFFSET>(Src1.S(), Src2.S(), STATE, Op->Offset); break;
|
||||
@@ -137,11 +137,8 @@ DEF_OP(LoadRegister) {
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
LOGMAN_THROW_A_FMT(Op->Reg < StaticRegisters.size(), "out of range reg");
|
||||
const auto reg = StaticRegisters[Op->Reg];
|
||||
|
||||
if (GetReg(Node).Idx() != reg.Idx()) {
|
||||
mov(GetReg(Node).X(), reg.X());
|
||||
}
|
||||
mov(GetReg(Node).X(), StaticRegisters[Op->Reg].X());
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
[[maybe_unused]] const auto regSize = HostSupportsAVX256 ? IR::OpSize::i256Bit : IR::OpSize::i128Bit;
|
||||
LOGMAN_THROW_A_FMT(Op->Reg < StaticFPRegisters.size(), "out of range reg");
|
||||
@@ -150,12 +147,10 @@ DEF_OP(LoadRegister) {
|
||||
const auto guest = StaticFPRegisters[Op->Reg];
|
||||
const auto host = GetVReg(Node);
|
||||
|
||||
if (host.Idx() != guest.Idx()) {
|
||||
if (HostSupportsAVX256) {
|
||||
mov(ARMEmitter::SubRegSize::i64Bit, host.Z(), PRED_TMP_32B.Merging(), guest.Z());
|
||||
} else {
|
||||
mov(host.Q(), guest.Q());
|
||||
}
|
||||
if (HostSupportsAVX256) {
|
||||
mov(ARMEmitter::SubRegSize::i64Bit, host.Z(), PRED_TMP_32B.Merging(), guest.Z());
|
||||
} else {
|
||||
mov(host.Q(), guest.Q());
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
@@ -180,44 +175,32 @@ DEF_OP(LoadAF) {
|
||||
|
||||
DEF_OP(StoreRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
auto Reg = IR::PhysicalRegister(Node);
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
unsigned Reg = Op->Reg == Core::CPUState::PF_AS_GREG ? (StaticRegisters.size() - 2) :
|
||||
Op->Reg == Core::CPUState::AF_AS_GREG ? (StaticRegisters.size() - 1) :
|
||||
Op->Reg;
|
||||
|
||||
LOGMAN_THROW_A_FMT(Reg < StaticRegisters.size(), "out of range reg");
|
||||
const auto reg = StaticRegisters[Reg];
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
if (Src.Idx() != reg.Idx()) {
|
||||
// Always use 64-bit, it's faster. Upper bits ignored for 32-bit mode.
|
||||
mov(ARMEmitter::Size::i64Bit, reg, Src);
|
||||
}
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
if (Reg.Class == IR::GPRFixedClass) {
|
||||
// Always use 64-bit, it's faster. Upper bits ignored for 32-bit mode.
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Reg), GetReg(Op->Value));
|
||||
} else if (Reg.Class == IR::FPRFixedClass) {
|
||||
[[maybe_unused]] const auto regSize = HostSupportsAVX256 ? IR::OpSize::i256Bit : IR::OpSize::i128Bit;
|
||||
LOGMAN_THROW_A_FMT(Op->Reg < StaticFPRegisters.size(), "reg out of range");
|
||||
LOGMAN_THROW_A_FMT(IROp->Size == regSize, "expected sized");
|
||||
|
||||
const auto guest = StaticFPRegisters[Op->Reg];
|
||||
const auto host = GetVReg(Op->Value.ID());
|
||||
const auto guest = GetVReg(Reg);
|
||||
const auto host = GetVReg(Op->Value);
|
||||
|
||||
if (guest.Idx() != host.Idx()) {
|
||||
if (HostSupportsAVX256) {
|
||||
mov(ARMEmitter::SubRegSize::i64Bit, guest.Z(), PRED_TMP_32B.Merging(), host.Z());
|
||||
} else {
|
||||
mov(guest.Q(), host.Q());
|
||||
}
|
||||
if (HostSupportsAVX256) {
|
||||
mov(ARMEmitter::SubRegSize::i64Bit, guest.Z(), PRED_TMP_32B.Merging(), host.Z());
|
||||
} else {
|
||||
mov(guest.Q(), host.Q());
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", Reg.Class);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StorePF) {
|
||||
const auto Op = IROp->C<IR::IROp_StorePF>();
|
||||
const auto reg = StaticRegisters[StaticRegisters.size() - 2];
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
const auto Src = GetReg(Op->Value);
|
||||
|
||||
if (Src.Idx() != reg.Idx()) {
|
||||
// Always use 64-bit, it's faster. Upper bits ignored for 32-bit mode.
|
||||
@@ -228,7 +211,7 @@ DEF_OP(StorePF) {
|
||||
DEF_OP(StoreAF) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreAF>();
|
||||
const auto reg = StaticRegisters[StaticRegisters.size() - 1];
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
const auto Src = GetReg(Op->Value);
|
||||
|
||||
if (Src.Idx() != reg.Idx()) {
|
||||
// Always use 64-bit, it's faster. Upper bits ignored for 32-bit mode.
|
||||
@@ -240,7 +223,7 @@ DEF_OP(LoadContextIndexed) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Index = GetReg(Op->Index.ID());
|
||||
const auto Index = GetReg(Op->Index);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
switch (Op->Stride) {
|
||||
@@ -303,10 +286,10 @@ DEF_OP(StoreContextIndexed) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Index = GetReg(Op->Index.ID());
|
||||
const auto Index = GetReg(Op->Index);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Value = GetReg(Op->Value.ID());
|
||||
const auto Value = GetReg(Op->Value);
|
||||
|
||||
switch (Op->Stride) {
|
||||
case 1:
|
||||
@@ -328,7 +311,7 @@ DEF_OP(StoreContextIndexed) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed stride: {}", Op->Stride); break;
|
||||
}
|
||||
} else {
|
||||
const auto Value = GetVReg(Op->Value.ID());
|
||||
const auto Value = GetVReg(Op->Value);
|
||||
|
||||
switch (Op->Stride) {
|
||||
case 1:
|
||||
@@ -371,7 +354,7 @@ DEF_OP(SpillRegister) {
|
||||
const uint32_t SlotOffset = Op->Slot * MaxSpillSlotSize;
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
const auto Src = GetReg(Op->Value);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: {
|
||||
if (SlotOffset > LSByteMaxUnsignedOffset) {
|
||||
@@ -412,7 +395,7 @@ DEF_OP(SpillRegister) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize); break;
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::FPRClass) {
|
||||
const auto Src = GetVReg(Op->Value.ID());
|
||||
const auto Src = GetVReg(Op->Value);
|
||||
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i32Bit: {
|
||||
@@ -552,7 +535,7 @@ DEF_OP(LoadNZCV) {
|
||||
DEF_OP(StoreNZCV) {
|
||||
auto Op = IROp->C<IR::IROp_StoreNZCV>();
|
||||
|
||||
msr(ARMEmitter::SystemRegister::NZCV, GetReg(Op->Value.ID()));
|
||||
msr(ARMEmitter::SystemRegister::NZCV, GetReg(Op->Value));
|
||||
}
|
||||
|
||||
DEF_OP(LoadDF) {
|
||||
@@ -575,7 +558,7 @@ ARMEmitter::ExtendedMemOperand Arm64JITCore::GenerateMemOperand(
|
||||
if (IsInlineConstant(Offset, &Const)) {
|
||||
return ARMEmitter::ExtendedMemOperand(Base.X(), ARMEmitter::IndexType::OFFSET, Const);
|
||||
} else {
|
||||
auto RegOffset = GetReg(Offset.ID());
|
||||
auto RegOffset = GetReg(Offset);
|
||||
switch (OffsetType.Val) {
|
||||
case IR::MEM_OFFSET_SXTX.Val:
|
||||
return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::SXTX, FEXCore::ilog2(OffsetScale));
|
||||
@@ -609,7 +592,7 @@ ARMEmitter::Register Arm64JITCore::ApplyMemOperand(IR::OpSize AccessSize, ARMEmi
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Tmp, Const);
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, Tmp, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(OffsetScale));
|
||||
} else {
|
||||
auto RegOffset = GetReg(Offset.ID());
|
||||
auto RegOffset = GetReg(Offset);
|
||||
switch (OffsetType.Val) {
|
||||
case IR::MEM_OFFSET_SXTX.Val:
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, RegOffset, ARMEmitter::ExtendedType::SXTX, FEXCore::ilog2(OffsetScale));
|
||||
@@ -676,7 +659,7 @@ ARMEmitter::SVEMemOperand Arm64JITCore::GenerateSVEMemOperand(IR::OpSize AccessS
|
||||
// optional extension or shift as part of their behavior.
|
||||
LOGMAN_THROW_A_FMT(OffsetType.Val == IR::MEM_OFFSET_SXTX.Val, "Currently only the default offset type (SXTX) is supported.");
|
||||
|
||||
const auto RegOffset = GetReg(Offset.ID());
|
||||
const auto RegOffset = GetReg(Offset);
|
||||
return ARMEmitter::SVEMemOperand(Base.X(), RegOffset.X());
|
||||
}
|
||||
|
||||
@@ -684,7 +667,7 @@ DEF_OP(LoadMem) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMem>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
@@ -719,11 +702,11 @@ DEF_OP(LoadMem) {
|
||||
|
||||
DEF_OP(LoadMemPair) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemPair>();
|
||||
const auto Addr = GetReg(Op->Addr.ID());
|
||||
const auto Addr = GetReg(Op->Addr);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst1 = GetReg(Op->OutValue1.ID());
|
||||
const auto Dst2 = GetReg(Op->OutValue2.ID());
|
||||
const auto Dst1 = GetReg(Op->OutValue1);
|
||||
const auto Dst2 = GetReg(Op->OutValue2);
|
||||
|
||||
switch (IROp->Size) {
|
||||
case IR::OpSize::i32Bit: ldp<ARMEmitter::IndexType::OFFSET>(Dst1.W(), Dst2.W(), Addr, Op->Offset); break;
|
||||
@@ -731,8 +714,8 @@ DEF_OP(LoadMemPair) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemPair size: {}", IROp->Size); break;
|
||||
}
|
||||
} else {
|
||||
const auto Dst1 = GetVReg(Op->OutValue1.ID());
|
||||
const auto Dst2 = GetVReg(Op->OutValue2.ID());
|
||||
const auto Dst1 = GetVReg(Op->OutValue1);
|
||||
const auto Dst2 = GetVReg(Op->OutValue2);
|
||||
|
||||
switch (IROp->Size) {
|
||||
case IR::OpSize::i32Bit: ldp<ARMEmitter::IndexType::OFFSET>(Dst1.S(), Dst2.S(), Addr, Op->Offset); break;
|
||||
@@ -747,7 +730,7 @@ DEF_OP(LoadMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid() || CTX->HostFeatures.SupportsTSOImm9, "unexpected offset");
|
||||
@@ -844,8 +827,8 @@ DEF_OP(VLoadVectorMasked) {
|
||||
const auto GoverningPredicate = Is256Bit ? PRED_TMP_32B : PRED_TMP_16B;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto MaskReg = GetVReg(Op->Mask.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto MaskReg = GetVReg(Op->Mask);
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
if (HostSupportsSVE128 || HostSupportsSVE256) {
|
||||
const auto MemSrc = GenerateSVEMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
@@ -946,9 +929,9 @@ DEF_OP(VStoreVectorMasked) {
|
||||
const auto CMPPredicate = ARMEmitter::PReg::p0;
|
||||
const auto GoverningPredicate = Is256Bit ? PRED_TMP_32B : PRED_TMP_16B;
|
||||
|
||||
const auto RegData = GetVReg(Op->Data.ID());
|
||||
const auto MaskReg = GetVReg(Op->Mask.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto RegData = GetVReg(Op->Data);
|
||||
const auto MaskReg = GetVReg(Op->Mask);
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
if (HostSupportsSVE128 || HostSupportsSVE256) {
|
||||
const auto MemDst = GenerateSVEMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
@@ -1171,13 +1154,13 @@ DEF_OP(VLoadVectorGatherMasked) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto IncomingDst = GetVReg(Op->Incoming.ID());
|
||||
const auto IncomingDst = GetVReg(Op->Incoming);
|
||||
|
||||
const auto MaskReg = GetVReg(Op->Mask.ID());
|
||||
std::optional<ARMEmitter::Register> BaseAddr = !Op->AddrBase.IsInvalid() ? std::make_optional(GetReg(Op->AddrBase.ID())) : std::nullopt;
|
||||
const auto VectorIndexLow = GetVReg(Op->VectorIndexLow.ID());
|
||||
const auto MaskReg = GetVReg(Op->Mask);
|
||||
std::optional<ARMEmitter::Register> BaseAddr = !Op->AddrBase.IsInvalid() ? std::make_optional(GetReg(Op->AddrBase)) : std::nullopt;
|
||||
const auto VectorIndexLow = GetVReg(Op->VectorIndexLow);
|
||||
std::optional<ARMEmitter::VRegister> VectorIndexHigh =
|
||||
!Op->VectorIndexHigh.IsInvalid() ? std::make_optional(GetVReg(Op->VectorIndexHigh.ID())) : std::nullopt;
|
||||
!Op->VectorIndexHigh.IsInvalid() ? std::make_optional(GetVReg(Op->VectorIndexHigh)) : std::nullopt;
|
||||
|
||||
///< If the host supports SVE and the offset scale matches SVE limitations then it can do an SVE style load.
|
||||
const bool SupportsSVELoad = (HostSupportsSVE128 || HostSupportsSVE256) &&
|
||||
@@ -1206,7 +1189,7 @@ DEF_OP(VLoadVectorGatherMasked) {
|
||||
if (BaseAddr.has_value() || OffsetScale != 1) {
|
||||
ARMEmitter::Register AddrReg = TMP1;
|
||||
if (BaseAddr.has_value()) {
|
||||
AddrReg = GetReg(Op->AddrBase.ID());
|
||||
AddrReg = GetReg(Op->AddrBase);
|
||||
} else {
|
||||
///< OpcodeDispatcher didn't provide a Base address while SVE requires one.
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, AddrReg, 0);
|
||||
@@ -1255,13 +1238,13 @@ DEF_OP(VLoadVectorGatherMaskedQPS) {
|
||||
/// - Matches VGATHERQPS/VPGATHERQD behaviour!
|
||||
const auto OffsetScale = Op->OffsetScale;
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto IncomingDst = GetVReg(Op->Incoming.ID());
|
||||
const auto IncomingDst = GetVReg(Op->Incoming);
|
||||
|
||||
const auto MaskReg = GetVReg(Op->MaskReg.ID());
|
||||
std::optional<ARMEmitter::Register> BaseAddr = !Op->AddrBase.IsInvalid() ? std::make_optional(GetReg(Op->AddrBase.ID())) : std::nullopt;
|
||||
const auto VectorIndexLow = GetVReg(Op->VectorIndexLow.ID());
|
||||
const auto MaskReg = GetVReg(Op->MaskReg);
|
||||
std::optional<ARMEmitter::Register> BaseAddr = !Op->AddrBase.IsInvalid() ? std::make_optional(GetReg(Op->AddrBase)) : std::nullopt;
|
||||
const auto VectorIndexLow = GetVReg(Op->VectorIndexLow);
|
||||
std::optional<ARMEmitter::VRegister> VectorIndexHigh =
|
||||
!Op->VectorIndexHigh.IsInvalid() ? std::make_optional(GetVReg(Op->VectorIndexHigh.ID())) : std::nullopt;
|
||||
!Op->VectorIndexHigh.IsInvalid() ? std::make_optional(GetVReg(Op->VectorIndexHigh)) : std::nullopt;
|
||||
|
||||
///< If the host supports SVE and the offset scale matches SVE limitations then it can do an SVE style load.
|
||||
if (HostSupportsSVE128 && (OffsetScale == 1 || OffsetScale == 4)) {
|
||||
@@ -1330,8 +1313,8 @@ DEF_OP(VLoadVectorElement) {
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto DstSrc = GetVReg(Op->DstSrc.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto DstSrc = GetVReg(Op->DstSrc);
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
LOGMAN_THROW_A_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
|
||||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
|
||||
@@ -1367,8 +1350,8 @@ DEF_OP(VStoreVectorElement) {
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
|
||||
const auto Value = GetVReg(Op->Value.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto Value = GetVReg(Op->Value);
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
LOGMAN_THROW_A_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
|
||||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
|
||||
@@ -1403,7 +1386,7 @@ DEF_OP(VBroadcastFromMem) {
|
||||
const auto ElementSize = IROp->ElementSize;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto MemReg = GetReg(Op->Address.ID());
|
||||
const auto MemReg = GetReg(Op->Address);
|
||||
|
||||
LOGMAN_THROW_A_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
|
||||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
|
||||
@@ -1444,8 +1427,8 @@ DEF_OP(VBroadcastFromMem) {
|
||||
DEF_OP(Push) {
|
||||
const auto Op = IROp->C<IR::IROp_Push>();
|
||||
const auto ValueSize = IR::OpSizeToSize(Op->ValueSize);
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
const auto AddrSrc = GetReg(Op->Addr.ID());
|
||||
auto Src = GetReg(Op->Value);
|
||||
const auto AddrSrc = GetReg(Op->Addr);
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
bool NeedsMoveAfterwards = false;
|
||||
@@ -1526,11 +1509,34 @@ DEF_OP(Push) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(PushTwo) {
|
||||
const auto Op = IROp->C<IR::IROp_PushTwo>();
|
||||
const auto ValueSize = IR::OpSizeToSize(Op->ValueSize);
|
||||
auto Src1 = GetReg(Op->Value1);
|
||||
auto Src2 = GetReg(Op->Value2);
|
||||
const auto Dst = GetReg(Op->Addr);
|
||||
|
||||
switch (ValueSize) {
|
||||
case 4: {
|
||||
stp<ARMEmitter::IndexType::PRE>(Src1.W(), Src2.W(), Dst, -2 * ValueSize);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
stp<ARMEmitter::IndexType::PRE>(Src1.X(), Src2.X(), Dst, -2 * ValueSize);
|
||||
break;
|
||||
}
|
||||
default: {
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, ValueSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Pop) {
|
||||
const auto Op = IROp->C<IR::IROp_Pop>();
|
||||
const auto Size = IR::OpSizeToSize(Op->Size);
|
||||
const auto Addr = GetReg(Op->InoutAddr.ID());
|
||||
const auto Dst = GetReg(Op->OutValue.ID());
|
||||
const auto Addr = GetReg(Op->InoutAddr);
|
||||
const auto Dst = GetReg(Op->OutValue);
|
||||
|
||||
LOGMAN_THROW_A_FMT(Dst != Addr, "Invalid");
|
||||
|
||||
@@ -1558,11 +1564,42 @@ DEF_OP(Pop) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(PopTwo) {
|
||||
const auto Op = IROp->C<IR::IROp_PopTwo>();
|
||||
const auto Size = IR::OpSizeToSize(Op->Size);
|
||||
const auto Addr = GetReg(Op->InoutAddr);
|
||||
auto Dst1 = GetReg(Op->OutValue1);
|
||||
const auto Dst2 = GetReg(Op->OutValue2);
|
||||
|
||||
// ldp x, x is invalid. Explicitly discard the first destination to encode.
|
||||
if (Dst1 == Dst2) {
|
||||
Dst1 = ARMEmitter::Reg::zr;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Dst1 != Addr && Dst2 != Addr, "Invalid");
|
||||
LOGMAN_THROW_A_FMT(Dst1 != Dst2, "Invalid");
|
||||
|
||||
switch (Size) {
|
||||
case 4: {
|
||||
ldp<ARMEmitter::IndexType::POST>(Dst1.W(), Dst2.W(), Addr, 2 * Size);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
ldp<ARMEmitter::IndexType::POST>(Dst1.X(), Dst2.X(), Addr, 2 * Size);
|
||||
break;
|
||||
}
|
||||
default: {
|
||||
LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Op->Size);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreMem) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMem>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
@@ -1575,7 +1612,7 @@ DEF_OP(StoreMem) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
const auto Src = GetVReg(Op->Value.ID());
|
||||
const auto Src = GetVReg(Op->Value);
|
||||
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: {
|
||||
@@ -1615,8 +1652,8 @@ DEF_OP(StoreMemX87SVEOptPredicate) {
|
||||
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE128 || HostSupportsSVE256, "StoreMemX87SVEOptPredicate needs SVE support");
|
||||
|
||||
const auto RegData = GetVReg(Op->Value.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto RegData = GetVReg(Op->Value);
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
const auto MemDst = ARMEmitter::SVEMemOperand(MemReg.X(), 0);
|
||||
|
||||
switch (IROp->ElementSize) {
|
||||
@@ -1644,7 +1681,7 @@ DEF_OP(LoadMemX87SVEOptPredicate) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemX87SVEOptPredicate>();
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Predicate = PRED_X87_SVEOPT;
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE128 || HostSupportsSVE256, "LoadMemX87SVEOptPredicate needs SVE support");
|
||||
|
||||
@@ -1674,7 +1711,7 @@ DEF_OP(LoadMemX87SVEOptPredicate) {
|
||||
DEF_OP(StoreMemPair) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMemPair>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Addr = GetReg(Op->Addr.ID());
|
||||
const auto Addr = GetReg(Op->Addr);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src1 = GetZeroableReg(Op->Value1);
|
||||
@@ -1685,8 +1722,8 @@ DEF_OP(StoreMemPair) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
const auto Src1 = GetVReg(Op->Value1.ID());
|
||||
const auto Src2 = GetVReg(Op->Value2.ID());
|
||||
const auto Src1 = GetVReg(Op->Value1);
|
||||
const auto Src2 = GetVReg(Op->Value2);
|
||||
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i32Bit: stp<ARMEmitter::IndexType::OFFSET>(Src1.S(), Src2.S(), Addr, Op->Offset); break;
|
||||
@@ -1701,7 +1738,7 @@ DEF_OP(StoreMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid() || CTX->HostFeatures.SupportsTSOImm9, "unexpected offset");
|
||||
@@ -1751,7 +1788,7 @@ DEF_OP(StoreMemTSO) {
|
||||
// Half-Barrier.
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
}
|
||||
const auto Src = GetVReg(Op->Value.ID());
|
||||
const auto Src = GetVReg(Op->Value);
|
||||
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: strb(Src, MemSrc); break;
|
||||
@@ -1782,16 +1819,16 @@ DEF_OP(MemSet) {
|
||||
|
||||
const bool IsAtomic = CTX->IsMemcpyAtomicTSOEnabled();
|
||||
const auto Size = IR::OpSizeToSize(Op->Size);
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
const auto Value = GetZeroableReg(Op->Value);
|
||||
const auto Length = GetReg(Op->Length.ID());
|
||||
const auto Length = GetReg(Op->Length);
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
uint64_t DirectionConstant;
|
||||
bool DirectionIsInline = IsInlineConstant(Op->Direction, &DirectionConstant);
|
||||
ARMEmitter::Register DirectionReg = ARMEmitter::Reg::r0;
|
||||
if (!DirectionIsInline) {
|
||||
DirectionReg = GetReg(Op->Direction.ID());
|
||||
DirectionReg = GetReg(Op->Direction);
|
||||
}
|
||||
|
||||
// If Direction > 0 then:
|
||||
@@ -1808,7 +1845,7 @@ DEF_OP(MemSet) {
|
||||
if (Op->Prefix.IsInvalid()) {
|
||||
mov(TMP2, MemReg.X());
|
||||
} else {
|
||||
const auto Prefix = GetReg(Op->Prefix.ID());
|
||||
const auto Prefix = GetReg(Op->Prefix);
|
||||
add(TMP2, Prefix.X(), MemReg.X());
|
||||
}
|
||||
|
||||
@@ -1971,19 +2008,19 @@ DEF_OP(MemCpy) {
|
||||
|
||||
const bool IsAtomic = CTX->IsMemcpyAtomicTSOEnabled();
|
||||
const auto Size = IR::OpSizeToSize(Op->Size);
|
||||
const auto MemRegDest = GetReg(Op->Dest.ID());
|
||||
const auto MemRegSrc = GetReg(Op->Src.ID());
|
||||
const auto MemRegDest = GetReg(Op->Dest);
|
||||
const auto MemRegSrc = GetReg(Op->Src);
|
||||
|
||||
const auto Length = GetReg(Op->Length.ID());
|
||||
const auto Length = GetReg(Op->Length);
|
||||
uint64_t DirectionConstant;
|
||||
bool DirectionIsInline = IsInlineConstant(Op->Direction, &DirectionConstant);
|
||||
ARMEmitter::Register DirectionReg = ARMEmitter::Reg::r0;
|
||||
if (!DirectionIsInline) {
|
||||
DirectionReg = GetReg(Op->Direction.ID());
|
||||
DirectionReg = GetReg(Op->Direction);
|
||||
}
|
||||
|
||||
auto Dst0 = GetReg(Op->OutDstAddress.ID());
|
||||
auto Dst1 = GetReg(Op->OutSrcAddress.ID());
|
||||
auto Dst0 = GetReg(Op->OutDstAddress);
|
||||
auto Dst1 = GetReg(Op->OutSrcAddress);
|
||||
// If Direction > 0 then:
|
||||
// MemRegDest is incremented (by size)
|
||||
// MemRegSrc is incremented (by size)
|
||||
@@ -2241,7 +2278,7 @@ DEF_OP(ParanoidLoadMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
auto MemReg = GetReg(Op->Addr.ID());
|
||||
auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
@@ -2329,7 +2366,7 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
auto MemReg = GetReg(Op->Addr.ID());
|
||||
auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
@@ -2362,7 +2399,7 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
const auto Src = GetVReg(Op->Value.ID());
|
||||
const auto Src = GetVReg(Op->Value);
|
||||
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
@@ -2416,7 +2453,7 @@ DEF_OP(CacheLineClear) {
|
||||
|
||||
auto Op = IROp->C<IR::IROp_CacheLineClear>();
|
||||
|
||||
auto MemReg = GetReg(Op->Addr.ID());
|
||||
auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
// Clear dcache only
|
||||
// icache doesn't matter here since the guest application shouldn't be calling clflush on JIT code.
|
||||
@@ -2445,7 +2482,7 @@ DEF_OP(CacheLineClean) {
|
||||
|
||||
auto Op = IROp->C<IR::IROp_CacheLineClean>();
|
||||
|
||||
auto MemReg = GetReg(Op->Addr.ID());
|
||||
auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
// Clean dcache only
|
||||
if (CTX->HostFeatures.DCacheLineSize >= 64U) {
|
||||
@@ -2463,7 +2500,7 @@ DEF_OP(CacheLineClean) {
|
||||
DEF_OP(CacheLineZero) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineZero>();
|
||||
|
||||
auto MemReg = GetReg(Op->Addr.ID());
|
||||
auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
if (CTX->HostFeatures.SupportsCLZERO) {
|
||||
// We can use this instruction directly
|
||||
@@ -2483,7 +2520,7 @@ DEF_OP(CacheLineZero) {
|
||||
|
||||
DEF_OP(Prefetch) {
|
||||
auto Op = IROp->C<IR::IROp_Prefetch>();
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
// Access size is only ever handled as 8-byte. Even though it is accesssed as a cacheline.
|
||||
const auto MemSrc = GenerateMemOperand(IR::OpSize::i64Bit, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
@@ -2527,8 +2564,8 @@ DEF_OP(VStoreNonTemporal) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
|
||||
|
||||
const auto Value = GetVReg(Op->Value.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto Value = GetVReg(Op->Value);
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
const auto Offset = Op->Offset;
|
||||
|
||||
if (Is256Bit) {
|
||||
@@ -2552,10 +2589,10 @@ DEF_OP(VStoreNonTemporalPair) {
|
||||
[[maybe_unused]] const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "This IR operation only operates at 128-bit wide");
|
||||
|
||||
const auto ValueLow = GetVReg(Op->ValueLow.ID());
|
||||
const auto ValueHigh = GetVReg(Op->ValueHigh.ID());
|
||||
const auto ValueLow = GetVReg(Op->ValueLow);
|
||||
const auto ValueHigh = GetVReg(Op->ValueHigh);
|
||||
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
const auto Offset = Op->Offset;
|
||||
|
||||
stnp(ValueLow.Q(), ValueHigh.Q(), MemReg, Offset);
|
||||
@@ -2570,7 +2607,7 @@ DEF_OP(VLoadNonTemporal) {
|
||||
const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
const auto Offset = Op->Offset;
|
||||
|
||||
if (Is256Bit) {
|
||||
@@ -2587,5 +2624,4 @@ DEF_OP(VLoadNonTemporal) {
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -16,11 +16,6 @@ $end_info$
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(AllocateGPR) {}
|
||||
DEF_OP(AllocateGPRAfter) {}
|
||||
DEF_OP(AllocateFPR) {}
|
||||
|
||||
DEF_OP(GuestOpcode) {
|
||||
auto Op = IROp->C<IR::IROp_GuestOpcode>();
|
||||
@@ -102,8 +97,8 @@ DEF_OP(GetRoundingMode) {
|
||||
|
||||
DEF_OP(SetRoundingMode) {
|
||||
auto Op = IROp->C<IR::IROp_SetRoundingMode>();
|
||||
auto Src = GetReg(Op->RoundMode.ID());
|
||||
auto MXCSR = GetReg(Op->MXCSR.ID());
|
||||
auto Src = GetReg(Op->RoundMode);
|
||||
auto MXCSR = GetReg(Op->MXCSR);
|
||||
|
||||
// As above, setup the rounding flags in [31:30]
|
||||
rbit(ARMEmitter::Size::i32Bit, TMP2, Src);
|
||||
@@ -161,7 +156,7 @@ DEF_OP(PushRoundingMode) {
|
||||
|
||||
DEF_OP(PopRoundingMode) {
|
||||
auto Op = IROp->C<IR::IROp_PopRoundingMode>();
|
||||
msr(ARMEmitter::SystemRegister::FPCR, GetReg(Op->FPCR.ID()));
|
||||
msr(ARMEmitter::SystemRegister::FPCR, GetReg(Op->FPCR));
|
||||
}
|
||||
|
||||
DEF_OP(Print) {
|
||||
@@ -170,17 +165,17 @@ DEF_OP(Print) {
|
||||
PushDynamicRegs(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
if (IsGPR(Op->Value.ID())) {
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetReg(Op->Value.ID()));
|
||||
if (IsGPR(Op->Value)) {
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetReg(Op->Value));
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintValue));
|
||||
} else {
|
||||
fmov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetVReg(Op->Value.ID()), false);
|
||||
fmov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, GetVReg(Op->Value.ID()), true);
|
||||
fmov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetVReg(Op->Value), false);
|
||||
fmov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, GetVReg(Op->Value), true);
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintVectorValue));
|
||||
}
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
if (IsGPR(Op->Value.ID())) {
|
||||
if (IsGPR(Op->Value)) {
|
||||
GenerateIndirectRuntimeCall<void, uint64_t>(ARMEmitter::Reg::r3);
|
||||
} else {
|
||||
GenerateIndirectRuntimeCall<void, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
@@ -271,5 +266,4 @@ DEF_OP(Yield) {
|
||||
yield();
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -8,26 +8,19 @@ $end_info$
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
DEF_OP(Copy) {
|
||||
auto Op = IROp->C<IR::IROp_Copy>();
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), GetReg(Op->Source.ID()));
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), GetReg(Op->Source));
|
||||
}
|
||||
|
||||
DEF_OP(RMWHandle) {
|
||||
auto Op = IROp->C<IR::IROp_RMWHandle>();
|
||||
auto Dest = GetReg(Node);
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
if (Dest != Src) {
|
||||
mov(ARMEmitter::Size::i64Bit, Dest, Src);
|
||||
}
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), GetReg(IROp->Args[0]));
|
||||
}
|
||||
|
||||
DEF_OP(Swap1) {
|
||||
auto Op = IROp->C<IR::IROp_Swap1>();
|
||||
auto A = GetReg(Op->A.ID()), B = GetReg(Op->B.ID());
|
||||
auto A = GetReg(Op->A), B = GetReg(Op->B);
|
||||
LOGMAN_THROW_A_FMT(B == GetReg(Node), "Invariant");
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, TMP1, A);
|
||||
@@ -39,5 +32,4 @@ DEF_OP(Swap2) {
|
||||
// Implemented above
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -11,7 +11,6 @@ $end_info$
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node)
|
||||
|
||||
#define DEF_UNOP(FEXOp, ARMOp, ScalarCase) \
|
||||
DEF_OP(FEXOp) { \
|
||||
@@ -24,7 +23,7 @@ namespace FEXCore::CPU {
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp); \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto Src = GetVReg(Op->Vector.ID()); \
|
||||
const auto Src = GetVReg(Op->Vector); \
|
||||
\
|
||||
if (HostSupportsSVE256 && Is256Bit) { \
|
||||
ARMOp(SubRegSize, Dst.Z(), PRED_TMP_32B.Merging(), Src.Z()); \
|
||||
@@ -45,8 +44,8 @@ namespace FEXCore::CPU {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__); \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID()); \
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID()); \
|
||||
const auto Vector1 = GetVReg(Op->Vector1); \
|
||||
const auto Vector2 = GetVReg(Op->Vector2); \
|
||||
\
|
||||
if (HostSupportsSVE256 && Is256Bit) { \
|
||||
ARMOp(Dst.Z(), Vector1.Z(), Vector2.Z()); \
|
||||
@@ -65,8 +64,8 @@ namespace FEXCore::CPU {
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp); \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID()); \
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID()); \
|
||||
const auto Vector1 = GetVReg(Op->Vector1); \
|
||||
const auto Vector2 = GetVReg(Op->Vector2); \
|
||||
\
|
||||
if (HostSupportsSVE256 && Is256Bit) { \
|
||||
ARMOp(SubRegSize, Dst.Z(), Vector1.Z(), Vector2.Z()); \
|
||||
@@ -85,8 +84,8 @@ namespace FEXCore::CPU {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__); \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto VectorLower = GetVReg(Op->VectorLower.ID()); \
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper.ID()); \
|
||||
const auto VectorLower = GetVReg(Op->VectorLower); \
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper); \
|
||||
\
|
||||
if (HostSupportsSVE256 && Is256Bit) { \
|
||||
ARMOp(SubRegSize, Dst.Z(), VectorLower.Z(), VectorUpper.Z()); \
|
||||
@@ -110,7 +109,7 @@ namespace FEXCore::CPU {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__); \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto Src = GetVReg(Op->Vector.ID()); \
|
||||
const auto Src = GetVReg(Op->Vector); \
|
||||
\
|
||||
if (HostSupportsSVE256 && Is256Bit) { \
|
||||
ARMOp(SubRegSize, Dst.Z(), PRED_TMP_32B.Merging(), Src.Z()); \
|
||||
@@ -149,8 +148,8 @@ namespace FEXCore::CPU {
|
||||
const auto IsScalar = ElementSize == OpSize; \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID()); \
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID()); \
|
||||
const auto Vector1 = GetVReg(Op->Vector1); \
|
||||
const auto Vector2 = GetVReg(Op->Vector2); \
|
||||
\
|
||||
if (HostSupportsSVE256 && Is256Bit) { \
|
||||
ARMOp(SubRegSize, Dst.Z(), Vector1.Z(), Vector2.Z()); \
|
||||
@@ -188,8 +187,8 @@ namespace FEXCore::CPU {
|
||||
}; \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID()); \
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID()); \
|
||||
const auto Vector1 = GetVReg(Op->Vector1); \
|
||||
const auto Vector2 = GetVReg(Op->Vector2); \
|
||||
\
|
||||
VFScalarOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2); \
|
||||
}
|
||||
@@ -211,10 +210,10 @@ namespace FEXCore::CPU {
|
||||
}; \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
const auto Upper = GetVReg(Op->Upper.ID()); \
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID()); \
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID()); \
|
||||
const auto Addend = GetVReg(Op->Addend.ID()); \
|
||||
const auto Upper = GetVReg(Op->Upper); \
|
||||
const auto Vector1 = GetVReg(Op->Vector1); \
|
||||
const auto Vector2 = GetVReg(Op->Vector2); \
|
||||
const auto Addend = GetVReg(Op->Addend); \
|
||||
\
|
||||
VFScalarFMAOperation(IROp->Size, ElementSize, ScalarEmit, Dst, Upper, Vector1, Vector2, Addend); \
|
||||
}
|
||||
@@ -458,8 +457,8 @@ DEF_OP(VFMinScalarInsert) {
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
VFScalarOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
|
||||
}
|
||||
@@ -484,8 +483,8 @@ DEF_OP(VFMaxScalarInsert) {
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
VFScalarOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
|
||||
}
|
||||
@@ -503,8 +502,8 @@ DEF_OP(VFSqrtScalarInsert) {
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
VFScalarUnaryOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
|
||||
}
|
||||
@@ -541,8 +540,8 @@ DEF_OP(VFRSqrtScalarInsert) {
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
VFScalarUnaryOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, Handlers[HandlerIndex], Dst, Vector1, Vector2);
|
||||
}
|
||||
@@ -578,8 +577,8 @@ DEF_OP(VFRecpScalarInsert) {
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
VFScalarUnaryOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, Handlers[HandlerIndex], Dst, Vector1, Vector2);
|
||||
}
|
||||
@@ -624,8 +623,8 @@ DEF_OP(VFToFScalarInsert) {
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
VFScalarUnaryOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
|
||||
}
|
||||
@@ -656,8 +655,8 @@ DEF_OP(VSToFVectorInsert) {
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
// Claim the element size is 8-bytes.
|
||||
// Might be scalar 8-byte (cvtsi2ss xmm0, rax)
|
||||
@@ -729,8 +728,8 @@ DEF_OP(VSToFGPRInsert) {
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto GPR = GetReg(Op->Src.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
const auto GPR = GetReg(Op->Src);
|
||||
|
||||
VFScalarUnaryOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector, GPR);
|
||||
}
|
||||
@@ -756,8 +755,8 @@ DEF_OP(VFToIScalarInsert) {
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
VFScalarUnaryOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
|
||||
}
|
||||
@@ -916,8 +915,8 @@ DEF_OP(VFCMPScalarInsert) {
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
VFScalarOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, Funcs[FEXCore::ToUnderlying(Op->Op)], Dst, Vector1, Vector2);
|
||||
}
|
||||
@@ -1033,7 +1032,7 @@ DEF_OP(VMov) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Source = GetVReg(Op->Source.ID());
|
||||
const auto Source = GetVReg(Op->Source);
|
||||
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: {
|
||||
@@ -1087,8 +1086,8 @@ DEF_OP(VAddP) {
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorLower = GetVReg(Op->VectorLower.ID());
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper.ID());
|
||||
const auto VectorLower = GetVReg(Op->VectorLower);
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Pred = PRED_TMP_32B.Merging();
|
||||
@@ -1125,7 +1124,7 @@ DEF_OP(VFAddV) {
|
||||
const auto SubRegSize = ConvertSubRegSizePair248(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i128Bit || OpSize == IR::OpSize::i256Bit, "Only AVX and SSE size supported");
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
@@ -1156,7 +1155,7 @@ DEF_OP(VAddV) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// SVE doesn't have an equivalent ADDV instruction, so we make do
|
||||
@@ -1192,7 +1191,7 @@ DEF_OP(VUMinV) {
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Pred = PRED_TMP_32B;
|
||||
@@ -1213,7 +1212,7 @@ DEF_OP(VUMaxV) {
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Pred = PRED_TMP_32B;
|
||||
@@ -1233,8 +1232,8 @@ DEF_OP(VURAvg) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -1267,8 +1266,8 @@ DEF_OP(VFAddP) {
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorLower = GetVReg(Op->VectorLower.ID());
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper.ID());
|
||||
const auto VectorLower = GetVReg(Op->VectorLower);
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Pred = PRED_TMP_32B.Merging();
|
||||
@@ -1302,8 +1301,8 @@ DEF_OP(VFDiv) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -1358,8 +1357,8 @@ DEF_OP(VFMin) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
// NOTE: We don't directly use FMIN** here for any of the implementations,
|
||||
// because it has undesirable NaN handling behavior (it sets
|
||||
@@ -1431,8 +1430,8 @@ DEF_OP(VFMax) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
// NOTE: See VFMin implementation for reasons why we
|
||||
// don't just use FMAX/FMIN for these implementations.
|
||||
@@ -1489,7 +1488,7 @@ DEF_OP(VFRecp) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Pred = PRED_TMP_32B.Merging();
|
||||
@@ -1559,7 +1558,7 @@ DEF_OP(VFRecpPrecision) {
|
||||
const auto IsScalar = OpSize == ElementSize;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (IsScalar) {
|
||||
if (ElementSize == IR::OpSize::i32Bit && HostSupportsRPRES) {
|
||||
@@ -1598,7 +1597,7 @@ DEF_OP(VFRSqrt) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Pred = PRED_TMP_32B.Merging();
|
||||
@@ -1669,7 +1668,7 @@ DEF_OP(VFRSqrtPrecision) {
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (IsScalar) {
|
||||
if (HostSupportsRPRES) {
|
||||
@@ -1707,7 +1706,7 @@ DEF_OP(VNot) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
not_(ARMEmitter::SubRegSize::i8Bit, Dst.Z(), PRED_TMP_32B.Merging(), Vector.Z());
|
||||
@@ -1726,8 +1725,8 @@ DEF_OP(VUMin) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Pred = PRED_TMP_32B.Merging();
|
||||
@@ -1775,8 +1774,8 @@ DEF_OP(VSMin) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Pred = PRED_TMP_32B.Merging();
|
||||
@@ -1824,8 +1823,8 @@ DEF_OP(VUMax) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Pred = PRED_TMP_32B.Merging();
|
||||
@@ -1873,8 +1872,8 @@ DEF_OP(VSMax) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Pred = PRED_TMP_32B.Merging();
|
||||
@@ -1921,9 +1920,9 @@ DEF_OP(VBSL) {
|
||||
const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorFalse = GetVReg(Op->VectorFalse.ID());
|
||||
const auto VectorTrue = GetVReg(Op->VectorTrue.ID());
|
||||
const auto VectorMask = GetVReg(Op->VectorMask.ID());
|
||||
const auto VectorFalse = GetVReg(Op->VectorFalse);
|
||||
const auto VectorTrue = GetVReg(Op->VectorTrue);
|
||||
const auto VectorMask = GetVReg(Op->VectorMask);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// NOTE: Slight parameter difference from ASIMD
|
||||
@@ -1990,8 +1989,8 @@ DEF_OP(VCMPEQ) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// FIXME: We should rework this op to avoid the NZCV spill/fill dance.
|
||||
@@ -2030,7 +2029,7 @@ DEF_OP(VCMPEQZ) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
@@ -2072,8 +2071,8 @@ DEF_OP(VCMPGT) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
@@ -2112,7 +2111,7 @@ DEF_OP(VCMPGTZ) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
@@ -2151,7 +2150,7 @@ DEF_OP(VCMPLTZ) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
@@ -2190,8 +2189,8 @@ DEF_OP(VFCMPEQ) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
@@ -2229,8 +2228,8 @@ DEF_OP(VFCMPNEQ) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
@@ -2270,8 +2269,8 @@ DEF_OP(VFCMPLT) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
@@ -2309,8 +2308,8 @@ DEF_OP(VFCMPGT) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
@@ -2348,8 +2347,8 @@ DEF_OP(VFCMPLE) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
@@ -2387,8 +2386,8 @@ DEF_OP(VFCMPORD) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
@@ -2438,8 +2437,8 @@ DEF_OP(VFCMPUNO) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Zeroing();
|
||||
@@ -2489,8 +2488,8 @@ DEF_OP(VUShl) {
|
||||
const auto MaxShift = IR::OpSizeAsBits(ElementSize);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
auto ShiftVector = GetVReg(Op->ShiftVector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
auto ShiftVector = GetVReg(Op->ShiftVector);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
const auto RangeCheck = Op->RangeCheck;
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
@@ -2545,8 +2544,8 @@ DEF_OP(VUShr) {
|
||||
const auto MaxShift = IR::OpSizeAsBits(ElementSize);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
auto ShiftVector = GetVReg(Op->ShiftVector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
auto ShiftVector = GetVReg(Op->ShiftVector);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
const auto RangeCheck = Op->RangeCheck;
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
@@ -2605,8 +2604,8 @@ DEF_OP(VSShr) {
|
||||
const auto RangeCheck = Op->RangeCheck;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
auto ShiftVector = GetVReg(Op->ShiftVector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
auto ShiftVector = GetVReg(Op->ShiftVector);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -2660,8 +2659,8 @@ DEF_OP(VUShlS) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar.ID());
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -2689,8 +2688,8 @@ DEF_OP(VUShrS) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar.ID());
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -2720,8 +2719,8 @@ DEF_OP(VUShrSWide) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar.ID());
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -2786,8 +2785,8 @@ DEF_OP(VSShrSWide) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar.ID());
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -2852,8 +2851,8 @@ DEF_OP(VUShlSWide) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar.ID());
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -2916,8 +2915,8 @@ DEF_OP(VSShrS) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar.ID());
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto ShiftScalar = GetVReg(Op->ShiftScalar);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -2950,8 +2949,8 @@ DEF_OP(VInsElement) {
|
||||
const uint32_t SrcIdx = Op->SrcIdx;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto SrcVector = GetVReg(Op->SrcVector.ID());
|
||||
auto Reg = GetVReg(Op->DestVector.ID());
|
||||
const auto SrcVector = GetVReg(Op->SrcVector);
|
||||
auto Reg = GetVReg(Op->DestVector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// Broadcast our source value across a temporary,
|
||||
@@ -3029,7 +3028,7 @@ DEF_OP(VDupElement) {
|
||||
const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
dup(SubRegSize, Dst.Z(), Vector.Z(), Index);
|
||||
@@ -3050,8 +3049,8 @@ DEF_OP(VExtr) {
|
||||
|
||||
// AArch64 ext op has bit arrangement as [Vm:Vn] so arguments need to be swapped
|
||||
const auto Dst = GetVReg(Node);
|
||||
auto UpperBits = GetVReg(Op->VectorLower.ID());
|
||||
auto LowerBits = GetVReg(Op->VectorUpper.ID());
|
||||
auto UpperBits = GetVReg(Op->VectorLower);
|
||||
auto LowerBits = GetVReg(Op->VectorUpper);
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
auto Index = Op->Index;
|
||||
@@ -3101,7 +3100,7 @@ DEF_OP(VUShrI) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (BitShift >= IR::OpSizeAsBits(ElementSize)) {
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), 0);
|
||||
@@ -3143,8 +3142,8 @@ DEF_OP(VUShraI) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto DestVector = GetVReg(Op->DestVector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto DestVector = GetVReg(Op->DestVector);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
if (Dst == DestVector) {
|
||||
@@ -3187,7 +3186,7 @@ DEF_OP(VSShrI) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -3226,7 +3225,7 @@ DEF_OP(VShlI) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (BitShift >= IR::OpSizeAsBits(ElementSize)) {
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), 0);
|
||||
@@ -3268,7 +3267,7 @@ DEF_OP(VUShrNI) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
shrnb(SubRegSize, Dst.Z(), Vector.Z(), BitShift);
|
||||
@@ -3292,8 +3291,8 @@ DEF_OP(VUShrNI2) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorLower = GetVReg(Op->VectorLower.ID());
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper.ID());
|
||||
const auto VectorLower = GetVReg(Op->VectorLower);
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_16B;
|
||||
@@ -3329,7 +3328,7 @@ DEF_OP(VSXTL) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if ((HostSupportsSVE128 && !Is256Bit && !HostSupportsSVE256) || (HostSupportsSVE256 && Is256Bit)) {
|
||||
sunpklo(SubRegSize, Dst.Z(), Vector.Z());
|
||||
@@ -3347,7 +3346,7 @@ DEF_OP(VSXTL2) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if ((HostSupportsSVE128 && !Is256Bit && !HostSupportsSVE256) || (HostSupportsSVE256 && Is256Bit)) {
|
||||
sunpkhi(SubRegSize, Dst.Z(), Vector.Z());
|
||||
@@ -3365,7 +3364,7 @@ DEF_OP(VSSHLL) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
const auto BitShift = Op->BitShift;
|
||||
LOGMAN_THROW_A_FMT(BitShift < IR::OpSizeAsBits(IROp->ElementSize / 2), "Bitshift size too large for source element size: {} < {}",
|
||||
BitShift, IR::OpSizeAsBits(IROp->ElementSize / 2));
|
||||
@@ -3387,7 +3386,7 @@ DEF_OP(VSSHLL2) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
const auto BitShift = Op->BitShift;
|
||||
LOGMAN_THROW_A_FMT(BitShift < IR::OpSizeAsBits(IROp->ElementSize / 2), "Bitshift size too large for source element size: {} < {}",
|
||||
BitShift, IR::OpSizeAsBits(IROp->ElementSize / 2));
|
||||
@@ -3409,7 +3408,7 @@ DEF_OP(VUXTL) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if ((HostSupportsSVE128 && !Is256Bit && !HostSupportsSVE256) || (HostSupportsSVE256 && Is256Bit)) {
|
||||
uunpklo(SubRegSize, Dst.Z(), Vector.Z());
|
||||
@@ -3427,7 +3426,7 @@ DEF_OP(VUXTL2) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if ((HostSupportsSVE128 && !Is256Bit && !HostSupportsSVE256) || (HostSupportsSVE256 && Is256Bit)) {
|
||||
uunpkhi(SubRegSize, Dst.Z(), Vector.Z());
|
||||
@@ -3445,7 +3444,7 @@ DEF_OP(VSQXTN) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// Note that SVE SQXTNB and SQXTNT are a tad different
|
||||
@@ -3497,8 +3496,8 @@ DEF_OP(VSQXTN2) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorLower = GetVReg(Op->VectorLower.ID());
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper.ID());
|
||||
const auto VectorLower = GetVReg(Op->VectorLower);
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// We use the 16 byte mask due to how SPLICE works. We only
|
||||
@@ -3541,8 +3540,8 @@ DEF_OP(VSQXTNPair) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorLower = GetVReg(Op->VectorLower.ID());
|
||||
auto VectorUpper = GetVReg(Op->VectorUpper.ID());
|
||||
const auto VectorLower = GetVReg(Op->VectorLower);
|
||||
auto VectorUpper = GetVReg(Op->VectorUpper);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// This combines the SVE versions of VSQXTN/VSQXTN2.
|
||||
@@ -3585,7 +3584,7 @@ DEF_OP(VSQXTUN) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
sqxtunb(SubRegSize, Dst.Z(), Vector.Z());
|
||||
@@ -3604,8 +3603,8 @@ DEF_OP(VSQXTUN2) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorLower = GetVReg(Op->VectorLower.ID());
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper.ID());
|
||||
const auto VectorLower = GetVReg(Op->VectorLower);
|
||||
const auto VectorUpper = GetVReg(Op->VectorUpper);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// NOTE: See VSQXTN2 implementation for an in-depth explanation
|
||||
@@ -3650,8 +3649,8 @@ DEF_OP(VSQXTUNPair) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorLower = GetVReg(Op->VectorLower.ID());
|
||||
auto VectorUpper = GetVReg(Op->VectorUpper.ID());
|
||||
const auto VectorLower = GetVReg(Op->VectorLower);
|
||||
auto VectorUpper = GetVReg(Op->VectorUpper);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// This combines the SVE versions of VSQXTUN/VSQXTUN2.
|
||||
@@ -3695,7 +3694,7 @@ DEF_OP(VSRSHR) {
|
||||
const auto SubRegSize = ConvertSubRegSize16(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
const auto BitShift = Op->BitShift;
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
@@ -3725,7 +3724,7 @@ DEF_OP(VSQSHL) {
|
||||
const auto SubRegSize = ConvertSubRegSize8(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
const auto BitShift = Op->BitShift;
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
@@ -3755,8 +3754,8 @@ DEF_OP(VMul) {
|
||||
const auto SubRegSize = ConvertSubRegSize16(IROp);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
mul(SubRegSize, Dst.Z(), Vector1.Z(), Vector2.Z());
|
||||
@@ -3774,8 +3773,8 @@ DEF_OP(VUMull) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
umullb(SubRegSize, VTMP1.Z(), Vector1.Z(), Vector2.Z());
|
||||
@@ -3795,8 +3794,8 @@ DEF_OP(VSMull) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
smullb(SubRegSize, VTMP1.Z(), Vector1.Z(), Vector2.Z());
|
||||
@@ -3816,8 +3815,8 @@ DEF_OP(VUMull2) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
umullb(SubRegSize, VTMP1.Z(), Vector1.Z(), Vector2.Z());
|
||||
@@ -3837,8 +3836,8 @@ DEF_OP(VSMull2) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
smullb(SubRegSize, VTMP1.Z(), Vector1.Z(), Vector2.Z());
|
||||
@@ -3861,8 +3860,8 @@ DEF_OP(VUMulH) {
|
||||
const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
const auto SubRegSizeLarger = ElementSize == IR::OpSize::i8Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == IR::OpSize::i16Bit ? ARMEmitter::SubRegSize::i32Bit :
|
||||
@@ -3912,8 +3911,8 @@ DEF_OP(VSMulH) {
|
||||
const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
const auto SubRegSizeLarger = ElementSize == IR::OpSize::i8Bit ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == IR::OpSize::i16Bit ? ARMEmitter::SubRegSize::i32Bit :
|
||||
@@ -3960,8 +3959,8 @@ DEF_OP(VUABDL) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// To mimic the behavior of AdvSIMD UABDL, we need to get the
|
||||
@@ -3986,8 +3985,8 @@ DEF_OP(VUABDL2) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// To mimic the behavior of AdvSIMD UABDL, we need to get the
|
||||
@@ -4008,8 +4007,8 @@ DEF_OP(VTBL1) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorIndices = GetVReg(Op->VectorIndices.ID());
|
||||
const auto VectorTable = GetVReg(Op->VectorTable.ID());
|
||||
const auto VectorIndices = GetVReg(Op->VectorIndices);
|
||||
const auto VectorTable = GetVReg(Op->VectorTable);
|
||||
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i64Bit: {
|
||||
@@ -4035,9 +4034,9 @@ DEF_OP(VTBL2) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorIndices = GetVReg(Op->VectorIndices.ID());
|
||||
auto VectorTable1 = GetVReg(Op->VectorTable1.ID());
|
||||
auto VectorTable2 = GetVReg(Op->VectorTable2.ID());
|
||||
const auto VectorIndices = GetVReg(Op->VectorIndices);
|
||||
auto VectorTable1 = GetVReg(Op->VectorTable1);
|
||||
auto VectorTable2 = GetVReg(Op->VectorTable2);
|
||||
|
||||
if (!ARMEmitter::AreVectorsSequential(VectorTable1, VectorTable2)) {
|
||||
// Vector registers aren't sequential, need to move to temporaries.
|
||||
@@ -4079,9 +4078,9 @@ DEF_OP(VTBX1) {
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto VectorSrcDst = GetVReg(Op->VectorSrcDst.ID());
|
||||
const auto VectorIndices = GetVReg(Op->VectorIndices.ID());
|
||||
const auto VectorTable = GetVReg(Op->VectorTable.ID());
|
||||
const auto VectorSrcDst = GetVReg(Op->VectorSrcDst);
|
||||
const auto VectorIndices = GetVReg(Op->VectorIndices);
|
||||
const auto VectorTable = GetVReg(Op->VectorTable);
|
||||
|
||||
if (Dst != VectorSrcDst) {
|
||||
switch (OpSize) {
|
||||
@@ -4136,7 +4135,7 @@ DEF_OP(VRev32) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
LOGMAN_THROW_A_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit, "Invalid size");
|
||||
const auto SubRegSize = ElementSize == IR::OpSize::i8Bit ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i16Bit;
|
||||
@@ -4175,7 +4174,7 @@ DEF_OP(VRev64) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -4213,8 +4212,8 @@ DEF_OP(VFCADD) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
LOGMAN_THROW_A_FMT(Op->Rotate == 90 || Op->Rotate == 270, "Invalidate Rotate");
|
||||
const auto Rotate = Op->Rotate == 90 ? ARMEmitter::Rotation::ROTATE_90 : ARMEmitter::Rotation::ROTATE_270;
|
||||
@@ -4260,9 +4259,9 @@ DEF_OP(VFMLA) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto VectorAddend = GetVReg(Op->Addend.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
const auto VectorAddend = GetVReg(Op->Addend);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -4327,9 +4326,9 @@ DEF_OP(VFMLS) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto VectorAddend = GetVReg(Op->Addend.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
const auto VectorAddend = GetVReg(Op->Addend);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -4417,9 +4416,9 @@ DEF_OP(VFNMLA) {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto VectorAddend = GetVReg(Op->Addend.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
const auto VectorAddend = GetVReg(Op->Addend);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -4487,9 +4486,9 @@ DEF_OP(VFNMLS) {
|
||||
const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
const auto VectorAddend = GetVReg(Op->Addend.ID());
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
const auto VectorAddend = GetVReg(Op->Addend);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
@@ -4568,8 +4567,8 @@ DEF_OP(VFCopySign) {
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
|
||||
ARMEmitter::VRegister Magnitude = GetVReg(Op->Vector1.ID());
|
||||
ARMEmitter::VRegister Sign = GetVReg(Op->Vector2.ID());
|
||||
ARMEmitter::VRegister Magnitude = GetVReg(Op->Vector1);
|
||||
ARMEmitter::VRegister Sign = GetVReg(Op->Vector2);
|
||||
|
||||
// We don't assign explicity to Dst but Dst and Magniture are tied to the same register.
|
||||
// Similar in semantics to C's copysignf.
|
||||
@@ -4586,6 +4585,4 @@ DEF_OP(VFCopySign) {
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
#undef DEF_OP
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -14,14 +14,17 @@ $end_info$
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
namespace FEXCore {
|
||||
LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
|
||||
: BlockLinks_mbr {fextl::pmr::get_default_resource()}
|
||||
, ctx {CTX} {
|
||||
|
||||
TotalCacheSize = ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE + L1_SIZE;
|
||||
GuestToHostMap::GuestToHostMap()
|
||||
: BlockLinks_mbr {fextl::pmr::get_default_resource()} {
|
||||
BlockLinks_pma = fextl::make_unique<std::pmr::polymorphic_allocator<std::byte>>(&BlockLinks_mbr);
|
||||
// Setup our PMR map.
|
||||
BlockLinks = BlockLinks_pma->new_object<BlockLinksMapType>();
|
||||
}
|
||||
|
||||
LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
|
||||
: ctx {CTX} {
|
||||
|
||||
TotalCacheSize = ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE + L1_SIZE;
|
||||
|
||||
// Block cache ends up looking like this
|
||||
// PageMemoryMap[VirtualMemoryRegion >> 12]
|
||||
@@ -62,18 +65,30 @@ LookupCache::~LookupCache() {
|
||||
}
|
||||
|
||||
void LookupCache::ClearL2Cache() {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
auto lk = Shared->AcquireLock();
|
||||
// Clear out the page memory
|
||||
// PagePointer and PageMemory are sequential with each other. Clear both at once.
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE, false);
|
||||
AllocateOffset = 0;
|
||||
}
|
||||
|
||||
void LookupCache::ClearCache() {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
void LookupCache::ClearThreadLocalCaches() {
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
// Clear L1 and L2 by clearing the full cache.
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), TotalCacheSize, false);
|
||||
}
|
||||
|
||||
void LookupCache::ClearCache() {
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
// Clear L1 and L2 by clearing the full cache.
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), TotalCacheSize, false);
|
||||
|
||||
Shared->ClearCache(lk);
|
||||
}
|
||||
|
||||
void GuestToHostMap::ClearCache(const LockToken&) {
|
||||
// Allocate a new pointer from the BlockLinks pma again.
|
||||
BlockLinks = BlockLinks_pma->new_object<BlockLinksMapType>();
|
||||
// All code is gone, clear the block list
|
||||
|
||||
@@ -16,6 +16,86 @@
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
struct GuestToHostMap {
|
||||
std::recursive_mutex WriteLock;
|
||||
|
||||
struct LockToken {
|
||||
std::lock_guard<std::recursive_mutex> Lock;
|
||||
};
|
||||
|
||||
[[nodiscard]]
|
||||
LockToken AcquireLock() {
|
||||
return LockToken {std::lock_guard {WriteLock}};
|
||||
}
|
||||
|
||||
struct BlockLinkTag {
|
||||
uint64_t GuestDestination;
|
||||
FEXCore::Context::ExitFunctionLinkData* HostLink;
|
||||
|
||||
bool operator<(const BlockLinkTag& other) const {
|
||||
if (GuestDestination < other.GuestDestination) {
|
||||
return true;
|
||||
} else if (GuestDestination == other.GuestDestination) {
|
||||
return HostLink < other.HostLink;
|
||||
} else {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
// Use a monotonic buffer resource to allocate both the std::pmr::map and its members.
|
||||
// This allows us to quickly clear the block link map by clearing the monotonic allocator.
|
||||
// If we had allocated the block link map without the MBR, then clearing the map would require slowly
|
||||
// walking each block member and destructing objects.
|
||||
//
|
||||
// This makes `BlockLinks` look like a raw pointer that could memory leak, but since it is backed by the MBR, it won't.
|
||||
std::pmr::monotonic_buffer_resource BlockLinks_mbr;
|
||||
using BlockLinksMapType = std::pmr::map<BlockLinkTag, FEXCore::Context::BlockDelinkerFunc>;
|
||||
fextl::unique_ptr<std::pmr::polymorphic_allocator<std::byte>> BlockLinks_pma;
|
||||
BlockLinksMapType* BlockLinks;
|
||||
|
||||
fextl::robin_map<uint64_t, uint64_t> BlockList;
|
||||
|
||||
GuestToHostMap();
|
||||
|
||||
// Adds to Guest -> Host code mapping
|
||||
void AddBlockMapping(uint64_t Address, void* HostCode, const LockToken&) {
|
||||
// This may replace an existing mapping
|
||||
// NOTE: Generally no previous entry should exist, however there is one exception:
|
||||
// If the backend updates the active thread's CodeBuffer, the new associated LookupCache
|
||||
// may already contain the block address. Since is comparatively rare, we'll just leak
|
||||
// one of the two blocks in this case.
|
||||
BlockList[Address] = (uintptr_t)HostCode;
|
||||
}
|
||||
|
||||
std::optional<uintptr_t> FindBlock(uint64_t Address, const LockToken&) {
|
||||
auto HostCode = BlockList.find(Address);
|
||||
if (HostCode == BlockList.end()) {
|
||||
return std::nullopt;
|
||||
}
|
||||
return HostCode->second;
|
||||
}
|
||||
|
||||
void Erase(FEXCore::Core::CpuStateFrame* Frame, uint64_t Address, const LockToken&) {
|
||||
// Sever any links to this block
|
||||
auto lower = BlockLinks->lower_bound({Address, nullptr});
|
||||
auto upper = BlockLinks->upper_bound({Address, reinterpret_cast<FEXCore::Context::ExitFunctionLinkData*>(UINTPTR_MAX)});
|
||||
for (auto it = lower; it != upper; it = BlockLinks->erase(it)) {
|
||||
it->second(Frame, it->first.HostLink);
|
||||
}
|
||||
|
||||
// Remove from BlockList
|
||||
BlockList.erase(Address);
|
||||
}
|
||||
|
||||
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink,
|
||||
const FEXCore::Context::BlockDelinkerFunc& delinker, const LockToken&) {
|
||||
BlockLinks->insert({{GuestDestination, HostLink}, delinker});
|
||||
}
|
||||
|
||||
void ClearCache(const LockToken&);
|
||||
};
|
||||
|
||||
class LookupCache {
|
||||
public:
|
||||
struct LookupCacheEntry {
|
||||
@@ -26,6 +106,13 @@ public:
|
||||
LookupCache(FEXCore::Context::ContextImpl* CTX);
|
||||
~LookupCache();
|
||||
|
||||
// Swaps out the underlying GuestToHostMap and clears all associated caches.
|
||||
// This interface requires the previous CodeBuffer to be provided despite not using it. This ensures the shared write lock is still valid.
|
||||
void ChangeGuestToHostMapping([[maybe_unused]] CPU::CodeBuffer& Prev, GuestToHostMap& NewMap) {
|
||||
ClearThreadLocalCaches();
|
||||
Shared = &NewMap;
|
||||
}
|
||||
|
||||
uintptr_t FindBlock(uint64_t Address) {
|
||||
// Try L1, no lock needed
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
@@ -34,7 +121,7 @@ public:
|
||||
}
|
||||
|
||||
// L2 and L3 need to be locked
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
// Try L2
|
||||
const auto PageIndex = (Address & (VirtualMemSize - 1)) >> 12;
|
||||
@@ -56,29 +143,30 @@ public:
|
||||
}
|
||||
|
||||
// Try L3
|
||||
auto HostCode = BlockList.find(Address);
|
||||
|
||||
if (HostCode != BlockList.end()) {
|
||||
CacheBlockMapping(Address, HostCode->second);
|
||||
return HostCode->second;
|
||||
auto HostCode = Shared->FindBlock(Address, lk);
|
||||
if (HostCode) {
|
||||
CacheBlockMapping(Address, HostCode.value());
|
||||
return HostCode.value();
|
||||
}
|
||||
|
||||
// Failed to find
|
||||
return 0;
|
||||
}
|
||||
|
||||
GuestToHostMap* Shared = nullptr;
|
||||
|
||||
fextl::map<uint64_t, fextl::vector<uint64_t>> CodePages;
|
||||
|
||||
// Appends Block {Address} to CodePages [Start, Start + Length)
|
||||
// Returns true if new pages are marked as containing code
|
||||
bool AddBlockExecutableRange(uint64_t Address, uint64_t Start, uint64_t Length) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
bool rv = false;
|
||||
|
||||
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length - 1) >> 12; CurrentPage <= EndPage; CurrentPage++) {
|
||||
auto& CodePage = CodePages[CurrentPage];
|
||||
rv |= CodePage.size() == 0;
|
||||
rv |= CodePage.empty();
|
||||
CodePage.push_back(Address);
|
||||
}
|
||||
|
||||
@@ -87,10 +175,9 @@ public:
|
||||
|
||||
// Adds to Guest -> Host code mapping
|
||||
void AddBlockMapping(uint64_t Address, void* HostCode) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
[[maybe_unused]] auto Inserted = BlockList.emplace(Address, (uintptr_t)HostCode).second;
|
||||
LOGMAN_THROW_A_FMT(Inserted, "Duplicate block mapping added");
|
||||
Shared->AddBlockMapping(Address, HostCode, lk);
|
||||
|
||||
// There is no need to update L1 or L2, they will get updated on first lookup
|
||||
// However, adding to L1 here increases performance
|
||||
@@ -99,19 +186,13 @@ public:
|
||||
L1Entry.HostCode = (uintptr_t)HostCode;
|
||||
}
|
||||
|
||||
// NOTE: It's the caller's responsibility to call Erase() for all other
|
||||
// GuestToHostMaps that share the same LookupCache. Otherwise, the
|
||||
// L1/L2 caches will contain stale references to deallocated memory.
|
||||
void Erase(FEXCore::Core::CpuStateFrame* Frame, uint64_t Address) {
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
// Sever any links to this block
|
||||
auto lower = BlockLinks->lower_bound({Address, nullptr});
|
||||
auto upper = BlockLinks->upper_bound({Address, reinterpret_cast<FEXCore::Context::ExitFunctionLinkData*>(UINTPTR_MAX)});
|
||||
for (auto it = lower; it != upper; it = BlockLinks->erase(it)) {
|
||||
it->second(Frame, it->first.HostLink);
|
||||
}
|
||||
|
||||
// Remove from BlockList
|
||||
BlockList.erase(Address);
|
||||
Shared->Erase(Frame, Address, lk);
|
||||
|
||||
// Do L1
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
@@ -141,13 +222,13 @@ public:
|
||||
}
|
||||
|
||||
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink, const FEXCore::Context::BlockDelinkerFunc& delinker) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
BlockLinks->insert({{GuestDestination, HostLink}, delinker});
|
||||
auto lk = Shared->AcquireLock();
|
||||
Shared->AddBlockLink(GuestDestination, HostLink, delinker, lk);
|
||||
}
|
||||
|
||||
void ClearCache();
|
||||
void ClearL2Cache();
|
||||
void ClearThreadLocalCaches();
|
||||
|
||||
uintptr_t GetL1Pointer() const {
|
||||
return L1Pointer;
|
||||
@@ -169,7 +250,9 @@ public:
|
||||
// Some care is taken so that L1 lookups can be done without locks, and even tearing is unlikely to lead to a crash.
|
||||
// This approach has not been fully vetted yet.
|
||||
// Also note that L1 lookups might be inlined in the JIT Dispatcher and/or block ends.
|
||||
std::recursive_mutex WriteLock;
|
||||
auto AcquireLock() {
|
||||
return Shared->AcquireLock();
|
||||
}
|
||||
|
||||
private:
|
||||
void CacheBlockMapping(uint64_t Address, uintptr_t HostCode) {
|
||||
@@ -226,34 +309,6 @@ private:
|
||||
uintptr_t PageMemory;
|
||||
uintptr_t L1Pointer;
|
||||
|
||||
struct BlockLinkTag {
|
||||
uint64_t GuestDestination;
|
||||
FEXCore::Context::ExitFunctionLinkData* HostLink;
|
||||
|
||||
bool operator<(const BlockLinkTag& other) const {
|
||||
if (GuestDestination < other.GuestDestination) {
|
||||
return true;
|
||||
} else if (GuestDestination == other.GuestDestination) {
|
||||
return HostLink < other.HostLink;
|
||||
} else {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
// Use a monotonic buffer resource to allocate both the std::pmr::map and its members.
|
||||
// This allows us to quickly clear the block link map by clearing the monotonic allocator.
|
||||
// If we had allocated the block link map without the MBR, then clearing the map would require slowly
|
||||
// walking each block member and destructing objects.
|
||||
//
|
||||
// This makes `BlockLinks` look like a raw pointer that could memory leak, but since it is backed by the MBR, it won't.
|
||||
std::pmr::monotonic_buffer_resource BlockLinks_mbr;
|
||||
using BlockLinksMapType = std::pmr::map<BlockLinkTag, FEXCore::Context::BlockDelinkerFunc>;
|
||||
fextl::unique_ptr<std::pmr::polymorphic_allocator<std::byte>> BlockLinks_pma;
|
||||
BlockLinksMapType* BlockLinks;
|
||||
|
||||
fextl::robin_map<uint64_t, uint64_t> BlockList;
|
||||
|
||||
size_t TotalCacheSize;
|
||||
|
||||
constexpr static size_t CODE_SIZE = 128 * 1024 * 1024;
|
||||
|
||||
@@ -355,68 +355,29 @@ void OpDispatchBuilder::SALCOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PUSHOp(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Push(Size, Src);
|
||||
FlushRegisterCache();
|
||||
Push(Size, LoadSource(GPRClass, Op, Op->Src[0], Op->Flags));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PUSHREGOp(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
Ref Src = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
Push(Size, Src);
|
||||
FlushRegisterCache();
|
||||
Push(Size, LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PUSHAOp(OpcodeArgs) {
|
||||
// 32bit only
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
auto OldSP = LoadGPRRegister(X86State::REG_RSP);
|
||||
Ref OldSP = _Copy(LoadGPRRegister(X86State::REG_RSP));
|
||||
|
||||
// PUSHA order:
|
||||
// Tmp = SP
|
||||
// push EAX
|
||||
// push ECX
|
||||
// push EDX
|
||||
// push EBX
|
||||
// push Tmp
|
||||
// push EBP
|
||||
// push ESI
|
||||
// push EDI
|
||||
|
||||
Ref Src {};
|
||||
Ref NewSP = OldSP;
|
||||
const auto GPRSize = CTX->GetGPROpSize();
|
||||
|
||||
Src = LoadGPRRegister(X86State::REG_RAX);
|
||||
NewSP = _Push(GPRSize, Size, Src, NewSP);
|
||||
|
||||
Src = LoadGPRRegister(X86State::REG_RCX);
|
||||
NewSP = _Push(GPRSize, Size, Src, NewSP);
|
||||
|
||||
Src = LoadGPRRegister(X86State::REG_RDX);
|
||||
NewSP = _Push(GPRSize, Size, Src, NewSP);
|
||||
|
||||
Src = LoadGPRRegister(X86State::REG_RBX);
|
||||
NewSP = _Push(GPRSize, Size, Src, NewSP);
|
||||
|
||||
// Push old-sp
|
||||
NewSP = _Push(GPRSize, Size, OldSP, NewSP);
|
||||
|
||||
Src = LoadGPRRegister(X86State::REG_RBP);
|
||||
NewSP = _Push(GPRSize, Size, Src, NewSP);
|
||||
|
||||
Src = LoadGPRRegister(X86State::REG_RSI);
|
||||
NewSP = _Push(GPRSize, Size, Src, NewSP);
|
||||
|
||||
Src = LoadGPRRegister(X86State::REG_RDI);
|
||||
NewSP = _Push(GPRSize, Size, Src, NewSP);
|
||||
|
||||
// Store the new stack pointer
|
||||
StoreGPRRegister(X86State::REG_RSP, NewSP, OpSize::i32Bit);
|
||||
FlushRegisterCache();
|
||||
Push(Size, LoadGPRRegister(X86State::REG_RAX));
|
||||
Push(Size, LoadGPRRegister(X86State::REG_RCX));
|
||||
Push(Size, LoadGPRRegister(X86State::REG_RDX));
|
||||
Push(Size, LoadGPRRegister(X86State::REG_RBX));
|
||||
Push(Size, OldSP);
|
||||
Push(Size, LoadGPRRegister(X86State::REG_RBP));
|
||||
Push(Size, LoadGPRRegister(X86State::REG_RSI));
|
||||
Push(Size, LoadGPRRegister(X86State::REG_RDI));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PUSHSegmentOp(OpcodeArgs, uint32_t SegmentReg) {
|
||||
@@ -1083,9 +1044,9 @@ void OpDispatchBuilder::CMPOp(OpcodeArgs, uint32_t SrcIndex) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CQOOp(OpcodeArgs) {
|
||||
Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
auto Size = GetSrcSize(Op);
|
||||
Ref Upper = _Sbfe(OpSize::i64Bit, 1, Size * 8 - 1, Src);
|
||||
Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Size = OpSizeFromSrc(Op);
|
||||
Ref Upper = _Sbfe(std::max(OpSize::i32Bit, Size), 1, GetSrcBitSize(Op) - 1, Src);
|
||||
|
||||
StoreResult(GPRClass, Op, Upper, OpSize::iInvalid);
|
||||
}
|
||||
@@ -1185,10 +1146,11 @@ void OpDispatchBuilder::FLAGControlOp(OpcodeArgs) {
|
||||
SetCFInverted(_Constant(0));
|
||||
break;
|
||||
case 0xFC: // CLD
|
||||
SetRFLAG(_Constant(0), FEXCore::X86State::RFLAG_DF_RAW_LOC);
|
||||
// Transformed
|
||||
StoreDF(_Constant(1));
|
||||
break;
|
||||
case 0xFD: // STD
|
||||
SetRFLAG(_Constant(1), FEXCore::X86State::RFLAG_DF_RAW_LOC);
|
||||
StoreDF(_Constant(-1));
|
||||
break;
|
||||
}
|
||||
}
|
||||
@@ -1349,7 +1311,6 @@ void OpDispatchBuilder::CPUIDOp(OpcodeArgs) {
|
||||
Ref RCX = _AllocateGPR(false);
|
||||
Ref RDX = _AllocateGPR(false);
|
||||
|
||||
_Fence({FEXCore::IR::Fence_Inst});
|
||||
_CPUID(Src, Leaf, RAX, RBX, RCX, RDX);
|
||||
|
||||
StoreGPRRegister(X86State::REG_RAX, RAX);
|
||||
@@ -1481,10 +1442,10 @@ void OpDispatchBuilder::SHLDImmediateOp(OpcodeArgs) {
|
||||
Ref Res {};
|
||||
if (Size < 32) {
|
||||
Ref ShiftLeft = _Constant(Shift);
|
||||
auto ShiftRight = _Constant(Size - Shift);
|
||||
auto ShiftRight = Size - Shift;
|
||||
|
||||
auto Tmp1 = _Lshl(OpSize::i64Bit, Dest, ShiftLeft);
|
||||
auto Tmp2 = _Lshr(OpSize::i32Bit, Src, ShiftRight);
|
||||
Ref Tmp2 = ShiftRight ? _Lshr(OpSize::i32Bit, Src, _Constant(ShiftRight)) : Src;
|
||||
|
||||
Res = _Or(OpSize::i64Bit, Tmp1, Tmp2);
|
||||
} else {
|
||||
@@ -1606,15 +1567,14 @@ void OpDispatchBuilder::RotateOp(OpcodeArgs, bool Left, bool IsImmediate, bool I
|
||||
// things tighter for 8-bit later in the function.
|
||||
uint64_t Mask = Size == 8 ? 7 : (Size == 64 ? 0x3F : 0x1F);
|
||||
|
||||
Ref Src, UnmaskedSrc;
|
||||
ArithRef UnmaskedSrc;
|
||||
if (Is1Bit || IsImmediate) {
|
||||
UnmaskedConst = LoadConstantShift(Op, Is1Bit);
|
||||
UnmaskedSrc = _Constant(UnmaskedConst);
|
||||
Src = _Constant(UnmaskedConst & Mask);
|
||||
UnmaskedSrc = ARef(UnmaskedConst);
|
||||
} else {
|
||||
UnmaskedSrc = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
Src = _And(OpSize::i64Bit, UnmaskedSrc, _InlineConstant(Mask));
|
||||
UnmaskedSrc = ARef(LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}));
|
||||
}
|
||||
auto Src = UnmaskedSrc.And(Mask);
|
||||
|
||||
// We fill the upper bits so we allow garbage on load.
|
||||
auto Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true});
|
||||
@@ -1626,18 +1586,18 @@ void OpDispatchBuilder::RotateOp(OpcodeArgs, bool Left, bool IsImmediate, bool I
|
||||
}
|
||||
|
||||
// To rotate 64-bits left, right-rotate by (64 - Shift) = -Shift mod 64.
|
||||
auto Res = _Ror(OpSize, Dest, Left ? _Neg(OpSize, Src) : Src);
|
||||
auto Res = _Ror(OpSize, Dest, (Left ? Src.Neg() : Src).Ref());
|
||||
StoreResult(GPRClass, Op, Res, OpSize::iInvalid);
|
||||
|
||||
if (Is1Bit || IsImmediate) {
|
||||
if (UnmaskedConst) {
|
||||
if (UnmaskedSrc.C) {
|
||||
// Extract the last bit shifted in to CF
|
||||
SetCFDirect(Res, Left ? 0 : Size - 1, true);
|
||||
|
||||
// For ROR, OF is the XOR of the new CF bit and the most significant bit of the result.
|
||||
// For ROL, OF is the LSB and MSB XOR'd together.
|
||||
// OF is architecturally only defined for 1-bit rotate.
|
||||
if (UnmaskedConst == 1) {
|
||||
if (UnmaskedSrc.C == 1) {
|
||||
auto NewOF = _XorShift(OpSize, Res, Res, ShiftType::LSR, Left ? Size - 1 : 1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, Left ? 0 : Size - 2, true);
|
||||
}
|
||||
@@ -1648,10 +1608,10 @@ void OpDispatchBuilder::RotateOp(OpcodeArgs, bool Left, bool IsImmediate, bool I
|
||||
|
||||
// We deferred the masking for 8-bit to the flag section, do it here.
|
||||
if (Size == 8) {
|
||||
Src = _And(OpSize::i64Bit, UnmaskedSrc, _InlineConstant(0x1F));
|
||||
Src = UnmaskedSrc.And(0x1F);
|
||||
}
|
||||
|
||||
_RotateFlags(OpSizeFromSrc(Op), Res, Src, Left);
|
||||
_RotateFlags(OpSizeFromSrc(Op), Res, Src.Ref(), Left);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1662,7 +1622,7 @@ void OpDispatchBuilder::ANDNBMIOp(OpcodeArgs) {
|
||||
auto Dest = _Andn(OpSizeFromSrc(Op), Src2, Src1);
|
||||
|
||||
StoreResult(GPRClass, Op, Dest, OpSize::iInvalid);
|
||||
CalculateFlags_Logical(OpSizeFromSrc(Op), Dest, Src1, Src2);
|
||||
CalculateFlags_Logical(OpSizeFromSrc(Op), Dest);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::BEXTRBMIOp(OpcodeArgs) {
|
||||
@@ -2109,15 +2069,15 @@ void OpDispatchBuilder::RCRSmallerOp(OpcodeArgs) {
|
||||
const auto Size = GetSrcBitSize(Op);
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op
|
||||
Ref Src = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
Src = AndConst(OpSize::i32Bit, Src, 0x1F);
|
||||
auto Src = ARef(LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}));
|
||||
Src = Src.And(0x1F);
|
||||
|
||||
// CF only changes if we actually shifted. OF undefined if we didn't shift.
|
||||
// The result is unchanged if we didn't shift. So branch over the whole thing.
|
||||
Calculate_ShiftVariable(Op, Src, [this, Op, Size]() {
|
||||
Calculate_ShiftVariable(Op, Src.Ref(), [this, Op, Size]() {
|
||||
// Rematerialized to avoid crossblock liveness
|
||||
Ref Src = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
Src = AndConst(OpSize::i32Bit, Src, 0x1F);
|
||||
auto Src = ARef(LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}));
|
||||
Src = Src.And(0x1F);
|
||||
|
||||
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
|
||||
@@ -2182,27 +2142,23 @@ void OpDispatchBuilder::RCRSmallerOp(OpcodeArgs) {
|
||||
// Entire bitfield has been setup. Just extract the 8 or 16bits we need.
|
||||
// 64-bit shift used because we want to rotate in our cascaded upper bits
|
||||
// rather than zeroes.
|
||||
Ref Res = _Lshr(OpSize::i64Bit, Tmp, Src);
|
||||
Ref Res = _Lshr(OpSize::i64Bit, Tmp, Src.Ref());
|
||||
|
||||
StoreResult(GPRClass, Op, Res, OpSize::iInvalid);
|
||||
|
||||
uint64_t SrcConst = 0;
|
||||
bool IsSrcConst = IsValueConstant(WrapNode(Src), &SrcConst);
|
||||
SrcConst &= 0x1f;
|
||||
|
||||
// Our new CF will be bit (Shift - 1) of the source. 32-bit Lshr masks the
|
||||
// same as x86, but if we constant fold we must mask ourselves.
|
||||
if (IsSrcConst) {
|
||||
SetCFDirect(Tmp, SrcConst - 1, true);
|
||||
if (Src.IsConstant) {
|
||||
SetCFDirect(Tmp, (Src.C & 0x1f) - 1, true);
|
||||
} else {
|
||||
auto One = _Constant(OpSizeFromSrc(Op), 1);
|
||||
auto NewCF = _Lshr(OpSize::i32Bit, Tmp, _Sub(OpSize::i32Bit, Src, One));
|
||||
auto NewCF = _Lshr(OpSize::i32Bit, Tmp, _Sub(OpSize::i32Bit, Src.Ref(), One));
|
||||
SetCFDirect(NewCF, 0, true);
|
||||
}
|
||||
|
||||
// OF is the top two MSBs XOR'd together
|
||||
// Only when Shift == 1, it is undefined otherwise
|
||||
if (!IsSrcConst || SrcConst == 1) {
|
||||
if (!Src.IsConstant || Src.C == 1) {
|
||||
auto NewOF = _XorShift(OpSize::i32Bit, Res, Res, ShiftType::LSR, 1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, Size - 2, true);
|
||||
}
|
||||
@@ -2329,15 +2285,15 @@ void OpDispatchBuilder::RCLSmallerOp(OpcodeArgs) {
|
||||
const auto Size = GetSrcBitSize(Op);
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op
|
||||
Ref Src = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
Src = AndConst(OpSize::i32Bit, Src, 0x1F);
|
||||
auto Src = ARef(LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}));
|
||||
Src = Src.And(0x1F);
|
||||
|
||||
// CF only changes if we actually shifted. OF undefined if we didn't shift.
|
||||
// The result is unchanged if we didn't shift. So branch over the whole thing.
|
||||
Calculate_ShiftVariable(Op, Src, [this, Op, Size]() {
|
||||
Calculate_ShiftVariable(Op, Src.Ref(), [this, Op, Size]() {
|
||||
// Rematerialized to avoid crossblock liveness
|
||||
Ref Src = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true});
|
||||
Src = AndConst(OpSize::i32Bit, Src, 0x1F);
|
||||
auto Src = ARef(LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}));
|
||||
Src = Src.And(0x1F);
|
||||
Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags);
|
||||
|
||||
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
@@ -2360,20 +2316,19 @@ void OpDispatchBuilder::RCLSmallerOp(OpcodeArgs) {
|
||||
// Shift 1 more bit that expected to get our result
|
||||
// Shifting to the right will now behave like a rotate to the left
|
||||
// Which we emulate with a _Ror
|
||||
Ref Res = _Ror(OpSize::i64Bit, Tmp, _Neg(OpSize::i32Bit, Src));
|
||||
Ref Res = _Ror(OpSize::i64Bit, Tmp, Src.Neg().Ref());
|
||||
|
||||
StoreResult(GPRClass, Op, Res, OpSize::iInvalid);
|
||||
|
||||
// Our new CF is now at the bit position that we are shifting
|
||||
// Either 0 if CF hasn't changed (CF is living in bit 0)
|
||||
// or higher
|
||||
auto NewCF = _Ror(OpSize::i64Bit, Tmp, _Sub(OpSize::i64Bit, _Constant(63), Src));
|
||||
auto NewCF = _Ror(OpSize::i64Bit, Tmp, Src.Presub(63).Ref());
|
||||
SetCFDirect(NewCF, 0, true);
|
||||
|
||||
// OF is the XOR of the NewCF and the MSB of the result
|
||||
// Only defined for 1-bit rotates.
|
||||
uint64_t SrcConst;
|
||||
if (!IsValueConstant(WrapNode(Src), &SrcConst) || SrcConst == 1) {
|
||||
if (!Src.IsConstant || Src.C == 1) {
|
||||
auto NewOF = _XorShift(OpSize::i64Bit, NewCF, Res, ShiftType::LSR, Size - 1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(NewOF, 0, true);
|
||||
}
|
||||
@@ -2382,21 +2337,19 @@ void OpDispatchBuilder::RCLSmallerOp(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
Ref Value;
|
||||
Ref Src {};
|
||||
ArithRef Src;
|
||||
bool IsNonconstant = Op->Src[SrcIndex].IsGPR();
|
||||
uint8_t ConstantShift = 0;
|
||||
|
||||
const uint32_t Size = GetDstBitSize(Op);
|
||||
const uint32_t Mask = Size - 1;
|
||||
|
||||
if (IsNonconstant) {
|
||||
// Because we mask explicitly with And/Bfe/Sbfe after, we can allow garbage here.
|
||||
Src = LoadSource(GPRClass, Op, Op->Src[SrcIndex], Op->Flags, {.AllowUpperGarbage = true});
|
||||
Src = ARef(LoadSource(GPRClass, Op, Op->Src[SrcIndex], Op->Flags, {.AllowUpperGarbage = true}));
|
||||
} else {
|
||||
// Can only be an immediate
|
||||
// Masked by operand size
|
||||
ConstantShift = Op->Src[SrcIndex].Data.Literal.Value & Mask;
|
||||
Src = _Constant(ConstantShift);
|
||||
Src = ARef(Op->Src[SrcIndex].Data.Literal.Value & Mask);
|
||||
}
|
||||
|
||||
if (Op->Dest.IsGPR()) {
|
||||
@@ -2408,7 +2361,8 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
// Get the bit selection from the src. We need to mask for 8/16-bit, but
|
||||
// rely on the implicit masking of Lshr for native sizes.
|
||||
unsigned LshrSize = std::max<uint8_t>(IR::OpSizeToSize(OpSize::i32Bit), Size / 8);
|
||||
auto BitSelect = (Size == (LshrSize * 8)) ? Src : _And(OpSize::i64Bit, Src, _Constant(Mask));
|
||||
auto BitSelect = (Size == (LshrSize * 8)) ? Src : Src.And(Mask);
|
||||
auto LshrOpSize = IR::SizeToOpSize(LshrSize);
|
||||
|
||||
// OF/SF/AF/PF undefined. ZF must be preserved. We choose to preserve OF/SF
|
||||
// too since we just use an rmif to insert into CF directly. We could
|
||||
@@ -2418,10 +2372,10 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
// can reuse the invert.
|
||||
if (Action != BTAction::BTComplement) {
|
||||
if (IsNonconstant) {
|
||||
Value = _Lshr(IR::SizeToOpSize(LshrSize), Value, BitSelect);
|
||||
Value = _Lshr(IR::SizeToOpSize(LshrSize), Value, BitSelect.Ref());
|
||||
}
|
||||
|
||||
SetRFLAG(Value, X86State::RFLAG_CF_RAW_LOC, ConstantShift, true);
|
||||
SetRFLAG(Value, X86State::RFLAG_CF_RAW_LOC, Src.IsConstant ? Src.C : 0, true);
|
||||
CFInverted = false;
|
||||
}
|
||||
|
||||
@@ -2432,30 +2386,27 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
}
|
||||
|
||||
case BTAction::BTClear: {
|
||||
Ref BitMask = _Lshl(IR::SizeToOpSize(LshrSize), _Constant(1), BitSelect);
|
||||
Dest = _Andn(IR::SizeToOpSize(LshrSize), Dest, BitMask);
|
||||
Dest = _Andn(LshrOpSize, Dest, BitSelect.MaskBit(LshrOpSize).Ref());
|
||||
StoreResult(GPRClass, Op, Dest, OpSize::iInvalid);
|
||||
break;
|
||||
}
|
||||
|
||||
case BTAction::BTSet: {
|
||||
Ref BitMask = _Lshl(IR::SizeToOpSize(LshrSize), _Constant(1), BitSelect);
|
||||
Dest = _Or(IR::SizeToOpSize(LshrSize), Dest, BitMask);
|
||||
Dest = _Or(LshrOpSize, Dest, BitSelect.MaskBit(LshrOpSize).Ref());
|
||||
StoreResult(GPRClass, Op, Dest, OpSize::iInvalid);
|
||||
break;
|
||||
}
|
||||
|
||||
case BTAction::BTComplement: {
|
||||
Ref BitMask = _Lshl(IR::SizeToOpSize(LshrSize), _Constant(1), BitSelect);
|
||||
Dest = _Xor(IR::SizeToOpSize(LshrSize), Dest, BitMask);
|
||||
Dest = _Xor(LshrOpSize, Dest, BitSelect.MaskBit(LshrOpSize).Ref());
|
||||
|
||||
if (IsNonconstant) {
|
||||
Value = _Lshr(IR::SizeToOpSize(LshrSize), Dest, BitSelect);
|
||||
Value = _Lshr(LshrOpSize, Dest, BitSelect.Ref());
|
||||
} else {
|
||||
Value = Dest;
|
||||
}
|
||||
|
||||
SetRFLAG(Value, X86State::RFLAG_CF_RAW_LOC, ConstantShift, true);
|
||||
SetRFLAG(Value, X86State::RFLAG_CF_RAW_LOC, Src.IsConstant ? Src.C : 0, true);
|
||||
CFInverted = true;
|
||||
|
||||
StoreResult(GPRClass, Op, Dest, OpSize::iInvalid);
|
||||
@@ -2466,17 +2417,15 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
// Load the address to the memory location
|
||||
Ref Dest = MakeSegmentAddress(Op, Op->Dest);
|
||||
// Get the bit selection from the src
|
||||
Ref BitSelect = _Bfe(std::max(OpSize::i32Bit, GetOpSize(Src)), 3, 0, Src);
|
||||
auto BitSelect = Src.Bfe(0, 3);
|
||||
|
||||
// Address is provided as bits we want BYTE offsets
|
||||
// Extract Signed offset
|
||||
Src = _Sbfe(OpSize::i64Bit, Size - 3, 3, Src);
|
||||
Src = Src.Sbfe(3, Size - 3);
|
||||
|
||||
// Get the address offset by shifting out the size of the op (To shift out the bit selection)
|
||||
// Then use that to index in to the memory location by size of op
|
||||
AddressMode Address = {.Base = Dest, .Index = Src, .AddrSize = OpSize::i64Bit};
|
||||
|
||||
ConstantShift = 0;
|
||||
AddressMode Address = {.Base = Dest, .Index = Src.Ref(), .AddrSize = OpSize::i64Bit};
|
||||
|
||||
switch (Action) {
|
||||
case BTAction::BTNone: {
|
||||
@@ -2485,7 +2434,7 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
}
|
||||
|
||||
case BTAction::BTClear: {
|
||||
Ref BitMask = _Lshl(OpSize::i64Bit, _Constant(1), BitSelect);
|
||||
Ref BitMask = BitSelect.MaskBit(OpSize::i64Bit).Ref();
|
||||
|
||||
if (DestIsLockedMem(Op)) {
|
||||
HandledLock = true;
|
||||
@@ -2500,7 +2449,7 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
}
|
||||
|
||||
case BTAction::BTSet: {
|
||||
Ref BitMask = _Lshl(OpSize::i64Bit, _Constant(1), BitSelect);
|
||||
Ref BitMask = BitSelect.MaskBit(OpSize::i64Bit).Ref();
|
||||
|
||||
if (DestIsLockedMem(Op)) {
|
||||
HandledLock = true;
|
||||
@@ -2515,7 +2464,7 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
}
|
||||
|
||||
case BTAction::BTComplement: {
|
||||
Ref BitMask = _Lshl(OpSize::i64Bit, _Constant(1), BitSelect);
|
||||
Ref BitMask = BitSelect.MaskBit(OpSize::i64Bit).Ref();
|
||||
|
||||
if (DestIsLockedMem(Op)) {
|
||||
HandledLock = true;
|
||||
@@ -2531,10 +2480,12 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
}
|
||||
|
||||
// Now shift in to the correct bit location
|
||||
Value = _Lshr(std::max(OpSize::i32Bit, GetOpSize(Value)), Value, BitSelect);
|
||||
if (!BitSelect.IsDefinitelyZero()) {
|
||||
Value = _Lshr(std::max(OpSize::i32Bit, GetOpSize(Value)), Value, BitSelect.Ref());
|
||||
}
|
||||
|
||||
// OF/SF/ZF/AF/PF undefined.
|
||||
SetCFDirect(Value, ConstantShift, true);
|
||||
SetCFDirect(Value, 0, true);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2591,7 +2542,7 @@ void OpDispatchBuilder::IMUL2SrcOp(OpcodeArgs) {
|
||||
case OpSize::i8Bit:
|
||||
case OpSize::i16Bit: {
|
||||
Src1 = _Sbfe(OpSize::i64Bit, SizeBits, 0, Src1);
|
||||
Src2 = _Sbfe(OpSize::i64Bit, SizeBits, 0, Src2);
|
||||
Src2 = ARef(Src2).Sbfe(0, SizeBits).Ref();
|
||||
Dest = _Mul(OpSize::i64Bit, Src1, Src2);
|
||||
ResultHigh = _Sbfe(OpSize::i64Bit, SizeBits, SizeBits, Dest);
|
||||
break;
|
||||
@@ -2900,9 +2851,10 @@ void OpDispatchBuilder::AASOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AAMOp(OpcodeArgs) {
|
||||
auto AL = LoadGPRRegister(X86State::REG_RAX, OpSize::i8Bit);
|
||||
auto Imm8 = _Constant(Op->Src[0].Data.Literal.Value & 0xFF);
|
||||
auto UDivOp = _UDiv(OpSize::i64Bit, AL, Imm8);
|
||||
auto URemOp = _URem(OpSize::i64Bit, AL, Imm8);
|
||||
auto Res = _AddShift(OpSize::i64Bit, URemOp, UDivOp, ShiftType::LSL, 8);
|
||||
Ref Quotient = _AllocateGPR(true);
|
||||
Ref Remainder = _AllocateGPR(true);
|
||||
_UDiv(OpSize::i64Bit, AL, Invalid(), Imm8, Quotient, Remainder);
|
||||
auto Res = _AddShift(OpSize::i64Bit, Remainder, Quotient, ShiftType::LSL, 8);
|
||||
StoreGPRRegister(X86State::REG_RAX, Res, OpSize::i16Bit);
|
||||
|
||||
SetNZ_ZeroCV(OpSize::i8Bit, Res);
|
||||
@@ -3594,8 +3546,7 @@ void OpDispatchBuilder::BSWAPOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PUSHFOp(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
Ref Src = GetPackedRFLAG();
|
||||
Push(Size, Src);
|
||||
Push(Size, GetPackedRFLAG());
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::POPFOp(OpcodeArgs) {
|
||||
@@ -3630,52 +3581,43 @@ void OpDispatchBuilder::NEGOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::DIVOp(OpcodeArgs) {
|
||||
// This loads the divisor
|
||||
Ref Divisor = LoadSource(GPRClass, Op, Op->Dest, Op->Flags);
|
||||
|
||||
const auto GPRSize = CTX->GetGPROpSize();
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
// This loads the divisor. 32-bit/64-bit paths mask inside the JIT, 8/16 do not.
|
||||
Ref Divisor = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = Size >= OpSize::i32Bit});
|
||||
|
||||
if (Size == OpSize::i64Bit && !CTX->Config.Is64BitMode) {
|
||||
LogMan::Msg::EFmt("Doesn't exist in 32bit mode");
|
||||
DecodeFailure = true;
|
||||
return;
|
||||
}
|
||||
|
||||
Ref Quotient = _AllocateGPR(true);
|
||||
Ref Remainder = _AllocateGPR(true);
|
||||
|
||||
if (Size == OpSize::i8Bit) {
|
||||
Ref Src1 = LoadGPRRegister(X86State::REG_RAX, OpSize::i16Bit);
|
||||
|
||||
auto UDivOp = _UDiv(OpSize::i16Bit, Src1, Divisor);
|
||||
auto URemOp = _URem(OpSize::i16Bit, Src1, Divisor);
|
||||
_UDiv(OpSize::i16Bit, Src1, Invalid(), Divisor, Quotient, Remainder);
|
||||
|
||||
// AX[15:0] = concat<URem[7:0]:UDiv[7:0]>
|
||||
auto ResultAX = _Bfi(GPRSize, 8, 8, UDivOp, URemOp);
|
||||
auto ResultAX = _Bfi(GPRSize, 8, 8, Quotient, Remainder);
|
||||
StoreGPRRegister(X86State::REG_RAX, ResultAX, OpSize::i16Bit);
|
||||
} else if (Size == OpSize::i16Bit) {
|
||||
Ref Src1 = LoadGPRRegister(X86State::REG_RAX);
|
||||
Ref Src2 = LoadGPRRegister(X86State::REG_RDX);
|
||||
auto UDivOp = _LUDiv(OpSize::i16Bit, Src1, Src2, Divisor);
|
||||
auto URemOp = _LURem(OpSize::i16Bit, Src1, Src2, Divisor);
|
||||
|
||||
StoreGPRRegister(X86State::REG_RAX, UDivOp, Size);
|
||||
StoreGPRRegister(X86State::REG_RDX, URemOp, Size);
|
||||
} else if (Size == OpSize::i32Bit) {
|
||||
} else {
|
||||
Ref Src1 = LoadGPRRegister(X86State::REG_RAX);
|
||||
Ref Src2 = LoadGPRRegister(X86State::REG_RDX);
|
||||
|
||||
Ref UDivOp = _Bfe(OpSize::i32Bit, IR::OpSizeAsBits(Size), 0, _LUDiv(OpSize::i32Bit, Src1, Src2, Divisor));
|
||||
Ref URemOp = _Bfe(OpSize::i32Bit, IR::OpSizeAsBits(Size), 0, _LURem(OpSize::i32Bit, Src1, Src2, Divisor));
|
||||
_UDiv(Size, Src1, Src2, Divisor, Quotient, Remainder);
|
||||
|
||||
StoreGPRRegister(X86State::REG_RAX, UDivOp);
|
||||
StoreGPRRegister(X86State::REG_RDX, URemOp);
|
||||
} else if (Size == OpSize::i64Bit) {
|
||||
if (!CTX->Config.Is64BitMode) {
|
||||
LogMan::Msg::EFmt("Doesn't exist in 32bit mode");
|
||||
DecodeFailure = true;
|
||||
return;
|
||||
if (Size == OpSize::i32Bit) {
|
||||
Quotient = _Bfe(OpSize::i32Bit, IR::OpSizeAsBits(Size), 0, Quotient);
|
||||
Remainder = _Bfe(OpSize::i32Bit, IR::OpSizeAsBits(Size), 0, Remainder);
|
||||
Size = OpSize::iInvalid;
|
||||
}
|
||||
Ref Src1 = LoadGPRRegister(X86State::REG_RAX);
|
||||
Ref Src2 = LoadGPRRegister(X86State::REG_RDX);
|
||||
|
||||
auto UDivOp = _LUDiv(OpSize::i64Bit, Src1, Src2, Divisor);
|
||||
auto URemOp = _LURem(OpSize::i64Bit, Src1, Src2, Divisor);
|
||||
|
||||
StoreGPRRegister(X86State::REG_RAX, UDivOp);
|
||||
StoreGPRRegister(X86State::REG_RDX, URemOp);
|
||||
StoreGPRRegister(X86State::REG_RAX, Quotient, Size);
|
||||
StoreGPRRegister(X86State::REG_RDX, Remainder, Size);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3684,50 +3626,41 @@ void OpDispatchBuilder::IDIVOp(OpcodeArgs) {
|
||||
Ref Divisor = LoadSource(GPRClass, Op, Op->Dest, Op->Flags);
|
||||
|
||||
const auto GPRSize = CTX->GetGPROpSize();
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
if (Size == OpSize::i64Bit && !CTX->Config.Is64BitMode) {
|
||||
LogMan::Msg::EFmt("Doesn't exist in 32bit mode");
|
||||
DecodeFailure = true;
|
||||
return;
|
||||
}
|
||||
|
||||
Ref Quotient = _AllocateGPR(true);
|
||||
Ref Remainder = _AllocateGPR(true);
|
||||
|
||||
if (Size == OpSize::i8Bit) {
|
||||
Ref Src1 = LoadGPRRegister(X86State::REG_RAX);
|
||||
Src1 = _Sbfe(OpSize::i64Bit, 16, 0, Src1);
|
||||
Divisor = _Sbfe(OpSize::i64Bit, 8, 0, Divisor);
|
||||
|
||||
auto UDivOp = _Div(OpSize::i64Bit, Src1, Divisor);
|
||||
auto URemOp = _Rem(OpSize::i64Bit, Src1, Divisor);
|
||||
_Div(OpSize::i64Bit, Src1, Invalid(), Divisor, Quotient, Remainder);
|
||||
|
||||
// AX[15:0] = concat<URem[7:0]:UDiv[7:0]>
|
||||
auto ResultAX = _Bfi(GPRSize, 8, 8, UDivOp, URemOp);
|
||||
auto ResultAX = _Bfi(GPRSize, 8, 8, Quotient, Remainder);
|
||||
StoreGPRRegister(X86State::REG_RAX, ResultAX, OpSize::i16Bit);
|
||||
} else if (Size == OpSize::i16Bit) {
|
||||
Ref Src1 = LoadGPRRegister(X86State::REG_RAX);
|
||||
Ref Src2 = LoadGPRRegister(X86State::REG_RDX);
|
||||
auto UDivOp = _LDiv(OpSize::i16Bit, Src1, Src2, Divisor);
|
||||
auto URemOp = _LRem(OpSize::i16Bit, Src1, Src2, Divisor);
|
||||
|
||||
StoreGPRRegister(X86State::REG_RAX, UDivOp, Size);
|
||||
StoreGPRRegister(X86State::REG_RDX, URemOp, Size);
|
||||
} else if (Size == OpSize::i32Bit) {
|
||||
} else {
|
||||
Ref Src1 = LoadGPRRegister(X86State::REG_RAX);
|
||||
Ref Src2 = LoadGPRRegister(X86State::REG_RDX);
|
||||
|
||||
Ref UDivOp = _Bfe(OpSize::i32Bit, IR::OpSizeAsBits(Size), 0, _LDiv(OpSize::i32Bit, Src1, Src2, Divisor));
|
||||
Ref URemOp = _Bfe(OpSize::i32Bit, IR::OpSizeAsBits(Size), 0, _LRem(OpSize::i32Bit, Src1, Src2, Divisor));
|
||||
_Div(Size, Src1, Src2, Divisor, Quotient, Remainder);
|
||||
|
||||
StoreGPRRegister(X86State::REG_RAX, UDivOp);
|
||||
StoreGPRRegister(X86State::REG_RDX, URemOp);
|
||||
} else if (Size == OpSize::i64Bit) {
|
||||
if (!CTX->Config.Is64BitMode) {
|
||||
LogMan::Msg::EFmt("Doesn't exist in 32bit mode");
|
||||
DecodeFailure = true;
|
||||
return;
|
||||
if (Size == OpSize::i32Bit) {
|
||||
Quotient = _Bfe(OpSize::i32Bit, IR::OpSizeAsBits(Size), 0, Quotient);
|
||||
Remainder = _Bfe(OpSize::i32Bit, IR::OpSizeAsBits(Size), 0, Remainder);
|
||||
Size = OpSize::iInvalid;
|
||||
}
|
||||
Ref Src1 = LoadGPRRegister(X86State::REG_RAX);
|
||||
Ref Src2 = LoadGPRRegister(X86State::REG_RDX);
|
||||
|
||||
auto UDivOp = _LDiv(OpSize::i64Bit, Src1, Src2, Divisor);
|
||||
auto URemOp = _LRem(OpSize::i64Bit, Src1, Src2, Divisor);
|
||||
|
||||
StoreGPRRegister(X86State::REG_RAX, UDivOp);
|
||||
StoreGPRRegister(X86State::REG_RDX, URemOp);
|
||||
StoreGPRRegister(X86State::REG_RAX, Quotient, Size);
|
||||
StoreGPRRegister(X86State::REG_RDX, Remainder, Size);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3933,7 +3866,7 @@ void OpDispatchBuilder::CreateJumpBlocks(const fextl::vector<FEXCore::Frontend::
|
||||
|
||||
void OpDispatchBuilder::BeginFunction(uint64_t RIP, const fextl::vector<FEXCore::Frontend::Decoder::DecodedBlocks>* Blocks, uint32_t NumInstructions) {
|
||||
Entry = RIP;
|
||||
auto IRHeader = _IRHeader(InvalidNode, RIP, 0, NumInstructions);
|
||||
auto IRHeader = _IRHeader(InvalidNode, RIP, 0, NumInstructions, 0, 0);
|
||||
CreateJumpBlocks(Blocks);
|
||||
|
||||
auto Block = GetNewJumpBlock(RIP);
|
||||
@@ -4299,7 +4232,7 @@ void OpDispatchBuilder::StoreGPRRegister(uint32_t GPR, const Ref Src, IR::OpSize
|
||||
Ref Reg = Src;
|
||||
if (Size != GPRSize || Offset != 0) {
|
||||
// Need to do an insert if not automatic size or zero offset.
|
||||
Reg = _Bfi(GPRSize, IR::OpSizeAsBits(Size), Offset, LoadGPRRegister(GPR), Src);
|
||||
Reg = ARef(Reg).BfiInto(LoadGPRRegister(GPR), Offset, IR::OpSizeAsBits(Size));
|
||||
}
|
||||
|
||||
StoreRegister(GPR, false, Reg);
|
||||
@@ -4358,7 +4291,7 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
if (GPRSize == OpSize::i64Bit && OpSize == OpSize::i32Bit) {
|
||||
// If the Source IR op is 64 bits, we need to zext the upper bits
|
||||
// For all other sizes, the upper bits are guaranteed to already be zero
|
||||
Ref Value = GetOpSize(Src) == OpSize::i64Bit ? _Bfe(OpSize::i32Bit, 32, 0, Src) : Src;
|
||||
Ref Value = GetOpSize(Src) == OpSize::i64Bit ? ARef(Src).Bfe(0, 32).Ref() : Src;
|
||||
StoreGPRRegister(gpr, Value, GPRSize);
|
||||
|
||||
LOGMAN_THROW_A_FMT(!Operand.Data.GPR.HighBits, "Can't handle 32bit store to high 8bit register");
|
||||
@@ -4456,18 +4389,23 @@ void OpDispatchBuilder::MOVGPRNTOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp, unsigned SrcIdx) {
|
||||
/* On x86, the canonical way to zero a register is XOR with itself... because
|
||||
* modern x86 detects this pattern in hardware. arm64 does not detect this
|
||||
* pattern, we should do it like the x86 hardware would. On arm64, "mov x0,
|
||||
* #0" is faster than "eor x0, x0, x0". Additionally this lets more constant
|
||||
* folding kick in for flags.
|
||||
*/
|
||||
// On x86, the canonical way to zero a register is XOR with itself. Detect and
|
||||
// emit optimal arm64 assembly.
|
||||
if (!DestIsLockedMem(Op) && ALUIROp == FEXCore::IR::IROps::OP_XOR && Op->Dest.IsGPR() && Op->Src[SrcIdx].IsGPR() &&
|
||||
Op->Dest.Data.GPR == Op->Src[SrcIdx].Data.GPR) {
|
||||
|
||||
auto Result = _Constant(0);
|
||||
StoreResult(GPRClass, Op, Result, OpSize::iInvalid);
|
||||
CalculateFlags_Logical(OpSizeFromSrc(Op), Result, Result, Result);
|
||||
// Set flags for zero result with inverted carry. We subtract an arbitrary
|
||||
// register from itself to get the zero, since `subs wzr, #0` is not
|
||||
// encodable. This is optimal and works regardless of the opsize.
|
||||
auto Zero = LoadGPR(Op->Dest.Data.GPR.GPR);
|
||||
HandleNZ00Write();
|
||||
InvalidateAF();
|
||||
CalculatePF(_SubWithFlags(OpSize::i32Bit, Zero, Zero));
|
||||
CFInverted = true;
|
||||
FlushRegisterCache();
|
||||
|
||||
// Move 0 into the register
|
||||
StoreResult(GPRClass, Op, _Constant(0), OpSize::iInvalid);
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -4485,7 +4423,8 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::I
|
||||
// Try to eliminate the masking after 8/16-bit operations with constants, by
|
||||
// promoting to a full size operation that preserves the upper bits.
|
||||
uint64_t Const;
|
||||
if (Size < OpSize::i32Bit && !DestIsLockedMem(Op) && Op->Dest.IsGPR() && !Op->Dest.Data.GPR.HighBits && IsValueConstant(WrapNode(Src), &Const) &&
|
||||
bool IsConst = IsValueConstant(WrapNode(Src), &Const);
|
||||
if (Size < OpSize::i32Bit && !DestIsLockedMem(Op) && Op->Dest.IsGPR() && !Op->Dest.Data.GPR.HighBits && IsConst &&
|
||||
(ALUIROp == IR::IROps::OP_XOR || ALUIROp == IR::IROps::OP_OR || ALUIROp == IR::IROps::OP_ANDWITHFLAGS)) {
|
||||
|
||||
RoundedSize = ResultSize = CTX->GetGPROpSize();
|
||||
@@ -4519,8 +4458,15 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::I
|
||||
}
|
||||
|
||||
const auto OpSize = RoundedSize;
|
||||
DeriveOp(ALUOp, ALUIROp, _AndWithFlags(OpSize, Dest, Src));
|
||||
Result = ALUOp;
|
||||
uint64_t Mask = Size == OpSize::i64Bit ? ~0ull : ((1ull << IR::OpSizeAsBits(Size)) - 1);
|
||||
if (IsConst && Const == Mask && !DestIsLockedMem(Op) && ALUIROp == IR::IROps::OP_XOR && Size >= OpSize::i32Bit) {
|
||||
Result = _Not(OpSize, Dest);
|
||||
} else if (IsConst && Const == Mask && !DestIsLockedMem(Op) && ALUIROp == IR::IROps::OP_AND) {
|
||||
Result = Dest;
|
||||
} else {
|
||||
DeriveOp(ALUOp, ALUIROp, _AndWithFlags(OpSize, Dest, Src));
|
||||
Result = ALUOp;
|
||||
}
|
||||
|
||||
// Flags set
|
||||
switch (ALUIROp) {
|
||||
@@ -4529,7 +4475,7 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::I
|
||||
case FEXCore::IR::IROps::OP_XOR:
|
||||
case FEXCore::IR::IROps::OP_AND:
|
||||
case FEXCore::IR::IROps::OP_OR: {
|
||||
CalculateFlags_Logical(Size, Result, Dest, Src);
|
||||
CalculateFlags_Logical(Size, Result);
|
||||
break;
|
||||
}
|
||||
case FEXCore::IR::IROps::OP_ANDWITHFLAGS: {
|
||||
|
||||
@@ -7,6 +7,7 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/RegisterAllocationData.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
@@ -161,7 +162,7 @@ public:
|
||||
}
|
||||
IRPair<IROp_ExitFunction> ExitFunction(Ref NewRIP) {
|
||||
FlushRegisterCache();
|
||||
return _ExitFunction(NewRIP);
|
||||
return _ExitFunction(GetOpSize(NewRIP), NewRIP);
|
||||
}
|
||||
IRPair<IROp_Break> Break(BreakDefinition Reason) {
|
||||
FlushRegisterCache();
|
||||
@@ -1208,13 +1209,15 @@ public:
|
||||
Ref Value = RegCache.Value[Index];
|
||||
|
||||
if (Index >= GPR0Index && Index <= GPR15Index) {
|
||||
_StoreRegister(Value, Index - GPR0Index, GPRClass, GPRSize);
|
||||
Ref R = _StoreRegister(Value, GPRSize);
|
||||
R->Reg = PhysicalRegister(GPRFixedClass, Index - GPR0Index).Raw;
|
||||
} else if (Index == PFIndex) {
|
||||
_StorePF(Value, GPRSize);
|
||||
} else if (Index == AFIndex) {
|
||||
_StoreAF(Value, GPRSize);
|
||||
} else if (Index >= FPR0Index && Index <= FPR15Index) {
|
||||
_StoreRegister(Value, Index - FPR0Index, FPRClass, VectorSize);
|
||||
Ref R = _StoreRegister(Value, VectorSize);
|
||||
R->Reg = PhysicalRegister(FPRFixedClass, Index - FPR0Index).Raw;
|
||||
} else if (Index == DFIndex) {
|
||||
_StoreContext(OpSize::i8Bit, GPRClass, Value, offsetof(Core::CPUState, flags[X86State::RFLAG_DF_RAW_LOC]));
|
||||
} else {
|
||||
@@ -2025,14 +2028,7 @@ private:
|
||||
|
||||
// Returns (DF ? -Size : Size)
|
||||
Ref LoadDir(const unsigned Size) {
|
||||
auto Dir = LoadDF();
|
||||
auto Shift = FEXCore::ilog2(Size);
|
||||
|
||||
if (Shift) {
|
||||
return _Lshl(CTX->GetGPROpSize(), Dir, _Constant(Shift));
|
||||
} else {
|
||||
return Dir;
|
||||
}
|
||||
return ARef(LoadDF()).Lshl(FEXCore::ilog2(Size)).Ref();
|
||||
}
|
||||
|
||||
// Returns DF ? (X - Size) : (X + Size)
|
||||
@@ -2310,7 +2306,7 @@ private:
|
||||
Ref CalculateFlags_ADD(IR::OpSize SrcSize, Ref Src1, Ref Src2, bool UpdateCF = true);
|
||||
void CalculateFlags_MUL(IR::OpSize SrcSize, Ref Res, Ref High);
|
||||
void CalculateFlags_UMUL(Ref High);
|
||||
void CalculateFlags_Logical(IR::OpSize SrcSize, Ref Res, Ref Src1, Ref Src2);
|
||||
void CalculateFlags_Logical(IR::OpSize SrcSize, Ref Res);
|
||||
void CalculateFlags_ShiftLeft(IR::OpSize SrcSize, Ref Res, Ref Src1, Ref Src2);
|
||||
void CalculateFlags_ShiftLeftImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRight(IR::OpSize SrcSize, Ref Res, Ref Src1, Ref Src2);
|
||||
@@ -2321,20 +2317,7 @@ private:
|
||||
void CalculateFlags_ZCNT(IR::OpSize SrcSize, Ref Result);
|
||||
/** @} */
|
||||
|
||||
Ref AndConst(FEXCore::IR::OpSize Size, Ref Node, uint64_t Const) {
|
||||
uint64_t NodeConst;
|
||||
|
||||
if (IsValueConstant(WrapNode(Node), &NodeConst)) {
|
||||
return _Constant(NodeConst & Const);
|
||||
} else {
|
||||
return _And(Size, Node, _Constant(Const));
|
||||
}
|
||||
}
|
||||
|
||||
/** @} */
|
||||
|
||||
Ref GetX87Top();
|
||||
Ref GetX87Tag(Ref Value, Ref AbridgedFTW);
|
||||
void SetX87FTW(Ref FTW);
|
||||
Ref GetX87FTW_Helper();
|
||||
void SetX87Top(Ref Value);
|
||||
@@ -2494,10 +2477,139 @@ private:
|
||||
return Value;
|
||||
}
|
||||
|
||||
Ref VZeroExtendOperand(OpSize Size, X86Tables::DecodedOperand Op, Ref Value) {
|
||||
bool IsMMX = Op.IsGPR() && Op.Data.GPR.GPR >= X86State::REG_MM_0;
|
||||
bool AlreadyExtended = Op.IsGPRDirect() || Op.IsGPRIndirect() || IsMMX;
|
||||
|
||||
return AlreadyExtended ? Value : _VMov(Size, Value);
|
||||
}
|
||||
|
||||
void Push(IR::OpSize Size, Ref Value) {
|
||||
auto OldSP = LoadGPRRegister(X86State::REG_RSP);
|
||||
auto NewSP = _Push(CTX->GetGPROpSize(), Size, Value, OldSP);
|
||||
StoreGPRRegister(X86State::REG_RSP, NewSP);
|
||||
FlushRegisterCache();
|
||||
}
|
||||
|
||||
struct ArithRef {
|
||||
IREmitter* E {};
|
||||
bool IsConstant {};
|
||||
union {
|
||||
Ref R {};
|
||||
uint64_t C;
|
||||
};
|
||||
|
||||
ArithRef() {}
|
||||
|
||||
ArithRef(IREmitter* IREmit, Ref Reference)
|
||||
: E(IREmit)
|
||||
, IsConstant(false)
|
||||
, R(Reference) {}
|
||||
|
||||
ArithRef(IREmitter* IREmit, uint64_t K)
|
||||
: E(IREmit)
|
||||
, IsConstant(true)
|
||||
, C(K) {}
|
||||
|
||||
ArithRef Neg() {
|
||||
return IsConstant ? ArithRef(E, -C) : ArithRef(E, E->_Neg(OpSize::i64Bit, R));
|
||||
}
|
||||
|
||||
ArithRef And(uint64_t K) {
|
||||
return IsConstant ? ArithRef(E, C & K) : ArithRef(E, E->_And(OpSize::i64Bit, R, E->_Constant(K)));
|
||||
}
|
||||
|
||||
ArithRef Presub(uint64_t K) {
|
||||
return IsConstant ? ArithRef(E, K - C) : ArithRef(E, E->_Sub(OpSize::i64Bit, E->_Constant(K), R));
|
||||
}
|
||||
|
||||
ArithRef Lshl(uint64_t Shift) {
|
||||
if (Shift == 0) {
|
||||
return *this;
|
||||
} else if (IsConstant) {
|
||||
return ArithRef(E, C << Shift);
|
||||
} else {
|
||||
return ArithRef(E, E->_Lshl(OpSize::i64Bit, R, E->_Constant(Shift)));
|
||||
}
|
||||
}
|
||||
|
||||
ArithRef Bfe(unsigned Start, unsigned Size) {
|
||||
if (IsConstant) {
|
||||
return ArithRef(E, (C >> Start) & ((1ull << Size) - 1));
|
||||
} else {
|
||||
return ArithRef(E, E->_Bfe(OpSize::i64Bit, Size, Start, R));
|
||||
}
|
||||
}
|
||||
|
||||
ArithRef Sbfe(unsigned Start, unsigned Size) {
|
||||
if (IsConstant) {
|
||||
uint64_t SourceMask = Size == 64 ? ~0ULL : ((1ULL << Size) - 1);
|
||||
SourceMask <<= Start;
|
||||
|
||||
int64_t NewConstant = (C & SourceMask) >> Start;
|
||||
NewConstant <<= 64 - Size;
|
||||
NewConstant >>= 64 - Size;
|
||||
|
||||
return ArithRef(E, NewConstant);
|
||||
} else {
|
||||
return ArithRef(E, E->_Sbfe(OpSize::i64Bit, Size, Start, R));
|
||||
}
|
||||
}
|
||||
|
||||
Ref BfiInto(Ref Bitfield, unsigned Start, unsigned Size) {
|
||||
if (IsConstant && (Size > 0 && Size < 64)) {
|
||||
uint64_t SourceMask = (1ULL << Size) - 1;
|
||||
uint64_t SourceMaskShifted = SourceMask << Start;
|
||||
|
||||
if (C == 0) {
|
||||
return E->_And(OpSize::i64Bit, Bitfield, E->_InlineConstant(~SourceMaskShifted));
|
||||
} else if (C == SourceMask) {
|
||||
return E->_Or(OpSize::i64Bit, Bitfield, E->_InlineConstant(SourceMaskShifted));
|
||||
}
|
||||
}
|
||||
|
||||
if (IsConstant) {
|
||||
return E->_Bfi(OpSize::i64Bit, Size, Start, Bitfield, E->_Constant(C));
|
||||
} else {
|
||||
return E->_Bfi(OpSize::i64Bit, Size, Start, Bitfield, R);
|
||||
}
|
||||
}
|
||||
|
||||
ArithRef MaskBit(OpSize Size) {
|
||||
if (IsConstant) {
|
||||
uint64_t ShiftMask = Size == OpSize::i64Bit ? 63 : 31;
|
||||
uint64_t Result = 1ull << (C & ShiftMask);
|
||||
if (ShiftMask == 31) {
|
||||
Result &= ((1ull << 32) - 1);
|
||||
}
|
||||
|
||||
return ArithRef(E, Result);
|
||||
} else {
|
||||
return ArithRef(E, E->_Lshl(Size, E->_Constant(1), R));
|
||||
}
|
||||
}
|
||||
|
||||
Ref Ref() {
|
||||
return IsConstant ? E->_Constant(C) : R;
|
||||
}
|
||||
|
||||
bool IsDefinitelyZero() {
|
||||
return IsConstant && C == 0;
|
||||
}
|
||||
};
|
||||
|
||||
ArithRef ARef(Ref R) {
|
||||
uint64_t C;
|
||||
|
||||
if (IsValueConstant(WrapNode(R), &C)) {
|
||||
return ARef(C);
|
||||
} else {
|
||||
return ArithRef(this, R);
|
||||
}
|
||||
}
|
||||
|
||||
ArithRef ARef(uint64_t K) {
|
||||
return ArithRef(this, K);
|
||||
}
|
||||
|
||||
void InstallHostSpecificOpcodeHandlers();
|
||||
|
||||
@@ -845,7 +845,7 @@ void OpDispatchBuilder::AVX128_MOVQ(OpcodeArgs) {
|
||||
// This instruction is a bit special that if the destination is a register then it'll ZEXT the 64bit source to 256bit
|
||||
if (Op->Dest.IsGPR()) {
|
||||
// Zero bits [127:64] as well.
|
||||
Src.Low = _VMov(OpSize::i64Bit, Src.Low);
|
||||
Src.Low = VZeroExtendOperand(OpSize::i64Bit, Op->Src[0], Src.Low);
|
||||
Ref ZeroVector = LoadZeroVector(OpSize::i128Bit);
|
||||
Src.High = ZeroVector;
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Src);
|
||||
|
||||
@@ -228,21 +228,26 @@ void OpDispatchBuilder::CalculateAF(Ref Src1, Ref Src2) {
|
||||
// We only care about bit 4 in the subsequent XOR. If we'll XOR with 0,
|
||||
// there's no sense XOR'ing at all. If we'll XOR with 1, that's just
|
||||
// inverting.
|
||||
uint64_t Const;
|
||||
if (IsValueConstant(WrapNode(Src2), &Const)) {
|
||||
if (Const & (1u << 4)) {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(_Not(OpSize::i32Bit, Src1));
|
||||
} else {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(Src1);
|
||||
}
|
||||
for (unsigned i = 0; i < 2; ++i) {
|
||||
Ref SrcA = i ? Src1 : Src2;
|
||||
Ref SrcB = i ? Src2 : Src1;
|
||||
|
||||
return;
|
||||
uint64_t Const;
|
||||
if (IsValueConstant(WrapNode(SrcA), &Const)) {
|
||||
if (Const & (1u << 4)) {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(_Not(OpSize::i32Bit, SrcB));
|
||||
} else {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(SrcB);
|
||||
}
|
||||
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
// We store the XOR of the arguments. At read time, we XOR with the
|
||||
// appropriate bit of the result (available as the PF flag) and extract the
|
||||
// appropriate bit. Again 64-bit to avoid masking.
|
||||
Ref XorRes = _Xor(OpSize::i64Bit, Src1, Src2);
|
||||
Ref XorRes = Src1 == Src2 ? _Constant(0) : _Xor(OpSize::i64Bit, Src1, Src2);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
|
||||
}
|
||||
|
||||
@@ -276,7 +281,7 @@ Ref OpDispatchBuilder::CalculateFlags_ADC(IR::OpSize SrcSize, Ref Src1, Ref Src2
|
||||
CFInverted = false;
|
||||
} else {
|
||||
// Need to zero-extend for correct comparisons below
|
||||
Src2 = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Src2);
|
||||
Src2 = ARef(Src2).Bfe(0, IR::OpSizeAsBits(SrcSize)).Ref();
|
||||
|
||||
// Note that we do not extend Src2PlusCF, since we depend on proper
|
||||
// 32-bit arithmetic to correctly handle the Src2 = 0xffff case.
|
||||
@@ -316,7 +321,7 @@ Ref OpDispatchBuilder::CalculateFlags_SBB(IR::OpSize SrcSize, Ref Src1, Ref Src2
|
||||
} else {
|
||||
// Zero extend for correct comparison behaviour with Src1 = 0xffff.
|
||||
Src1 = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Src1);
|
||||
Src2 = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Src2);
|
||||
Src2 = ARef(Src2).Bfe(0, IR::OpSizeAsBits(SrcSize)).Ref();
|
||||
|
||||
auto Src2PlusCF = IncrementByCarry(OpSize, Src2);
|
||||
|
||||
@@ -426,13 +431,9 @@ void OpDispatchBuilder::CalculateFlags_UMUL(Ref High) {
|
||||
CFInverted = true;
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_Logical(IR::OpSize SrcSize, Ref Res, Ref Src1, Ref Src2) {
|
||||
void OpDispatchBuilder::CalculateFlags_Logical(IR::OpSize SrcSize, Ref Res) {
|
||||
InvalidateAF();
|
||||
|
||||
CalculatePF(Res);
|
||||
|
||||
// SF/ZF/CF/OF
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
SetNZP_ZeroCV(SrcSize, Res);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateFlags_ShiftLeftImmediate(IR::OpSize SrcSize, Ref UnmaskedRes, Ref Src1, uint64_t Shift) {
|
||||
|
||||
@@ -78,7 +78,7 @@ void OpDispatchBuilder::VMOVAPS_VMOVAPDOp(OpcodeArgs) {
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
if (Is128Bit && Op->Dest.IsGPR()) {
|
||||
Src = _VMov(OpSize::i128Bit, Src);
|
||||
Src = VZeroExtendOperand(OpSize::i128Bit, Op->Src[0], Src);
|
||||
}
|
||||
StoreResult(FPRClass, Op, Src, OpSize::iInvalid);
|
||||
}
|
||||
@@ -90,7 +90,7 @@ void OpDispatchBuilder::VMOVUPS_VMOVUPDOp(OpcodeArgs) {
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, {.Align = OpSize::i8Bit});
|
||||
|
||||
if (Is128Bit && Op->Dest.IsGPR()) {
|
||||
Src = _VMov(OpSize::i128Bit, Src);
|
||||
Src = VZeroExtendOperand(OpSize::i128Bit, Op->Src[0], Src);
|
||||
}
|
||||
StoreResult(FPRClass, Op, Src, OpSize::i8Bit);
|
||||
}
|
||||
@@ -706,7 +706,7 @@ void OpDispatchBuilder::MOVQOp(OpcodeArgs, VectorOpType VectorType) {
|
||||
const auto gpr = Op->Dest.Data.GPR.GPR;
|
||||
const auto gprIndex = gpr - X86State::REG_XMM_0;
|
||||
|
||||
auto Reg = _VMov(OpSize::i64Bit, Src);
|
||||
auto Reg = VZeroExtendOperand(OpSize::i64Bit, Op->Src[0], Src);
|
||||
StoreXMMRegister_WithAVXInsert(VectorType, gprIndex, Reg);
|
||||
} else {
|
||||
// This is simple, just store the result
|
||||
@@ -762,11 +762,12 @@ void OpDispatchBuilder::MOVMSKOp(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
Ref Tmp = _VExtractToGPR(Size, ElementSize, Src, i);
|
||||
Tmp = _Bfe(ElementSize, 1, IR::OpSizeAsBits(ElementSize) - 1, Tmp);
|
||||
|
||||
// Shift it to the correct location
|
||||
Tmp = _Lshl(ElementSize, Tmp, _Constant(i));
|
||||
|
||||
// Or it with the current value
|
||||
CurrentVal = _Or(OpSize::i64Bit, CurrentVal, Tmp);
|
||||
// Shift it to the correct location and or it with the current value
|
||||
if (i != 0) {
|
||||
CurrentVal = _Orlshl(OpSize::i64Bit, CurrentVal, Tmp, i);
|
||||
} else {
|
||||
CurrentVal = Tmp;
|
||||
}
|
||||
}
|
||||
StoreResult(GPRClass, Op, CurrentVal, OpSize::iInvalid);
|
||||
}
|
||||
@@ -3006,7 +3007,7 @@ void OpDispatchBuilder::MOVQ2DQ(OpcodeArgs) {
|
||||
if constexpr (ToXMM) {
|
||||
const auto Index = Op->Dest.Data.GPR.GPR - FEXCore::X86State::REG_XMM_0;
|
||||
|
||||
Src = _VMov(OpSize::i128Bit, Src);
|
||||
Src = VZeroExtendOperand(OpSize::i128Bit, Op->Src[0], Src);
|
||||
StoreXMMRegister(Index, Src);
|
||||
} else {
|
||||
// This is simple, just store the result
|
||||
|
||||
@@ -31,14 +31,6 @@ Ref OpDispatchBuilder::GetX87Top() {
|
||||
return _LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::GetX87Tag(Ref Value, Ref AbridgedFTW) {
|
||||
Ref RegValid = _And(OpSize::i32Bit, _Lshr(OpSize::i32Bit, AbridgedFTW, Value), _Constant(1));
|
||||
Ref X87Empty = _Constant(static_cast<uint8_t>(FPState::X87Tag::Empty));
|
||||
Ref X87Valid = _Constant(static_cast<uint8_t>(FPState::X87Tag::Valid));
|
||||
|
||||
return _Select(FEXCore::IR::COND_EQ, RegValid, _Constant(0), X87Empty, X87Valid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SetX87FTW(Ref FTW) {
|
||||
Ref X87Empty = _Constant(static_cast<uint8_t>(FPState::X87Tag::Empty));
|
||||
Ref NewAbridgedFTW {};
|
||||
@@ -312,14 +304,25 @@ void OpDispatchBuilder::FSUB(OpcodeArgs, IR::OpSize Width, bool Integer, bool Re
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::GetX87FTW_Helper() {
|
||||
Ref FTW = _Constant(0);
|
||||
// AbridgedFTWIndex has 1-bit per slot (8 slots). Duplicate each bit to get
|
||||
// 2-bits per slot (16-bit result). Duplicating bits is equivalent to
|
||||
// Morton interleaving a number with itself. To interleave efficiently two
|
||||
// bytes, we use the well-known bit twiddling algorithm:
|
||||
//
|
||||
// https://graphics.stanford.edu/~seander/bithacks.html#InterleaveBMN
|
||||
Ref X = LoadContext(AbridgedFTWIndex);
|
||||
X = _Orlshl(OpSize::i32Bit, X, X, 4);
|
||||
X = _And(OpSize::i32Bit, X, _Constant(0x0f0f0f0f));
|
||||
X = _Orlshl(OpSize::i32Bit, X, X, 2);
|
||||
X = _And(OpSize::i32Bit, X, _Constant(0x33333333));
|
||||
X = _Orlshl(OpSize::i32Bit, X, X, 1);
|
||||
X = _And(OpSize::i32Bit, X, _Constant(0x55555555));
|
||||
X = _Orlshl(OpSize::i32Bit, X, X, 1);
|
||||
|
||||
for (int i = 0; i < 8; i++) {
|
||||
Ref RegTag = GetX87Tag(_Constant(i), LoadContext(AbridgedFTWIndex));
|
||||
FTW = _Orlshl(OpSize::i32Bit, FTW, RegTag, i * 2);
|
||||
}
|
||||
|
||||
return FTW;
|
||||
// The above sequence sets valid to 11 and empty to 00, so invert to finalize.
|
||||
static_assert(static_cast<uint8_t>(FPState::X87Tag::Valid) == 0b00);
|
||||
static_assert(static_cast<uint8_t>(FPState::X87Tag::Empty) == 0b11);
|
||||
return _Xor(OpSize::i32Bit, X, _Constant(0xffff));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) {
|
||||
|
||||
@@ -47,19 +47,12 @@ AOTIRInlineEntry* AOTIRInlineIndex::Find(uint64_t GuestStart) {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
IR::RegisterAllocationData* AOTIRInlineEntry::GetRAData() {
|
||||
return (IR::RegisterAllocationData*)InlineData;
|
||||
}
|
||||
|
||||
IR::IRListView* AOTIRInlineEntry::GetIRData() {
|
||||
auto RAData = GetRAData();
|
||||
auto Offset = RAData->Size(RAData->MapCount);
|
||||
|
||||
return (IR::IRListView*)&InlineData[Offset];
|
||||
return (IR::IRListView*)InlineData;
|
||||
}
|
||||
|
||||
void AOTIRCaptureCacheEntry::AppendAOTIRCaptureCache(uint64_t GuestRIP, uint64_t Start, uint64_t Length, uint64_t Hash,
|
||||
const FEXCore::IR::IRListView& IRList, const FEXCore::IR::RegisterAllocationData* RAData) {
|
||||
const FEXCore::IR::IRListView& IRList) {
|
||||
auto Inserted = Index.emplace(GuestRIP, Stream->Offset());
|
||||
|
||||
if (Inserted.second) {
|
||||
@@ -69,8 +62,6 @@ void AOTIRCaptureCacheEntry::AppendAOTIRCaptureCache(uint64_t GuestRIP, uint64_t
|
||||
};
|
||||
Stream->Write((const char*)&entry, sizeof(entry));
|
||||
|
||||
RAData->Serialize(*Stream);
|
||||
|
||||
// IRData (inline)
|
||||
IRList.Serialize(*Stream);
|
||||
}
|
||||
@@ -265,9 +256,6 @@ class IRInlineStorage : public IRStorageBase {
|
||||
public:
|
||||
IRInlineStorage(AOTIRInlineEntry& entry)
|
||||
: entry(entry) {}
|
||||
const RegisterAllocationData* RAData() override {
|
||||
return entry.GetRAData();
|
||||
}
|
||||
IRListView GetIRView() override {
|
||||
return entry.GetIRData();
|
||||
}
|
||||
@@ -332,7 +320,7 @@ bool AOTIRCaptureCache::PostCompileCode(FEXCore::Core::InternalThreadState* Thre
|
||||
}
|
||||
|
||||
// Add to AOT cache if aot generation is enabled
|
||||
if (GeneratedIR && IR->RAData() && (CTX->Config.AOTIRCapture() || CTX->Config.AOTIRGenerate())) {
|
||||
if (GeneratedIR && (CTX->Config.AOTIRCapture() || CTX->Config.AOTIRGenerate())) {
|
||||
|
||||
auto hash = XXH3_64bits((void*)StartAddr, Length);
|
||||
|
||||
@@ -356,7 +344,7 @@ bool AOTIRCaptureCache::PostCompileCode(FEXCore::Core::InternalThreadState* Thre
|
||||
uint64_t tag = FEXCore::IR::AOTIR_COOKIE;
|
||||
AotFile->Stream->Write(&tag, sizeof(tag));
|
||||
}
|
||||
AotFile->AppendAOTIRCaptureCache(LocalRIP, LocalStartAddr, Length, hash, IR->GetIRView(), IR->RAData());
|
||||
AotFile->AppendAOTIRCaptureCache(LocalRIP, LocalStartAddr, Length, hash, IR->GetIRView());
|
||||
});
|
||||
|
||||
if (CTX->Config.AOTIRGenerate()) {
|
||||
|
||||
@@ -50,7 +50,6 @@ class ContextImpl;
|
||||
}
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class RegisterAllocationData;
|
||||
class IRListView;
|
||||
|
||||
constexpr auto COOKIE_VERSION = [](const char CookieText[4], uint32_t Version) {
|
||||
@@ -75,10 +74,9 @@ struct AOTIRInlineEntry {
|
||||
uint64_t GuestHash;
|
||||
uint64_t GuestLength;
|
||||
|
||||
/* RAData followed by IRData */
|
||||
/* IRData */
|
||||
uint8_t InlineData[0];
|
||||
|
||||
IR::RegisterAllocationData* GetRAData();
|
||||
IR::IRListView* GetIRData();
|
||||
};
|
||||
|
||||
@@ -100,8 +98,7 @@ struct AOTIRCaptureCacheEntry {
|
||||
fextl::unique_ptr<FEXCore::Context::AOTIRWriter> Stream;
|
||||
fextl::map<uint64_t, uint64_t> Index;
|
||||
|
||||
void AppendAOTIRCaptureCache(uint64_t GuestRIP, uint64_t Start, uint64_t Length, uint64_t Hash, const FEXCore::IR::IRListView& IRList,
|
||||
const FEXCore::IR::RegisterAllocationData* RAData);
|
||||
void AppendAOTIRCaptureCache(uint64_t GuestRIP, uint64_t Start, uint64_t Length, uint64_t Hash, const FEXCore::IR::IRListView& IRList);
|
||||
};
|
||||
|
||||
struct AOTIRCacheEntry {
|
||||
|
||||
@@ -11,7 +11,6 @@ namespace FEXCore::IR {
|
||||
|
||||
class OrderedNode;
|
||||
class RegisterAllocationPass;
|
||||
class RegisterAllocationData;
|
||||
|
||||
/**
|
||||
* @brief The IROp_Header is an dynamically sized array
|
||||
@@ -106,8 +105,7 @@ struct NodeID final {
|
||||
*/
|
||||
template<typename Type>
|
||||
struct FEX_PACKED NodeWrapperBase final {
|
||||
// On x86-64 using a uint64_t type is more efficient since RIP addressing gives you [<Base> + <Index> + <imm offset>]
|
||||
// On AArch64 using uint32_t is just more memory efficient. 32bit or 64bit offset doesn't matter
|
||||
// 32bit or 64bit offset doesn't matter for addressing.
|
||||
// We use uint32_t to be more memory efficient (Cuts our node list size in half)
|
||||
using NodeOffsetType = uint32_t;
|
||||
NodeOffsetType NodeOffset;
|
||||
@@ -141,22 +139,59 @@ struct FEX_PACKED NodeWrapperBase final {
|
||||
return NodeOffset == 0;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsImmediate() const {
|
||||
return NodeOffset & (1u << 31);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsPointer() const {
|
||||
return !IsImmediate();
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
Type* GetNode(uintptr_t Base) {
|
||||
LOGMAN_THROW_A_FMT(IsPointer(), "Precondition");
|
||||
return reinterpret_cast<Type*>(Base + NodeOffset);
|
||||
}
|
||||
[[nodiscard]]
|
||||
const Type* GetNode(uintptr_t Base) const {
|
||||
LOGMAN_THROW_A_FMT(IsPointer(), "Precondition");
|
||||
return reinterpret_cast<const Type*>(Base + NodeOffset);
|
||||
}
|
||||
|
||||
void SetOffset(uintptr_t Base, uintptr_t Value) {
|
||||
NodeOffset = Value - Base;
|
||||
LOGMAN_THROW_A_FMT(IsPointer(), "Offsets are within 2GiB range");
|
||||
}
|
||||
|
||||
void SetInvalid() {
|
||||
NodeOffset = 0;
|
||||
LOGMAN_THROW_A_FMT(IsInvalid(), "Zero state");
|
||||
}
|
||||
|
||||
void SetImmediate(uint32_t Immediate) {
|
||||
LOGMAN_THROW_A_FMT(Immediate < (1u << 31), "Bounded");
|
||||
NodeOffset = Immediate | (1u << 31);
|
||||
LOGMAN_THROW_A_FMT(IsImmediate(), "Encoded above");
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
uint32_t GetImmediate() const {
|
||||
LOGMAN_THROW_A_FMT(IsImmediate(), "Precondition: must be an immediate");
|
||||
return NodeOffset & ~(1u << 31);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
friend constexpr bool
|
||||
operator==(const NodeWrapperBase<Type>&, const NodeWrapperBase<Type>&) = default;
|
||||
|
||||
[[nodiscard]]
|
||||
static NodeWrapperBase<Type> FromImmediate(uint32_t Immediate) {
|
||||
NodeWrapperBase<Type> A;
|
||||
A.SetImmediate(Immediate);
|
||||
return A;
|
||||
}
|
||||
};
|
||||
|
||||
static_assert(std::is_trivially_copyable_v<NodeWrapperBase<OrderedNode>>);
|
||||
@@ -196,6 +231,15 @@ public:
|
||||
OrderedNodeHeader Header;
|
||||
uint32_t NumUses;
|
||||
|
||||
// After RA, the register allocated for the node. This is the register for the
|
||||
// node at the time it is written, even if it is shuffled into other registers
|
||||
// later. In other words, it is the register destination of the instruction
|
||||
// represented by this OrderedNode.
|
||||
//
|
||||
// This is the raw value of a PhysicalRegister data structure.
|
||||
uint8_t Reg;
|
||||
uint8_t Pad[3];
|
||||
|
||||
using value_type = OrderedNodeWrapper;
|
||||
|
||||
OrderedNode() = default;
|
||||
@@ -358,7 +402,7 @@ private:
|
||||
static_assert(std::is_trivially_constructible_v<OrderedNode>);
|
||||
static_assert(std::is_trivially_copyable_v<OrderedNode>);
|
||||
static_assert(offsetof(OrderedNode, Header) == 0);
|
||||
static_assert(sizeof(OrderedNode) == (sizeof(OrderedNodeHeader) + sizeof(uint32_t)));
|
||||
static_assert(sizeof(OrderedNode) == (sizeof(OrderedNodeHeader) + 2 * sizeof(uint32_t)));
|
||||
|
||||
// This is temporary. We are transitioning away from OrderedNode's in favour of
|
||||
// flat Ref words. To ease porting, we have this typedef. Eventually OrderedNode
|
||||
@@ -726,7 +770,7 @@ inline NodeID NodeWrapperBase<Type>::ID() const {
|
||||
bool IsFragmentExit(FEXCore::IR::IROps Op);
|
||||
bool IsBlockExit(FEXCore::IR::IROps Op);
|
||||
|
||||
void Dump(fextl::stringstream* out, const IRListView* IR, const IR::RegisterAllocationData* RAData);
|
||||
void Dump(fextl::stringstream* out, const IRListView* IR);
|
||||
} // namespace FEXCore::IR
|
||||
|
||||
template<>
|
||||
|
||||
@@ -79,15 +79,13 @@
|
||||
|
||||
"constexpr uint8_t COND_AL = 32 /* always */",
|
||||
|
||||
"constexpr FEXCore::IR::RegisterClassType GPRClass {0}",
|
||||
"constexpr FEXCore::IR::RegisterClassType GPRFixedClass {1}",
|
||||
"constexpr FEXCore::IR::RegisterClassType FPRClass {2}",
|
||||
"constexpr FEXCore::IR::RegisterClassType FPRFixedClass {3}",
|
||||
"constexpr FEXCore::IR::RegisterClassType InvalidClass {0}",
|
||||
"constexpr FEXCore::IR::RegisterClassType GPRClass {1}",
|
||||
"constexpr FEXCore::IR::RegisterClassType GPRFixedClass {2}",
|
||||
"constexpr FEXCore::IR::RegisterClassType FPRClass {3}",
|
||||
"constexpr FEXCore::IR::RegisterClassType FPRFixedClass {4}",
|
||||
"constexpr FEXCore::IR::RegisterClassType ComplexClass {5}",
|
||||
"constexpr FEXCore::IR::RegisterClassType InvalidClass {7}",
|
||||
"",
|
||||
"// Only up to 30 registers per register class",
|
||||
"constexpr uint8_t InvalidReg {31}",
|
||||
"constexpr uint8_t NumClasses {6}",
|
||||
"",
|
||||
"constexpr FEXCore::IR::TypeDefinition i8 {TypeDefinition::Create(1, 0)}",
|
||||
"constexpr FEXCore::IR::TypeDefinition i16 {TypeDefinition::Create(2, 0)}",
|
||||
@@ -169,11 +167,11 @@
|
||||
"SwitchGen": false,
|
||||
"JITDispatchOverride": "NoOp"
|
||||
},
|
||||
"IRHeader SSA:$Blocks, u64:$OriginalRIP, u32:$BlockCount, u32:$NumHostInstructions, i1:$HasX87{false}, i1:$ReadsParity{false}": {
|
||||
"IRHeader SSA:$Blocks, u64:$OriginalRIP, u32:$BlockCount, u32:$NumHostInstructions, u32:$SpillSlots, i1:$PostRA{false}, i1:$HasX87{false}, i1:$ReadsParity{false}": {
|
||||
"SwitchGen": false,
|
||||
"JITDispatchOverride": "NoOp"
|
||||
},
|
||||
"CodeBlock SSA:$Begin, SSA:$Last": {
|
||||
"CodeBlock SSA:$Begin, SSA:$Last, u32:$ID": {
|
||||
"SwitchGen": false,
|
||||
"RAOverride": "0",
|
||||
"JITDispatchOverride": "NoOp"
|
||||
@@ -254,19 +252,22 @@
|
||||
"it cannot use a regular destination too. This ensures RA correctness.",
|
||||
"This is a kludge to deal with the IR's lack of multiple destinations",
|
||||
"If ForPair is set, RA will try to allocate the base of a register pair"],
|
||||
"DestSize": "OpSize::i64Bit"
|
||||
"DestSize": "OpSize::i64Bit",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = AllocateFPR OpSize:#RegisterSize, OpSize:#ElementSize": {
|
||||
"Desc": ["Like AllocateGPR, but for FPR"],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
"ElementSize": "ElementSize",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"GPR = AllocateGPRAfter GPR:$After": {
|
||||
"Desc": ["Silly pseudo-instruction to allocate a register for a future destination",
|
||||
"This is a kludge to deal with the IR's lack of multiple destinations",
|
||||
"RA will attempt to allocate to the register after $After.",
|
||||
"It may not succeed."],
|
||||
"DestSize": "OpSize::i64Bit"
|
||||
"DestSize": "OpSize::i64Bit",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"GPR = RDRAND i1:$GetReseeded": {
|
||||
"Desc": ["Uses the hardware random number generator to generate a 64bit number",
|
||||
@@ -294,11 +295,11 @@
|
||||
"HasSideEffects": true,
|
||||
"RAOverride": "2"
|
||||
},
|
||||
"ExitFunction GPR:$NewRIP": {
|
||||
"ExitFunction OpSize:#Size, GPR:$NewRIP": {
|
||||
"Desc": ["Exits the current JIT function with a target RIP"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "GetOpSize(NewRIP)"
|
||||
"DestSize": "Size"
|
||||
},
|
||||
"Break BreakDefinition:$Reason": {
|
||||
"HasSideEffects": true
|
||||
@@ -377,14 +378,11 @@
|
||||
"DestSize": "Size"
|
||||
},
|
||||
|
||||
"StoreRegister SSA:$Value, u32:$Reg, RegisterClass:$Class, OpSize:#Size": {
|
||||
"SSA = StoreRegister SSA:$Value, OpSize:#Size": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Stores a value to a given register.",
|
||||
"Size must match the execution mode."],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value) == $Class"
|
||||
]
|
||||
"DestSize": "Size"
|
||||
},
|
||||
|
||||
"StorePF GPR:$Value, OpSize:#Size": {
|
||||
@@ -500,17 +498,13 @@
|
||||
]
|
||||
},
|
||||
|
||||
"SSA = FillRegister SSA:$OriginalValue, u32:$Slot, RegisterClass:$Class": {
|
||||
"SSA = FillRegister OpSize:#Size, OpSize:#ElementSize, u32:$Slot, RegisterClass:$Class": {
|
||||
"Desc": ["Fills a register from a spill slot",
|
||||
"Spill slots are register allocated and has live ranges calculated to handle slot calculation",
|
||||
"```diff\n- !Don't use this op. It is for RA to handle spilling and filling!\n```",
|
||||
"",
|
||||
"The OriginalValue SSA arg points at the original SSA value spilled, and only exists for",
|
||||
"RA validation purposes"
|
||||
"```diff\n- !Don't use this op. It is for RA to handle spilling and filling!\n```"
|
||||
],
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($OriginalValue) == $Class"
|
||||
]
|
||||
"DestSize": "Size",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
|
||||
"GPR = LoadNZCV": {
|
||||
@@ -672,11 +666,19 @@
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size"
|
||||
},
|
||||
"PushTwo OpSize:#Size, OpSize:$ValueSize, GPR:$Value1, GPR:$Value2, GPR:$Addr": {
|
||||
"Desc": [
|
||||
"Push two values to the address, incrementing the pointer in the place.",
|
||||
"Fused post-RA so doesn't have a destination."
|
||||
],
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"GPR = RMWHandle GPR:$Value": {
|
||||
"Desc": [
|
||||
"This is a special move that indicates the result will be poisoned by a non-SSA instruction writing to its result.",
|
||||
"In effect, it serves to prevent invalid optimizations with non-SSA instructions."
|
||||
],
|
||||
"DestSize": "OpSize::i64Bit",
|
||||
"HasSideEffects": true,
|
||||
"TiedSource": 0
|
||||
},
|
||||
@@ -688,6 +690,11 @@
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size"
|
||||
},
|
||||
"GPR:$Addr, GPR:$Value1, GPR:$Value2 = PopTwo OpSize:$Size, GPR:$Addr": {
|
||||
"Desc": ["Pop two values from the address. Fused post-RA."],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size"
|
||||
},
|
||||
"GPR = MemSet i1:$IsAtomic, OpSize:$Size, GPR:$Prefix, GPR:$Addr, GPR:$Value, GPR:$Length, GPR:$Direction": {
|
||||
"Desc": ["Duplicates behaviour of x86 STOS repeat",
|
||||
"Returns the final address that gets generated without the prefix appended."
|
||||
@@ -1447,38 +1454,6 @@
|
||||
],
|
||||
"DestSize": "FEXCore::IR::OpSize::i64Bit"
|
||||
},
|
||||
"GPR = Div OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer signed division"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i8Bit || Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = UDiv OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer unsigned division"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i8Bit || Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = Rem OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer signed remainder"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i8Bit || Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = URem OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer unsigned remainder"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i8Bit || Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = MulH OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer signed multiply returning high results",
|
||||
"op:",
|
||||
@@ -1623,45 +1598,23 @@
|
||||
]
|
||||
},
|
||||
|
||||
"GPR = LDiv OpSize:#Size, GPR:$Lower, GPR:$Upper, GPR:$Divisor": {
|
||||
"GPR:$Quotient, GPR:$Remainder = Div OpSize:#Size, GPR:$Lower, GPR:$Upper, GPR:$Divisor": {
|
||||
"Desc": ["Integer long signed division returning lower bits",
|
||||
"The Lower and Upper registers will be concated together to generate a dividend twice the size",
|
||||
"Then the divisor divides the temporary dividend and returns the results in the original sized register",
|
||||
"If Upper is invalid, this is a non-long division."
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"GPR:$Quotient, GPR:$Remainder = UDiv OpSize:#Size, GPR:$Lower, GPR:$Upper, GPR:$Divisor": {
|
||||
"Desc": ["Integer long unsigned division returning lower bits",
|
||||
"The Lower and Upper registers will be concated together to generate a dividend twice the size",
|
||||
"Then the divisor divides the temporary dividend and returns the results in the original sized register"
|
||||
"Then the divisor divides the temporary dividend and returns the results in the original sized register",
|
||||
"If Upper is invalid, this is a non-long division."
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = LUDiv OpSize:#Size, GPR:$Lower, GPR:$Upper, GPR:$Divisor": {
|
||||
"Desc": ["Integer long unsigned division returning lower bits",
|
||||
"The Lower and Upper registers will be concated together to generate a dividend twice the size",
|
||||
"Then the divisor divides the temporary dividend and returns the results in the original sized register"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = LRem OpSize:#Size, GPR:$Lower, GPR:$Upper, GPR:$Divisor": {
|
||||
"Desc": ["Integer long signed remainder returning lower bits",
|
||||
"The Lower and Upper registers will be concated together to generate a dividend twice the size",
|
||||
"Then the divisor divides the temporary dividend and returns the remainder results in the original sized register"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = LURem OpSize:#Size, GPR:$Lower, GPR:$Upper, GPR:$Divisor": {
|
||||
"Desc": ["Integer long unsigned remainder returning lower bits",
|
||||
"The Lower and Upper registers will be concated together to generate a dividend twice the size",
|
||||
"Then the divisor divides the temporary dividend and returns the remainder results in the original sized register"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
"HasSideEffects": true
|
||||
},
|
||||
|
||||
"Float to GPR": {"Ignore": 1},
|
||||
@@ -2835,6 +2788,11 @@
|
||||
"FPR = F64COS FPR:$Src": {
|
||||
"DestSize": "OpSize::i64Bit",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR:$Sin, FPR:$Cos = F64SINCOS FPR:$Src": {
|
||||
"DestSize": "OpSize::i64Bit",
|
||||
"HasSideEffects": true,
|
||||
"JITDispatch": false
|
||||
}
|
||||
},
|
||||
"F80": {
|
||||
@@ -3174,6 +3132,11 @@
|
||||
"DestSize": "OpSize::i128Bit",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR:$Sin, FPR:$Cos = F80SINCOS FPR:$X80Src": {
|
||||
"DestSize": "OpSize::i128Bit",
|
||||
"HasSideEffects": true,
|
||||
"JITDispatch": false
|
||||
},
|
||||
"F80SINCOSStack": {
|
||||
"X87": true,
|
||||
"HasSideEffects": true
|
||||
|
||||
@@ -82,7 +82,27 @@ static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNodeWrapper Arg, const IR::RegisterAllocationData* RAData) {
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNodeWrapper Arg) {
|
||||
if (Arg.IsImmediate()) {
|
||||
auto PhyReg = PhysicalRegister(Arg);
|
||||
|
||||
switch (PhyReg.Class) {
|
||||
case FEXCore::IR::GPRClass.Val: *out << "r"; break;
|
||||
case FEXCore::IR::GPRFixedClass.Val: *out << "R"; break;
|
||||
case FEXCore::IR::FPRClass.Val: *out << "v"; break;
|
||||
case FEXCore::IR::FPRFixedClass.Val: *out << "V"; break;
|
||||
case FEXCore::IR::ComplexClass.Val: *out << "c"; break;
|
||||
case FEXCore::IR::InvalidClass.Val: *out << "invalid"; break;
|
||||
default: *out << "unknown"; break;
|
||||
}
|
||||
|
||||
if (PhyReg.Class != FEXCore::IR::InvalidClass.Val) {
|
||||
*out << std::dec << (uint32_t)PhyReg.Reg;
|
||||
}
|
||||
|
||||
return;
|
||||
}
|
||||
|
||||
auto [CodeNode, IROp] = IR->at(Arg)();
|
||||
const auto ArgID = Arg.ID();
|
||||
|
||||
@@ -90,25 +110,6 @@ static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNode
|
||||
*out << "%Invalid";
|
||||
} else {
|
||||
*out << "%" << std::dec << ArgID;
|
||||
if (RAData) {
|
||||
auto PhyReg = RAData->GetNodeRegister(ArgID);
|
||||
|
||||
switch (PhyReg.Class) {
|
||||
case FEXCore::IR::GPRClass.Val: *out << "(GPR"; break;
|
||||
case FEXCore::IR::GPRFixedClass.Val: *out << "(GPRFixed"; break;
|
||||
case FEXCore::IR::FPRClass.Val: *out << "(FPR"; break;
|
||||
case FEXCore::IR::FPRFixedClass.Val: *out << "(FPRFixed"; break;
|
||||
case FEXCore::IR::ComplexClass.Val: *out << "(Complex"; break;
|
||||
case FEXCore::IR::InvalidClass.Val: *out << "(Invalid"; break;
|
||||
default: *out << "(Unknown"; break;
|
||||
}
|
||||
|
||||
if (PhyReg.Class != FEXCore::IR::InvalidClass.Val) {
|
||||
*out << std::dec << (uint32_t)PhyReg.Reg << ")";
|
||||
} else {
|
||||
*out << ")";
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (GetHasDest(IROp->Op)) {
|
||||
@@ -271,7 +272,7 @@ static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView
|
||||
}
|
||||
}
|
||||
|
||||
void Dump(fextl::stringstream* out, const IRListView* IR, const IR::RegisterAllocationData* RAData) {
|
||||
void Dump(fextl::stringstream* out, const IRListView* IR) {
|
||||
auto HeaderOp = IR->GetHeader();
|
||||
|
||||
int8_t CurrentIndent = 0;
|
||||
@@ -323,16 +324,16 @@ void Dump(fextl::stringstream* out, const IRListView* IR, const IR::RegisterAllo
|
||||
|
||||
*out << "%" << std::dec << ID;
|
||||
|
||||
if (RAData) {
|
||||
auto PhyReg = RAData->GetNodeRegister(ID);
|
||||
auto PhyReg = PhysicalRegister(CodeNode);
|
||||
if (!PhyReg.IsInvalid()) {
|
||||
switch (PhyReg.Class) {
|
||||
case FEXCore::IR::GPRClass.Val: *out << "(GPR"; break;
|
||||
case FEXCore::IR::GPRFixedClass.Val: *out << "(GPRFixed"; break;
|
||||
case FEXCore::IR::FPRClass.Val: *out << "(FPR"; break;
|
||||
case FEXCore::IR::FPRFixedClass.Val: *out << "(FPRFixed"; break;
|
||||
case FEXCore::IR::ComplexClass.Val: *out << "(Complex"; break;
|
||||
case FEXCore::IR::InvalidClass.Val: *out << "(Invalid"; break;
|
||||
default: *out << "(Unknown"; break;
|
||||
case FEXCore::IR::GPRClass.Val: *out << "(r"; break;
|
||||
case FEXCore::IR::GPRFixedClass.Val: *out << "(R"; break;
|
||||
case FEXCore::IR::FPRClass.Val: *out << "(v"; break;
|
||||
case FEXCore::IR::FPRFixedClass.Val: *out << "(V"; break;
|
||||
case FEXCore::IR::ComplexClass.Val: *out << "(complex"; break;
|
||||
case FEXCore::IR::InvalidClass.Val: *out << "(invalid"; break;
|
||||
default: *out << "(unknown"; break;
|
||||
}
|
||||
if (PhyReg.Class != FEXCore::IR::InvalidClass.Val) {
|
||||
*out << std::dec << (uint32_t)PhyReg.Reg << ")";
|
||||
|
||||
@@ -146,6 +146,10 @@ void IREmitter::RemoveArgUses(Ref Node) {
|
||||
}
|
||||
}
|
||||
|
||||
void IREmitter::RemovePostRA(Ref Node) {
|
||||
Node->Unlink(DualListData.ListBegin());
|
||||
}
|
||||
|
||||
void IREmitter::Remove(Ref Node) {
|
||||
RemoveArgUses(Node);
|
||||
|
||||
@@ -185,27 +189,4 @@ void IREmitter::SetCurrentCodeBlock(Ref Node) {
|
||||
SetWriteCursor(Node->Op(DualListData.DataBegin())->CW<IROp_CodeBlock>()->Begin.GetNode(DualListData.ListBegin()));
|
||||
}
|
||||
|
||||
void IREmitter::ReplaceWithConstant(Ref Node, uint64_t Value) {
|
||||
auto Header = Node->Op(DualListData.DataBegin());
|
||||
|
||||
if (IRSizes[Header->Op] >= sizeof(IROp_Constant)) {
|
||||
// Unlink any arguments the node currently has
|
||||
RemoveArgUses(Node);
|
||||
|
||||
// Overwrite data with the new constant op
|
||||
Header->Op = OP_CONSTANT;
|
||||
auto Const = Header->CW<IROp_Constant>();
|
||||
Const->Constant = Value;
|
||||
} else {
|
||||
// Fallback path for when the node to overwrite is too small
|
||||
auto cursor = GetWriteCursor();
|
||||
SetWriteCursor(Node);
|
||||
|
||||
auto NewNode = _Constant(Value);
|
||||
ReplaceAllUsesWith(Node, NewNode);
|
||||
|
||||
SetWriteCursor(cursor);
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -182,11 +182,6 @@ public:
|
||||
return NodeIterator(DualListData.ListBegin(), DualListData.DataBegin(), wrapper);
|
||||
}
|
||||
|
||||
// Overwrite a node with a constant
|
||||
// Depending on what node has been overwritten, there might be some unallocated space around the node
|
||||
// Because we are overwriting the node, we don't have to worry about update all the arguments which use it
|
||||
void ReplaceWithConstant(Ref Node, uint64_t Value);
|
||||
|
||||
void ReplaceAllUsesWithRange(Ref Node, Ref NewNode, AllNodesIterator Begin, AllNodesIterator End);
|
||||
|
||||
void ReplaceUsesWithAfter(Ref Node, Ref NewNode, AllNodesIterator After) {
|
||||
@@ -201,24 +196,10 @@ public:
|
||||
ReplaceUsesWithAfter(Node, NewNode, It);
|
||||
}
|
||||
|
||||
void ReplaceAllUsesWith(Ref Node, Ref NewNode) {
|
||||
auto Start = AllNodesIterator(DualListData.ListBegin(), DualListData.DataBegin(), Node->Wrapped(DualListData.ListBegin()));
|
||||
|
||||
ReplaceAllUsesWithRange(Node, NewNode, Start, AllNodesIterator(DualListData.ListBegin(), DualListData.DataBegin()));
|
||||
|
||||
LOGMAN_THROW_A_FMT(Node->NumUses == 0, "Node still used");
|
||||
|
||||
auto IROp = Node->Op(DualListData.DataBegin())->CW<FEXCore::IR::IROp_Header>();
|
||||
// We can not remove the op if there are side-effects
|
||||
if (!IR::HasSideEffects(IROp->Op)) {
|
||||
// Since we have deleted ALL uses, we can safely delete the node.
|
||||
Remove(Node);
|
||||
}
|
||||
}
|
||||
|
||||
void ReplaceNodeArgument(Ref Node, uint8_t Arg, Ref NewArg);
|
||||
|
||||
void Remove(Ref Node);
|
||||
void RemovePostRA(Ref Node);
|
||||
|
||||
void SetPackedRFLAG(bool Lower8, Ref Src);
|
||||
Ref GetPackedRFLAG(bool Lower8);
|
||||
@@ -270,7 +251,8 @@ public:
|
||||
IRPair<IROp_CodeBlock> CreateCodeNode() {
|
||||
SetWriteCursor(nullptr); // Orphan from any previous nodes
|
||||
|
||||
auto CodeNode = _CodeBlock(InvalidNode, InvalidNode);
|
||||
auto ID = ViewIR().GetHeader()->BlockCount++;
|
||||
auto CodeNode = _CodeBlock(InvalidNode, InvalidNode, ID);
|
||||
|
||||
CodeBlocks.emplace_back(CodeNode);
|
||||
|
||||
|
||||
@@ -141,7 +141,7 @@ public:
|
||||
}
|
||||
|
||||
private:
|
||||
Utils::FixedSizePooledAllocation<uintptr_t, 5000, 500> PoolObject;
|
||||
Utils::PoolBufferWithTimedRetirement<uintptr_t, 5000, 500> PoolObject;
|
||||
};
|
||||
|
||||
class IRListView final {
|
||||
@@ -234,6 +234,16 @@ public:
|
||||
return GetOp<IROp_IRHeader>(GetHeaderNode());
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
unsigned PostRA() const {
|
||||
return GetHeader()->PostRA;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
unsigned SpillSlots() const {
|
||||
return GetHeader()->SpillSlots;
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
[[nodiscard]]
|
||||
T* GetOp(Ref Node) const {
|
||||
@@ -409,9 +419,6 @@ class IRStorageBase {
|
||||
public:
|
||||
virtual ~IRStorageBase() = default;
|
||||
|
||||
// Optional RA data. Returns nullptr if none present
|
||||
virtual const RegisterAllocationData* RAData() = 0;
|
||||
|
||||
virtual IRListView GetIRView() = 0;
|
||||
};
|
||||
|
||||
|
||||
@@ -71,7 +71,7 @@ void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl* ctx) {
|
||||
|
||||
if (!DisablePasses()) {
|
||||
InsertPass(CreateX87StackOptimizationPass(ctx->HostFeatures, ctx->GetGPROpSize()));
|
||||
InsertPass(CreateConstProp(ctx->HostFeatures.SupportsTSOImm9, &ctx->CPUID));
|
||||
InsertPass(CreateConstProp(ctx->HostFeatures.SupportsTSOImm9));
|
||||
InsertPass(CreateDeadFlagCalculationEliminination());
|
||||
}
|
||||
}
|
||||
@@ -79,12 +79,11 @@ void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl* ctx) {
|
||||
void PassManager::AddDefaultValidationPasses() {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
InsertValidationPass(Validation::CreateIRValidation(), "IRValidation");
|
||||
InsertValidationPass(Validation::CreateRAValidation());
|
||||
#endif
|
||||
}
|
||||
|
||||
void PassManager::InsertRegisterAllocationPass() {
|
||||
InsertPass(IR::CreateRegisterAllocationPass(), "RA");
|
||||
void PassManager::InsertRegisterAllocationPass(FEXCore::Context::ContextImpl* ctx) {
|
||||
InsertPass(IR::CreateRegisterAllocationPass(&ctx->CPUID), "RA");
|
||||
}
|
||||
|
||||
void PassManager::Run(IREmitter* IREmit) {
|
||||
|
||||
@@ -56,7 +56,7 @@ public:
|
||||
return PassPtr;
|
||||
}
|
||||
|
||||
void InsertRegisterAllocationPass();
|
||||
void InsertRegisterAllocationPass(FEXCore::Context::ContextImpl* ctx);
|
||||
|
||||
void Run(IREmitter* IREmit);
|
||||
|
||||
|
||||
@@ -15,16 +15,14 @@ class IntrusivePooledAllocator;
|
||||
namespace FEXCore::IR {
|
||||
class Pass;
|
||||
class RegisterAllocationPass;
|
||||
class RegisterAllocationData;
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool SupportsTSOImm9, const FEXCore::CPUIDEmu* CPUID);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool SupportsTSOImm9);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadFlagCalculationEliminination();
|
||||
fextl::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass();
|
||||
fextl::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass(const FEXCore::CPUIDEmu* CPUID);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateX87StackOptimizationPass(const FEXCore::HostFeatures&, OpSize GPROpSize);
|
||||
|
||||
namespace Validation {
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateIRValidation();
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateRAValidation();
|
||||
} // namespace Validation
|
||||
|
||||
namespace Debug {
|
||||
|
||||
@@ -9,7 +9,6 @@ $end_info$
|
||||
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
@@ -23,23 +22,6 @@ $end_info$
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
uint64_t getMask(IROp_Header* Op) {
|
||||
LOGMAN_THROW_A_FMT(Op->Size >= IR::OpSize::i8Bit && Op->Size <= IR::OpSize::i64Bit, "Invalid mask size");
|
||||
uint64_t NumBits = IR::OpSizeAsBits(Op->Size);
|
||||
return (~0ULL) >> (64 - NumBits);
|
||||
}
|
||||
|
||||
// Returns true if the number bits from [0:width) contain the same bit.
|
||||
// Ensuring that the consecutive bits in the range are entirely 0 or 1.
|
||||
static bool HasConsecutiveBits(uint64_t imm, unsigned width) {
|
||||
if (width == 0) {
|
||||
return true;
|
||||
}
|
||||
|
||||
// Credit to https://github.com/dougallj for this implementation.
|
||||
return ((imm ^ (imm >> 1)) & ((1ULL << (width - 1)) - 1)) == 0;
|
||||
}
|
||||
|
||||
// aarch64 heuristics
|
||||
static bool IsImmLogical(uint64_t imm, unsigned width) {
|
||||
if (width < 32) {
|
||||
@@ -50,18 +32,15 @@ static bool IsImmLogical(uint64_t imm, unsigned width) {
|
||||
|
||||
class ConstProp final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
explicit ConstProp(bool SupportsTSOImm9, const FEXCore::CPUIDEmu* CPUID)
|
||||
: SupportsTSOImm9 {SupportsTSOImm9}
|
||||
, CPUID {CPUID} {}
|
||||
explicit ConstProp(bool SupportsTSOImm9)
|
||||
: SupportsTSOImm9 {SupportsTSOImm9} {}
|
||||
|
||||
void Run(IREmitter* IREmit) override;
|
||||
|
||||
private:
|
||||
void HandleConstantPools(IREmitter* IREmit, const IRListView& CurrentIR);
|
||||
void ConstantPropagation(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp);
|
||||
|
||||
bool SupportsTSOImm9 {};
|
||||
const FEXCore::CPUIDEmu* CPUID;
|
||||
|
||||
template<class F>
|
||||
bool InlineIf(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp, unsigned Index, F Filter) {
|
||||
@@ -122,100 +101,6 @@ private:
|
||||
}
|
||||
};
|
||||
|
||||
// Constants are pooled per block.
|
||||
void ConstProp::HandleConstantPools(IREmitter* IREmit, const IRListView& CurrentIR) {
|
||||
const uint32_t SSACount = CurrentIR.GetSSACount();
|
||||
|
||||
// Allocation/initialization deferred until first use, since many multiblocks
|
||||
// don't have constants leftover after all inlining.
|
||||
fextl::vector<Ref> Remap {};
|
||||
|
||||
struct Entry {
|
||||
int64_t Value;
|
||||
Ref R;
|
||||
};
|
||||
|
||||
|
||||
fextl::vector<Entry> Pool {};
|
||||
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
Pool.clear();
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
if (IROp->Op == OP_CONSTANT) {
|
||||
auto Op = IROp->C<IR::IROp_Constant>();
|
||||
bool Found = false;
|
||||
|
||||
// Search for the constant. This is O(n^2) but n is small since it's
|
||||
// local and most constants are inlined. In practice, it ends up much
|
||||
// faster than a hash table.
|
||||
for (auto K : Pool) {
|
||||
if (K.Value == Op->Constant) {
|
||||
uint32_t Value = CurrentIR.GetID(CodeNode).Value;
|
||||
LOGMAN_THROW_A_FMT(Value < SSACount, "def not yet remapped");
|
||||
|
||||
if (Remap.empty()) {
|
||||
Remap.resize(SSACount, nullptr);
|
||||
}
|
||||
|
||||
Remap[Value] = K.R;
|
||||
Found = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (!Found) {
|
||||
Pool.push_back({.Value = Op->Constant, .R = CodeNode});
|
||||
}
|
||||
} else if (!Remap.empty()) {
|
||||
const uint8_t NumArgs = IR::GetArgs(IROp->Op);
|
||||
for (uint8_t i = 0; i < NumArgs; ++i) {
|
||||
if (IROp->Args[i].IsInvalid()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
uint32_t Value = IROp->Args[i].ID().Value;
|
||||
LOGMAN_THROW_A_FMT(Value < SSACount, "src not yet remapped");
|
||||
|
||||
Ref New = Remap[Value];
|
||||
if (New) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, i, New);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Helper to replace the destination of an instruction with one of its sources,
|
||||
// to implement algebraic identities. This is surprisingly tricky due to
|
||||
// implicit masking in our IR.
|
||||
//
|
||||
// FEX's IR uses sized opcodes, matching arm64 semantics. 64-bit opcodes do not
|
||||
// mask, whereas smaller opcodes mask/zero-extend from 32-bits. Therefore, if
|
||||
// the instruction is 32-bit, we need to mask the source for a sound
|
||||
// replacement, in case there was garbage in the upper bits.
|
||||
//
|
||||
// However, if that source is in turn written by a 32-bit instruction, it is
|
||||
// guaranteed to have already been masked, so we know there's no garbage and we
|
||||
// can avoid the zero-extension. This is the case 99% of the time, but the
|
||||
// masking here is correctness-bearing nevertheless (and new versions of Denuvo
|
||||
// break if you get this wrong!)
|
||||
static inline void ReplaceWithSource(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp, unsigned Idx) {
|
||||
Ref Arg = CurrentIR.GetNode(IROp->Args[Idx]);
|
||||
|
||||
if (IROp->Size < OpSize::i64Bit) {
|
||||
LOGMAN_THROW_A_FMT(IROp->Size == OpSize::i32Bit, "other sizes not here");
|
||||
|
||||
auto Header = IREmit->GetOpHeader(IROp->Args[Idx]);
|
||||
if (Header->Size > OpSize::i32Bit) {
|
||||
Arg = IREmit->_Bfe(OpSize::i32Bit, 32, 0, Arg);
|
||||
}
|
||||
}
|
||||
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, Arg);
|
||||
}
|
||||
|
||||
// constprop + some more per instruction logic
|
||||
void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp) {
|
||||
switch (IROp->Op) {
|
||||
@@ -226,7 +111,6 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
auto Op = IROp->C<IR::IROp_Add>();
|
||||
uint64_t Constant1 {};
|
||||
uint64_t Constant2 {};
|
||||
bool IsConstant1 = IREmit->IsValueConstant(IROp->Args[0], &Constant1);
|
||||
bool IsConstant2 = IREmit->IsValueConstant(IROp->Args[1], &Constant2);
|
||||
|
||||
/* IsImmAddSub assumes the constants are sign-extended, take care of that
|
||||
@@ -237,16 +121,6 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
Constant2 = (int64_t)(int32_t)Constant2;
|
||||
}
|
||||
|
||||
if (IsConstant1 && IsConstant2 && IROp->Op == OP_ADD) {
|
||||
uint64_t NewConstant = (Constant1 + Constant2) & getMask(IROp);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
break;
|
||||
} else if (IsConstant1 && IsConstant2 && IROp->Op == OP_SUB) {
|
||||
uint64_t NewConstant = (Constant1 - Constant2) & getMask(IROp);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
break;
|
||||
}
|
||||
|
||||
if (IsConstant2 && !ARMEmitter::IsImmAddSub(Constant2) && ARMEmitter::IsImmAddSub(-Constant2)) {
|
||||
// If the second argument is constant, the immediate is not ImmAddSub, but when negated is.
|
||||
// So, negate the operation to negate (and inline) the constant.
|
||||
@@ -287,74 +161,10 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_SUBSHIFT: {
|
||||
auto Op = IROp->C<IR::IROp_SubShift>();
|
||||
|
||||
uint64_t Constant1, Constant2;
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1) && IREmit->IsValueConstant(IROp->Args[1], &Constant2) &&
|
||||
Op->Shift == IR::ShiftType::LSL) {
|
||||
// Optimize the LSL case when we know both sources are constant.
|
||||
// This is a pattern that shows up with direction flag calculations if DF was set just before the operation.
|
||||
uint64_t NewConstant = (Constant1 - (Constant2 << Op->ShiftAmount)) & getMask(IROp);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_AND: {
|
||||
uint64_t Constant1 {};
|
||||
uint64_t Constant2 {};
|
||||
|
||||
bool Replaced = false;
|
||||
|
||||
// Order matter for short circuit evaluation, subsequent ifs read constant2.
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant2) && IREmit->IsValueConstant(IROp->Args[0], &Constant1)) {
|
||||
uint64_t NewConstant = (Constant1 & Constant2) & getMask(IROp);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
Replaced = true;
|
||||
} else if (IROp->Args[0].ID() == IROp->Args[1].ID() || (Constant2 & getMask(IROp)) == getMask(IROp)) {
|
||||
// AND with same value results in original value
|
||||
ReplaceWithSource(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
Replaced = true;
|
||||
}
|
||||
|
||||
if (!Replaced) {
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, [&IROp](uint64_t X) { return IsImmLogical(X, IR::OpSizeAsBits(IROp->Size)); });
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_OR: {
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, [&IROp](uint64_t X) { return IsImmLogical(X, IR::OpSizeAsBits(IROp->Size)); });
|
||||
break;
|
||||
}
|
||||
case OP_AND:
|
||||
case OP_OR:
|
||||
case OP_XOR: {
|
||||
uint64_t Constant1 {};
|
||||
|
||||
if (IROp->Args[0].ID() == IROp->Args[1].ID()) {
|
||||
// XOR with same value results to zero
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, IREmit->_Constant(0));
|
||||
} else {
|
||||
// XOR with zero results in the nonzero source
|
||||
bool Replaced = false;
|
||||
for (unsigned i = 0; i < 2; ++i) {
|
||||
if (!IREmit->IsValueConstant(IROp->Args[i], &Constant1)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if (Constant1 != 0) {
|
||||
continue;
|
||||
}
|
||||
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
ReplaceWithSource(IREmit, CurrentIR, CodeNode, IROp, 1 - i);
|
||||
Replaced = true;
|
||||
break;
|
||||
}
|
||||
|
||||
if (!Replaced) {
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, [&IROp](uint64_t X) { return IsImmLogical(X, IR::OpSizeAsBits(IROp->Size)); });
|
||||
}
|
||||
}
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, [&IROp](uint64_t X) { return IsImmLogical(X, IR::OpSizeAsBits(IROp->Size)); });
|
||||
break;
|
||||
}
|
||||
case OP_ANDWITHFLAGS:
|
||||
@@ -363,249 +173,19 @@ void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& Current
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, [&IROp](uint64_t X) { return IsImmLogical(X, IR::OpSizeAsBits(IROp->Size)); });
|
||||
break;
|
||||
}
|
||||
case OP_NEG: {
|
||||
uint64_t Constant {};
|
||||
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant)) {
|
||||
uint64_t NewConstant = -Constant;
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_ASHR:
|
||||
case OP_ROR: {
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, 1);
|
||||
break;
|
||||
}
|
||||
case OP_LSHL: {
|
||||
uint64_t Constant1 {};
|
||||
uint64_t Constant2 {};
|
||||
|
||||
if (IREmit->IsValueConstant(IROp->Args[0], &Constant1) && IREmit->IsValueConstant(IROp->Args[1], &Constant2)) {
|
||||
// Shifts mask the shift amount by 63 or 31 depending on operating size;
|
||||
uint64_t ShiftMask = IROp->Size == OpSize::i64Bit ? 63 : 31;
|
||||
uint64_t NewConstant = (Constant1 << (Constant2 & ShiftMask)) & getMask(IROp);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
} else if (IREmit->IsValueConstant(IROp->Args[1], &Constant2) && Constant2 == 0) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
ReplaceWithSource(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
} else {
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, 1);
|
||||
}
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, 1);
|
||||
break;
|
||||
}
|
||||
case OP_LSHR: {
|
||||
uint64_t Constant2 {};
|
||||
|
||||
if (IREmit->IsValueConstant(IROp->Args[1], &Constant2) && Constant2 == 0) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
ReplaceWithSource(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
} else {
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, 1);
|
||||
}
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, 1);
|
||||
break;
|
||||
}
|
||||
case OP_BFE: {
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
uint64_t Constant;
|
||||
|
||||
if (IROp->Size <= OpSize::i64Bit && IREmit->IsValueConstant(Op->Src, &Constant)) {
|
||||
uint64_t SourceMask = Op->Width == 64 ? ~0ULL : ((1ULL << Op->Width) - 1);
|
||||
SourceMask <<= Op->lsb;
|
||||
|
||||
uint64_t NewConstant = (Constant & SourceMask) >> Op->lsb;
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_SBFE: {
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
uint64_t Constant;
|
||||
if (IREmit->IsValueConstant(Op->Src, &Constant)) {
|
||||
LOGMAN_THROW_A_FMT(IROp->Size >= IR::OpSize::i8Bit && IROp->Size <= IR::OpSize::i64Bit, "Invalid size");
|
||||
// SBFE of a constant can be converted to a constant.
|
||||
uint64_t SourceMask = Op->Width == 64 ? ~0ULL : ((1ULL << Op->Width) - 1);
|
||||
uint64_t DestSizeInBits = IR::OpSizeAsBits(IROp->Size);
|
||||
uint64_t DestMask = DestSizeInBits == 64 ? ~0ULL : ((1ULL << DestSizeInBits) - 1);
|
||||
SourceMask <<= Op->lsb;
|
||||
|
||||
int64_t NewConstant = (Constant & SourceMask) >> Op->lsb;
|
||||
NewConstant <<= 64 - Op->Width;
|
||||
NewConstant >>= 64 - Op->Width;
|
||||
NewConstant &= DestMask;
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_BFI: {
|
||||
auto Op = IROp->C<IR::IROp_Bfi>();
|
||||
uint64_t ConstantSrc {};
|
||||
bool SrcIsConstant = IREmit->IsValueConstant(IROp->Args[1], &ConstantSrc);
|
||||
|
||||
if (SrcIsConstant && HasConsecutiveBits(ConstantSrc, Op->Width)) {
|
||||
// We are trying to insert constant, if it is a bitfield of only set bits then we can orr or and it.
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
uint64_t SourceMask = Op->Width == 64 ? ~0ULL : ((1ULL << Op->Width) - 1);
|
||||
uint64_t NewConstant = SourceMask << Op->lsb;
|
||||
|
||||
if (ConstantSrc & 1) {
|
||||
auto orr = IREmit->_Or(IROp->Size, CurrentIR.GetNode(IROp->Args[0]), IREmit->_Constant(NewConstant));
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, orr);
|
||||
} else {
|
||||
// We are wanting to clear the bitfield.
|
||||
auto andn = IREmit->_Andn(IROp->Size, CurrentIR.GetNode(IROp->Args[0]), IREmit->_Constant(NewConstant));
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, andn);
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_VMOV: {
|
||||
// elim from load mem
|
||||
auto source = IROp->Args[0];
|
||||
auto sourceHeader = IREmit->GetOpHeader(source);
|
||||
|
||||
if (IROp->Size >= sourceHeader->Size &&
|
||||
(sourceHeader->Op == OP_LOADMEM || sourceHeader->Op == OP_LOADMEMTSO || sourceHeader->Op == OP_LOADCONTEXT)) {
|
||||
// Load mem / load ctx zexts, no need to vmem
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(source));
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_SYSCALL: {
|
||||
auto Op = IROp->CW<IR::IROp_Syscall>();
|
||||
|
||||
// Is the first argument a constant?
|
||||
uint64_t Constant;
|
||||
if (IREmit->IsValueConstant(Op->SyscallID, &Constant)) {
|
||||
auto SyscallDef = Manager->SyscallHandler->GetSyscallABI(Constant);
|
||||
auto SyscallFlags = Manager->SyscallHandler->GetSyscallFlags(Constant);
|
||||
|
||||
// Update the syscall flags
|
||||
Op->Flags = SyscallFlags;
|
||||
|
||||
// XXX: Once we have the ability to do real function calls then we can call directly in to the syscall handler
|
||||
if (SyscallDef.NumArgs < FEXCore::HLE::SyscallArguments::MAX_ARGS) {
|
||||
// If the number of args are less than what the IR op supports then we can remove arg usage
|
||||
// We need +1 since we are still passing in syscall number here
|
||||
for (uint8_t Arg = (SyscallDef.NumArgs + 1); Arg < FEXCore::HLE::SyscallArguments::MAX_ARGS; ++Arg) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Arg, IREmit->Invalid());
|
||||
}
|
||||
// Replace syscall with inline passthrough syscall if we can
|
||||
if (SyscallDef.HostSyscallNumber != -1) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
// Skip Args[0] since that is the syscallid
|
||||
auto InlineSyscall =
|
||||
IREmit->_InlineSyscall(CurrentIR.GetNode(IROp->Args[1]), CurrentIR.GetNode(IROp->Args[2]), CurrentIR.GetNode(IROp->Args[3]),
|
||||
CurrentIR.GetNode(IROp->Args[4]), CurrentIR.GetNode(IROp->Args[5]), CurrentIR.GetNode(IROp->Args[6]),
|
||||
SyscallDef.HostSyscallNumber, Op->Flags);
|
||||
|
||||
// Replace all syscall uses with this inline one
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, InlineSyscall);
|
||||
|
||||
// We must remove here since DCE can't remove a IROp with sideeffects
|
||||
IREmit->Remove(CodeNode);
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_CPUID: {
|
||||
auto Op = IROp->CW<IR::IROp_CPUID>();
|
||||
|
||||
uint64_t ConstantFunction {}, ConstantLeaf {};
|
||||
bool IsConstantFunction = IREmit->IsValueConstant(Op->Function, &ConstantFunction);
|
||||
bool IsConstantLeaf = IREmit->IsValueConstant(Op->Leaf, &ConstantLeaf);
|
||||
// If the CPUID function is constant then we can try and optimize.
|
||||
if (IsConstantFunction) { // && ConstantFunction != 1) {
|
||||
// Check if it supports constant data reporting for this function.
|
||||
const auto SupportsConstant = CPUID->DoesFunctionReportConstantData(ConstantFunction);
|
||||
if (SupportsConstant.SupportsConstantFunction == CPUIDEmu::SupportsConstant::CONSTANT) {
|
||||
// If the CPUID needs a constant leaf to be optimized then this can't work if we didn't const-prop the leaf register.
|
||||
if (!(SupportsConstant.NeedsLeaf == CPUIDEmu::NeedsLeafConstant::NEEDSLEAFCONSTANT && !IsConstantLeaf)) {
|
||||
// Calculate the constant data and replace all uses.
|
||||
const auto Result = CPUID->RunFunction(ConstantFunction, ConstantLeaf);
|
||||
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
IREmit->ReplaceAllUsesWith(CurrentIR.GetNode(Op->OutEAX), IREmit->_Constant(Result.eax));
|
||||
IREmit->ReplaceAllUsesWith(CurrentIR.GetNode(Op->OutEBX), IREmit->_Constant(Result.ebx));
|
||||
IREmit->ReplaceAllUsesWith(CurrentIR.GetNode(Op->OutECX), IREmit->_Constant(Result.ecx));
|
||||
IREmit->ReplaceAllUsesWith(CurrentIR.GetNode(Op->OutEDX), IREmit->_Constant(Result.edx));
|
||||
IREmit->Remove(CodeNode);
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_XGETBV: {
|
||||
auto Op = IROp->CW<IR::IROp_XGetBV>();
|
||||
|
||||
uint64_t ConstantFunction {};
|
||||
if (IREmit->IsValueConstant(Op->Function, &ConstantFunction) && CPUID->DoesXCRFunctionReportConstantData(ConstantFunction)) {
|
||||
const auto Result = CPUID->RunXCRFunction(ConstantFunction);
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
IREmit->ReplaceAllUsesWith(CurrentIR.GetNode(Op->OutEAX), IREmit->_Constant(Result.eax));
|
||||
IREmit->ReplaceAllUsesWith(CurrentIR.GetNode(Op->OutEDX), IREmit->_Constant(Result.edx));
|
||||
IREmit->Remove(CodeNode);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_LDIV:
|
||||
case OP_LREM: {
|
||||
auto Op = IROp->C<IR::IROp_LDiv>();
|
||||
auto UpperIROp = IREmit->GetOpHeader(Op->Upper);
|
||||
|
||||
// Check upper Op to see if it came from a sign-extension
|
||||
if (UpperIROp->Op != OP_SBFE) {
|
||||
break;
|
||||
}
|
||||
|
||||
auto Sbfe = UpperIROp->C<IR::IROp_Sbfe>();
|
||||
if (Sbfe->Width != 1 || Sbfe->lsb != 63 || Sbfe->Header.Args[0] != Op->Lower) {
|
||||
break;
|
||||
}
|
||||
|
||||
// If it does then it we only need a 64bit SDIV
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
Ref Lower = CurrentIR.GetNode(Op->Lower);
|
||||
Ref Divisor = CurrentIR.GetNode(Op->Divisor);
|
||||
Ref SDivOp {};
|
||||
if (IROp->Op == OP_LDIV) {
|
||||
SDivOp = IREmit->_Div(OpSize::i64Bit, Lower, Divisor);
|
||||
} else {
|
||||
SDivOp = IREmit->_Rem(OpSize::i64Bit, Lower, Divisor);
|
||||
}
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, SDivOp);
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_LUDIV:
|
||||
case OP_LUREM: {
|
||||
auto Op = IROp->C<IR::IROp_LUDiv>();
|
||||
// Check upper Op to see if it came from a zeroing op
|
||||
// If it does then it we only need a 64bit UDIV
|
||||
uint64_t Value;
|
||||
if (!IREmit->IsValueConstant(Op->Upper, &Value) || Value != 0) {
|
||||
break;
|
||||
}
|
||||
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
Ref Lower = CurrentIR.GetNode(Op->Lower);
|
||||
Ref Divisor = CurrentIR.GetNode(Op->Divisor);
|
||||
Ref UDivOp {};
|
||||
if (IROp->Op == OP_LUDIV) {
|
||||
UDivOp = IREmit->_UDiv(OpSize::i64Bit, Lower, Divisor);
|
||||
} else {
|
||||
UDivOp = IREmit->_URem(OpSize::i64Bit, Lower, Divisor);
|
||||
}
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, UDivOp);
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_ADC:
|
||||
case OP_ADCWITHFLAGS:
|
||||
case OP_RMIFNZCV: {
|
||||
@@ -747,15 +327,75 @@ void ConstProp::Run(IREmitter* IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::ConstProp");
|
||||
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
const uint32_t SSACount = CurrentIR.GetSSACount();
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) {
|
||||
ConstantPropagation(IREmit, CurrentIR, CodeNode, IROp);
|
||||
// Allocation/initialization deferred until first use, since many multiblocks
|
||||
// don't have constants leftover after all inlining.
|
||||
fextl::vector<Ref> Remap {};
|
||||
|
||||
struct Entry {
|
||||
int64_t Value;
|
||||
Ref R;
|
||||
};
|
||||
|
||||
fextl::vector<Entry> Pool {};
|
||||
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
Pool.clear();
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
if (IROp->Op == OP_CONSTANT) {
|
||||
auto Op = IROp->C<IR::IROp_Constant>();
|
||||
bool Found = false;
|
||||
|
||||
// Search for the constant. This is O(n^2) but n is small since it's
|
||||
// local and most constants are inlined. In practice, it ends up much
|
||||
// faster than a hash table.
|
||||
for (auto K : Pool) {
|
||||
if (K.Value == Op->Constant) {
|
||||
uint32_t Value = CurrentIR.GetID(CodeNode).Value;
|
||||
if (Value < SSACount) {
|
||||
if (Remap.empty()) {
|
||||
Remap.resize(SSACount, nullptr);
|
||||
}
|
||||
|
||||
Remap[Value] = K.R;
|
||||
}
|
||||
Found = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (!Found) {
|
||||
Pool.push_back({.Value = Op->Constant, .R = CodeNode});
|
||||
}
|
||||
|
||||
continue;
|
||||
}
|
||||
|
||||
ConstantPropagation(IREmit, CurrentIR, CodeNode, IROp);
|
||||
|
||||
if (!Remap.empty()) {
|
||||
const uint8_t NumArgs = IR::GetArgs(IROp->Op);
|
||||
for (uint8_t i = 0; i < NumArgs; ++i) {
|
||||
if (IROp->Args[i].IsInvalid()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
uint32_t Value = IROp->Args[i].ID().Value;
|
||||
if (Value < SSACount) {
|
||||
Ref New = Remap[Value];
|
||||
if (New) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, i, New);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
HandleConstantPools(IREmit, IREmit->ViewIR());
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool SupportsTSOImm9, const FEXCore::CPUIDEmu* CPUID) {
|
||||
return fextl::make_unique<ConstProp>(SupportsTSOImm9, CPUID);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool SupportsTSOImm9) {
|
||||
return fextl::make_unique<ConstProp>(SupportsTSOImm9);
|
||||
}
|
||||
} // namespace FEXCore::IR
|
||||
@@ -38,12 +38,6 @@ IRDumper::IRDumper() {
|
||||
}
|
||||
|
||||
void IRDumper::Run(IREmitter* IREmit) {
|
||||
auto RAPass = Manager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
IR::RegisterAllocationData* RA {};
|
||||
if (RAPass) {
|
||||
RA = RAPass->GetAllocationData();
|
||||
}
|
||||
|
||||
FEXCore::File::File FD {};
|
||||
if (DumpIR() == "stderr") {
|
||||
FD = FEXCore::File::File::GetStdERR();
|
||||
@@ -57,18 +51,18 @@ void IRDumper::Run(IREmitter* IREmit) {
|
||||
|
||||
// DumpIRStr might be no if not dumping but ShouldDump is set in OpDisp
|
||||
if (DumpToFile) {
|
||||
const auto fileName = fextl::fmt::format("{}/{:x}{}", DumpIR(), HeaderOp->OriginalRIP, RA ? "-post.ir" : "-pre.ir");
|
||||
const auto fileName = fextl::fmt::format("{}/{:x}{}", DumpIR(), HeaderOp->OriginalRIP, IR.PostRA() ? "-post.ir" : "-pre.ir");
|
||||
FD = FEXCore::File::File(fileName.c_str(),
|
||||
FEXCore::File::FileModes::WRITE | FEXCore::File::FileModes::CREATE | FEXCore::File::FileModes::TRUNCATE);
|
||||
}
|
||||
|
||||
if (FD.IsValid() || DumpToLog) {
|
||||
fextl::stringstream out;
|
||||
FEXCore::IR::Dump(&out, &IR, RA);
|
||||
FEXCore::IR::Dump(&out, &IR);
|
||||
if (FD.IsValid()) {
|
||||
fextl::fmt::print(FD, "IR-{} 0x{:x}:\n{}\n@@@@@\n", RA ? "post" : "pre", HeaderOp->OriginalRIP, out.str());
|
||||
fextl::fmt::print(FD, "IR-{} 0x{:x}:\n{}\n@@@@@\n", IR.PostRA() ? "post" : "pre", HeaderOp->OriginalRIP, out.str());
|
||||
} else {
|
||||
LogMan::Msg::IFmt("IR-{} 0x{:x}:\n{}\n@@@@@\n", RA ? "post" : "pre", HeaderOp->OriginalRIP, out.str());
|
||||
LogMan::Msg::IFmt("IR-{} 0x{:x}:\n{}\n@@@@@\n", IR.PostRA() ? "post" : "pre", HeaderOp->OriginalRIP, out.str());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -58,11 +58,6 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
LOGMAN_THROW_A_FMT(HeaderOp->Header.Op == OP_IRHEADER, "First op wasn't IRHeader");
|
||||
#endif
|
||||
|
||||
IR::RegisterAllocationData* RAData {};
|
||||
if (Manager->HasPass("RA")) {
|
||||
RAData = Manager->GetPass<IR::RegisterAllocationPass>("RA")->GetAllocationData();
|
||||
}
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
@@ -94,9 +89,9 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
Warnings << "%" << ID << ": Destination created but had no uses" << std::endl;
|
||||
}
|
||||
|
||||
if (RAData) {
|
||||
// If we have a register allocator then the destination needs to be assigned a register and class
|
||||
auto PhyReg = RAData->GetNodeRegister(ID);
|
||||
if (CurrentIR.PostRA()) {
|
||||
// After RA, the destination needs to be assigned a register and class
|
||||
auto PhyReg = PhysicalRegister(CodeNode);
|
||||
|
||||
FEXCore::IR::RegisterClassType ExpectedClass = IR::GetRegClass(IROp->Op);
|
||||
FEXCore::IR::RegisterClassType AssignedClass = FEXCore::IR::RegisterClassType {PhyReg.Class};
|
||||
@@ -108,7 +103,7 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
}
|
||||
|
||||
// If no physical register was assigned
|
||||
if (PhyReg.Reg == IR::InvalidReg) {
|
||||
if (PhyReg.IsInvalid()) {
|
||||
HadError |= true;
|
||||
Errors << "%" << ID << ": Had destination but with no register assigned" << std::endl;
|
||||
}
|
||||
@@ -127,6 +122,10 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
for (uint32_t i = 0; i < NumArgs; ++i) {
|
||||
OrderedNodeWrapper Arg = IROp->Args[i];
|
||||
const auto ArgID = Arg.ID();
|
||||
if (Arg.IsImmediate()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
IROps Op = CurrentIR.GetOp<IROp_Header>(Arg)->Op;
|
||||
|
||||
if (ArgID.IsValid()) {
|
||||
@@ -239,18 +238,21 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
}
|
||||
}
|
||||
|
||||
for (uint32_t i = 0; i < CurrentIR.GetSSACount(); i++) {
|
||||
auto [Node, IROp] = CurrentIR.at(IR::NodeID {i})();
|
||||
if (Node->NumUses != Uses[i] && IROp->Op != OP_CODEBLOCK && IROp->Op != OP_IRHEADER) {
|
||||
HadError |= true;
|
||||
Errors << "%" << i << " Has " << Uses[i] << " Uses, but reports " << Node->NumUses << std::endl;
|
||||
// Use counts are only relevant pre-RA.
|
||||
if (!CurrentIR.PostRA()) {
|
||||
for (uint32_t i = 0; i < CurrentIR.GetSSACount(); i++) {
|
||||
auto [Node, IROp] = CurrentIR.at(IR::NodeID {i})();
|
||||
if (Node->NumUses != Uses[i] && IROp->Op != OP_CODEBLOCK && IROp->Op != OP_IRHEADER) {
|
||||
HadError |= true;
|
||||
Errors << "%" << i << " Has " << Uses[i] << " Uses, but reports " << Node->NumUses << std::endl;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
HadWarning = false;
|
||||
if (HadError || HadWarning) {
|
||||
fextl::stringstream Out;
|
||||
FEXCore::IR::Dump(&Out, &CurrentIR, RAData);
|
||||
FEXCore::IR::Dump(&Out, &CurrentIR);
|
||||
|
||||
if (HadError) {
|
||||
Out << "Errors:" << std::endl << Errors.str() << std::endl;
|
||||
|
||||
@@ -16,8 +16,6 @@ struct BlockInfo {
|
||||
fextl::vector<OrderedNode*> Successors;
|
||||
};
|
||||
|
||||
class RAValidation;
|
||||
|
||||
class IRValidation final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
~IRValidation();
|
||||
@@ -29,7 +27,5 @@ private:
|
||||
OrderedNode* EntryBlock {};
|
||||
fextl::unordered_map<IR::NodeID, BlockInfo> OffsetToBlockMap;
|
||||
size_t MaxNodes {};
|
||||
|
||||
friend class RAValidation;
|
||||
};
|
||||
} // namespace FEXCore::IR::Validation
|
||||
@@ -1,197 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "Interface/IR/RegisterAllocationData.h"
|
||||
#include "Interface/IR/Passes/IRValidation.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/fextl/deque.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
|
||||
#include <algorithm>
|
||||
|
||||
namespace FEXCore::IR::Validation {
|
||||
|
||||
// Hold the mapping of physical registers to the SSA id it holds at any given point in the IR
|
||||
struct RegState {
|
||||
static constexpr IR::NodeID UninitializedValue {0};
|
||||
static constexpr IR::NodeID InvalidReg {0xffff'ffff};
|
||||
|
||||
// This class makes some assumptions about how the host registers are arranged and mapped to virtual registers:
|
||||
// 1. There will be less than 32 GPRs and 32 FPRs
|
||||
// 2. If the GPRFixed class is used, there will be 16 GPRs and 16 FixedGPRs max
|
||||
// 3. Same with FPRFixed
|
||||
|
||||
// These assumptions were all true for the state of the arm64 and x86 jits at the time this was written
|
||||
|
||||
// Mark a physical register as containing a SSA id
|
||||
bool Set(PhysicalRegister Reg, IR::NodeID ssa) {
|
||||
LOGMAN_THROW_A_FMT(ssa.IsValid(), "RegState assumes ssa0 will be the block header and never assigned to a register");
|
||||
|
||||
// PhysicalRegisters aren't fully mapped until assembly emission
|
||||
// We need to apply a generic mapping here to catch any aliasing
|
||||
switch (Reg.Class) {
|
||||
case GPRClass: GPRs[Reg.Reg] = ssa; return true;
|
||||
case GPRFixedClass:
|
||||
// On arm64, there are 16 Fixed and 9 normal
|
||||
GPRsFixed[Reg.Reg] = ssa;
|
||||
return true;
|
||||
case FPRClass: FPRs[Reg.Reg] = ssa; return true;
|
||||
case FPRFixedClass:
|
||||
// On arm64, there are 16 Fixed and 12 normal
|
||||
FPRsFixed[Reg.Reg] = ssa;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
// Get the current SSA id
|
||||
// Or an error value there isn't a (sane) SSA id
|
||||
IR::NodeID Get(PhysicalRegister Reg) const {
|
||||
switch (Reg.Class) {
|
||||
case GPRClass: return GPRs[Reg.Reg];
|
||||
case GPRFixedClass: return GPRsFixed[Reg.Reg];
|
||||
case FPRClass: return FPRs[Reg.Reg];
|
||||
case FPRFixedClass: return FPRsFixed[Reg.Reg];
|
||||
}
|
||||
return InvalidReg;
|
||||
}
|
||||
|
||||
// Mark a spill slot as containing a SSA id
|
||||
void Spill(uint32_t SpillSlot, IR::NodeID ssa) {
|
||||
Spills[SpillSlot] = ssa;
|
||||
}
|
||||
|
||||
// Return the SSA id currently in a spill slot
|
||||
IR::NodeID Unspill(uint32_t SpillSlot) {
|
||||
if (Spills.contains(SpillSlot)) {
|
||||
return Spills[SpillSlot];
|
||||
} else {
|
||||
return UninitializedValue;
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
std::array<IR::NodeID, 32> GPRsFixed = {};
|
||||
std::array<IR::NodeID, 32> FPRsFixed = {};
|
||||
std::array<IR::NodeID, 32> GPRs = {};
|
||||
std::array<IR::NodeID, 32> FPRs = {};
|
||||
|
||||
fextl::unordered_map<uint32_t, IR::NodeID> Spills;
|
||||
};
|
||||
|
||||
class RAValidation final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
~RAValidation() {}
|
||||
void Run(IREmitter* IREmit) override;
|
||||
};
|
||||
|
||||
|
||||
void RAValidation::Run(IREmitter* IREmit) {
|
||||
if (!Manager->HasPass("RA")) {
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::RAValidation");
|
||||
|
||||
IR::RegisterAllocationData* RAData = Manager->GetPass<IR::RegisterAllocationPass>("RA")->GetAllocationData();
|
||||
|
||||
bool HadError = false;
|
||||
fextl::ostringstream Errors;
|
||||
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
// We only allocate registers locally, so state is reset each block
|
||||
struct RegState BlockRegState = {};
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
const auto ID = CurrentIR.GetID(CodeNode);
|
||||
|
||||
const auto CheckArg = [&](uint32_t i, OrderedNodeWrapper Arg) {
|
||||
const auto ArgID = Arg.ID();
|
||||
const auto PhyReg = RAData->GetNodeRegister(ArgID);
|
||||
|
||||
if (PhyReg.IsInvalid()) {
|
||||
return;
|
||||
}
|
||||
|
||||
auto CurrentSSAAtReg = BlockRegState.Get(PhyReg);
|
||||
if (CurrentSSAAtReg == RegState::InvalidReg) {
|
||||
HadError |= true;
|
||||
Errors << fextl::fmt::format("%{}: Arg[{}] unknown Reg: {}, class: {}\n", ID, i, PhyReg.Reg, PhyReg.Class);
|
||||
} else if (CurrentSSAAtReg == RegState::UninitializedValue) {
|
||||
HadError |= true;
|
||||
|
||||
Errors << fextl::fmt::format("%{}: Arg[{}] expects reg{} to contain %{}, but it is uninitialized\n", ID, i, PhyReg.Reg, ArgID);
|
||||
} else if (CurrentSSAAtReg != ArgID) {
|
||||
HadError |= true;
|
||||
Errors << fextl::fmt::format("%{}: Arg[{}] expects reg{} to contain %{}, but it actually contains %{}\n", ID, i, PhyReg.Reg,
|
||||
ArgID, CurrentSSAAtReg);
|
||||
}
|
||||
};
|
||||
|
||||
switch (IROp->Op) {
|
||||
case OP_SPILLREGISTER: {
|
||||
auto SpillRegister = IROp->C<IROp_SpillRegister>();
|
||||
CheckArg(0, SpillRegister->Value);
|
||||
|
||||
BlockRegState.Spill(SpillRegister->Slot, SpillRegister->Value.ID());
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_FILLREGISTER: {
|
||||
auto FillRegister = IROp->C<IROp_FillRegister>();
|
||||
const auto ExpectedValue = FillRegister->OriginalValue.ID();
|
||||
const auto Value = BlockRegState.Unspill(FillRegister->Slot);
|
||||
|
||||
// TODO: This only proves that the Spill has a consistent SSA value
|
||||
// In the future we need to prove it contains the correct SSA value. For
|
||||
// this we need to analyze copies/swaps properly. As a hot fix, don't
|
||||
// compare Value with ExpectedValue.
|
||||
|
||||
if (Value == RegState::UninitializedValue) {
|
||||
HadError |= true;
|
||||
Errors << fextl::fmt::format("%{}: FillRegister expected %{} in Slot {}, but was undefined\n", ID, ExpectedValue, FillRegister->Slot);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
default: {
|
||||
// And check that all args point at the correct SSA
|
||||
uint8_t NumArgs = IR::GetArgs(IROp->Op);
|
||||
for (uint32_t i = 0; i < NumArgs; ++i) {
|
||||
CheckArg(i, IROp->Args[i]);
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Update BlockState map
|
||||
if (IROp->Op != OP_SPILLREGISTER) {
|
||||
BlockRegState.Set(RAData->GetNodeRegister(ID), ID);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (HadError) {
|
||||
fextl::stringstream IrDump;
|
||||
FEXCore::IR::Dump(&IrDump, &CurrentIR, RAData);
|
||||
|
||||
LogMan::Msg::EFmt("RA Validation Error\n{}\nErrors:\n{}\n", IrDump.str(), Errors.str());
|
||||
LOGMAN_MSG_A_FMT("Encountered RA validation Error");
|
||||
|
||||
Errors.clear();
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateRAValidation() {
|
||||
return fextl::make_unique<RAValidation>();
|
||||
}
|
||||
} // namespace FEXCore::IR::Validation
|
||||
@@ -97,46 +97,49 @@ private:
|
||||
};
|
||||
|
||||
struct BlockInfo {
|
||||
fextl::vector<Ref> Predecessors;
|
||||
fextl::vector<uint32_t> Predecessors;
|
||||
Ref Node;
|
||||
uint8_t Flags;
|
||||
bool InWorklist;
|
||||
};
|
||||
|
||||
struct ControlFlowGraph {
|
||||
fextl::unordered_map<uint32_t, BlockInfo> BlockMap;
|
||||
fextl::vector<BlockInfo> BlockMap;
|
||||
IRListView& IR;
|
||||
|
||||
void AddBlock(fextl::deque<Ref>& Worklist, Ref Block) {
|
||||
uint32_t ID = IR.GetID(Block).Value;
|
||||
void Init(fextl::deque<uint32_t>& Worklist, uint32_t BlockCount) {
|
||||
BlockMap.resize(BlockCount);
|
||||
|
||||
// Add the block with conservative flags and already in the worklist.
|
||||
auto Info = &BlockMap.emplace(ID, BlockInfo {{}, FLAG_ALL, true}).first->second;
|
||||
for (unsigned ID = 0; ID < BlockCount; ++ID) {
|
||||
// Add the block with conservative flags and already in the worklist.
|
||||
auto Info = BlockInfo {{}, nullptr, FLAG_ALL, true};
|
||||
|
||||
// Add some initial capacity
|
||||
Info->Predecessors.reserve(2);
|
||||
// Add some initial capacity
|
||||
Info.Predecessors.reserve(2);
|
||||
|
||||
// Add to worklist
|
||||
Worklist.push_back(Block);
|
||||
BlockMap[ID] = std::move(Info);
|
||||
Worklist.push_back(ID);
|
||||
}
|
||||
}
|
||||
|
||||
BlockInfo* Get(uint32_t Block) {
|
||||
return &BlockMap.try_emplace(Block).first->second;
|
||||
return &BlockMap[Block];
|
||||
}
|
||||
|
||||
BlockInfo* Get(Ref Block) {
|
||||
return Get(IR.GetID(Block).Value);
|
||||
BlockInfo* Get(IROp_CodeBlock* Block) {
|
||||
return &BlockMap[Block->ID];
|
||||
}
|
||||
|
||||
BlockInfo* Get(OrderedNodeWrapper Block) {
|
||||
return Get(Block.ID().Value);
|
||||
return Get(IR.GetOp<IR::IROp_CodeBlock>(Block));
|
||||
}
|
||||
|
||||
void RecordEdge(Ref From, Ref To) {
|
||||
void RecordEdge(uint32_t From, OrderedNodeWrapper To) {
|
||||
auto Info = Get(To);
|
||||
Info->Predecessors.push_back(From);
|
||||
}
|
||||
|
||||
void AddWorklist(fextl::deque<Ref>& Worklist, Ref Block) {
|
||||
void AddWorklist(fextl::deque<uint32_t>& Worklist, uint32_t Block) {
|
||||
auto Info = Get(Block);
|
||||
if (!Info->InWorklist) {
|
||||
Info->InWorklist = true;
|
||||
@@ -637,8 +640,8 @@ bool DeadFlagCalculationEliminination::ProcessBlock(IREmitter* IREmit, IRListVie
|
||||
// For the purposes of global propagation, the content of our progress doesn't
|
||||
// matter -- only the difference in our final FlagsRead contributes to changes
|
||||
// in the predecessors.
|
||||
uint32_t OldFlagsRead = CFG.Get(Block)->Flags;
|
||||
CFG.Get(Block)->Flags = FlagsRead;
|
||||
uint32_t OldFlagsRead = CFG.Get(BlockIROp->ID)->Flags;
|
||||
CFG.Get(BlockIROp->ID)->Flags = FlagsRead;
|
||||
return (OldFlagsRead != FlagsRead);
|
||||
}
|
||||
|
||||
@@ -650,12 +653,14 @@ void DeadFlagCalculationEliminination::OptimizeParity(IREmitter* IREmit, IRListV
|
||||
// Initialize conservatively: all blocks need full parity. This initialization
|
||||
// matters for proper handling of backedges.
|
||||
for (auto [Block, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
CFG.Get(Block)->Flags = FULL;
|
||||
auto ID = BlockHeader->C<IROp_CodeBlock>()->ID;
|
||||
CFG.Get(ID)->Flags = FULL;
|
||||
}
|
||||
|
||||
for (auto [Block, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
auto ID = BlockHeader->C<IROp_CodeBlock>()->ID;
|
||||
bool Full = false;
|
||||
auto Predecessors = CFG.Get(Block)->Predecessors;
|
||||
auto Predecessors = CFG.Get(ID)->Predecessors;
|
||||
|
||||
if (Predecessors.empty()) {
|
||||
// Conservatively assume there was full parity before the start block
|
||||
@@ -701,7 +706,7 @@ void DeadFlagCalculationEliminination::OptimizeParity(IREmitter* IREmit, IRListV
|
||||
}
|
||||
|
||||
// Record our final state for our successors to read.
|
||||
CFG.Get(Block)->Flags = Full ? FULL : PARTIAL;
|
||||
CFG.Get(ID)->Flags = Full ? FULL : PARTIAL;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -709,28 +714,28 @@ void DeadFlagCalculationEliminination::Run(IREmitter* IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::DFE");
|
||||
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
fextl::deque<Ref> Worklist;
|
||||
fextl::deque<uint32_t> Worklist;
|
||||
|
||||
// Initialize CFG
|
||||
ControlFlowGraph CFG {.IR = CurrentIR};
|
||||
|
||||
// Gather blocks
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
CFG.AddBlock(Worklist, BlockNode);
|
||||
}
|
||||
CFG.Init(Worklist, CurrentIR.GetHeader()->BlockCount);
|
||||
|
||||
// Gather CFG
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
auto CodeLast = CurrentIR.at(BlockHeader->C<IROp_CodeBlock>()->Last);
|
||||
auto Block = BlockHeader->C<IROp_CodeBlock>();
|
||||
auto CodeLast = CurrentIR.at(Block->Last);
|
||||
--CodeLast;
|
||||
auto [ExitNode, ExitOp] = CodeLast();
|
||||
if (ExitOp->Op == IR::OP_CONDJUMP) {
|
||||
auto Op = ExitOp->CW<IR::IROp_CondJump>();
|
||||
|
||||
CFG.RecordEdge(BlockNode, CurrentIR.GetNode(Op->TrueBlock));
|
||||
CFG.RecordEdge(BlockNode, CurrentIR.GetNode(Op->FalseBlock));
|
||||
CFG.RecordEdge(Block->ID, Op->TrueBlock);
|
||||
CFG.RecordEdge(Block->ID, Op->FalseBlock);
|
||||
} else if (ExitOp->Op == IR::OP_JUMP) {
|
||||
CFG.RecordEdge(BlockNode, CurrentIR.GetNode(ExitOp->Args[0]));
|
||||
CFG.RecordEdge(Block->ID, ExitOp->Args[0]);
|
||||
}
|
||||
|
||||
CFG.Get(Block->ID)->Node = BlockNode;
|
||||
}
|
||||
|
||||
// After processing a block, if we made progress, we must process its
|
||||
@@ -741,7 +746,7 @@ void DeadFlagCalculationEliminination::Run(IREmitter* IREmit) {
|
||||
auto Info = CFG.Get(Block);
|
||||
Info->InWorklist = false;
|
||||
|
||||
if (ProcessBlock(IREmit, CurrentIR, Block, CFG)) {
|
||||
if (ProcessBlock(IREmit, CurrentIR, Info->Node, CFG)) {
|
||||
for (auto Pred : Info->Predecessors) {
|
||||
CFG.AddWorklist(Worklist, Pred);
|
||||
}
|
||||
|
||||
@@ -10,6 +10,7 @@ $end_info$
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/RegisterAllocationData.h"
|
||||
#include "Interface/IR/Passes.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
@@ -21,16 +22,13 @@ using namespace FEXCore;
|
||||
|
||||
namespace FEXCore::IR {
|
||||
namespace {
|
||||
[[maybe_unused]] constexpr uint32_t INVALID_REG = IR::InvalidReg;
|
||||
constexpr uint32_t INVALID_CLASS = IR::InvalidClass.Val;
|
||||
|
||||
struct RegisterClass {
|
||||
uint32_t Available;
|
||||
uint32_t Count;
|
||||
|
||||
// If bit R of Available is 0, then RegToSSA[R] is the Old node
|
||||
// currently allocated to R. Else, RegToSSA[R] is UNDEFINED, no need to
|
||||
// clear this when freeing registers.
|
||||
// If bit R of Available is 0, then RegToSSA[R] is the node currently
|
||||
// allocated to R. Else, RegToSSA[R] is UNDEFINED, no need to clear this
|
||||
// when freeing registers.
|
||||
Ref RegToSSA[32];
|
||||
};
|
||||
|
||||
@@ -55,102 +53,34 @@ namespace {
|
||||
|
||||
class ConstrainedRAPass final : public RegisterAllocationPass {
|
||||
public:
|
||||
explicit ConstrainedRAPass(const FEXCore::CPUIDEmu* CPUID)
|
||||
: CPUID {CPUID} {}
|
||||
void Run(IREmitter* IREmit) override;
|
||||
void AddRegisters(IR::RegisterClassType Class, uint32_t RegisterCount) override;
|
||||
|
||||
RegisterAllocationData* GetAllocationData() override;
|
||||
RegisterAllocationData::UniquePtr PullAllocationData() override;
|
||||
bool TryPostRAMerge(Ref LastNode, Ref CodeNode, IROp_Header* IROp);
|
||||
|
||||
private:
|
||||
IR::RegisterAllocationData::UniquePtr AllocData;
|
||||
RegisterClass Classes[INVALID_CLASS];
|
||||
RegisterClass Classes[IR::NumClasses];
|
||||
|
||||
IREmitter* IREmit;
|
||||
IRListView* IR;
|
||||
const FEXCore::CPUIDEmu* CPUID;
|
||||
|
||||
// Map of Old nodes to their preferred register, to coalesce load/store reg.
|
||||
// Map of nodes to their preferred register, to coalesce load/store reg.
|
||||
fextl::vector<PhysicalRegister> PreferredReg;
|
||||
|
||||
// FEX's original RA could only assign a single register to a given def for
|
||||
// its entire live range, and this limitation is baked deep into the IR.
|
||||
// However, we split live ranges to implement register pairs and spilling.
|
||||
//
|
||||
// To reconcile, we generate new SSA nodes when we split live ranges, and
|
||||
// remap SSA sources accordingly. This means SSAToReg can grow.
|
||||
//
|
||||
// We define "Old" nodes as nodes present in the original IR, and "New" nodes
|
||||
// as nodes added to split live ranges. Helpful properties:
|
||||
//
|
||||
// - A node is Old <===> it is not New
|
||||
// - A node is Old <===> its ID < IR.GetSSACount() at the start
|
||||
// - All sources are Old before remapping an instruction
|
||||
//
|
||||
// SSAToNewSSA tracks the current remapping. nullptr indicates no remapping.
|
||||
//
|
||||
// Since its indexed by Old nodes, SSAToNewSSA does not grow after allocation.
|
||||
fextl::vector<Ref> SSAToNewSSA;
|
||||
|
||||
// Inverse of SSAToNewSSA. Since it's indexed by new nodes, it grows.
|
||||
fextl::vector<Ref> NewSSAToSSA;
|
||||
|
||||
// Map of assigned registers. Grows.
|
||||
// Map of assigned registers. Does not grow beyond the initial set.
|
||||
fextl::vector<PhysicalRegister> SSAToReg;
|
||||
|
||||
bool IsOld(Ref Node) {
|
||||
return IR->GetID(Node).Value < PreferredReg.size();
|
||||
};
|
||||
|
||||
// Return the New node (if it exists) for an Old node, else the Old node.
|
||||
Ref Map(Ref Old) {
|
||||
LOGMAN_THROW_A_FMT(IsOld(Old), "Pre-condition");
|
||||
|
||||
if (SSAToNewSSA.empty()) {
|
||||
return Old;
|
||||
} else {
|
||||
return SSAToNewSSA[IR->GetID(Old).Value] ?: Old;
|
||||
}
|
||||
};
|
||||
|
||||
// Return the Old node for a possibly-remapped node.
|
||||
Ref Unmap(Ref Node) {
|
||||
if (NewSSAToSSA.empty()) {
|
||||
return Node;
|
||||
} else {
|
||||
return NewSSAToSSA[IR->GetID(Node).Value] ?: Node;
|
||||
}
|
||||
};
|
||||
|
||||
// Record a remapping of Old to New.
|
||||
void Remap(Ref Old, Ref New) {
|
||||
LOGMAN_THROW_A_FMT(IsOld(Old) && !IsOld(New), "Pre-condition");
|
||||
|
||||
uint32_t OldID = IR->GetID(Old).Value;
|
||||
uint32_t NewID = IR->GetID(New).Value;
|
||||
|
||||
LOGMAN_THROW_A_FMT(NewID >= NewSSAToSSA.size(), "Brand new SSA def");
|
||||
NewSSAToSSA.resize(NewID + 1, 0);
|
||||
|
||||
if (SSAToNewSSA.empty()) {
|
||||
SSAToNewSSA.resize(PreferredReg.size(), nullptr);
|
||||
}
|
||||
|
||||
SSAToNewSSA[OldID] = New;
|
||||
NewSSAToSSA[NewID] = Old;
|
||||
|
||||
LOGMAN_THROW_A_FMT(Map(Old) == New && Unmap(New) == Old, "Post-condition");
|
||||
LOGMAN_THROW_A_FMT(Unmap(Old) == Old, "Invariant1");
|
||||
};
|
||||
|
||||
// Maps Old defs to their assigned spill slot + 1, or 0 if not spilled.
|
||||
// Maps defs to their assigned spill slot + 1, or 0 if not spilled.
|
||||
fextl::vector<unsigned> SpillSlots;
|
||||
|
||||
bool Rematerializable(IROp_Header* IROp) {
|
||||
return IROp->Op == OP_CONSTANT;
|
||||
}
|
||||
|
||||
Ref InsertFill(Ref Old) {
|
||||
LOGMAN_THROW_A_FMT(IsOld(Old), "Precondition");
|
||||
IROp_Header* IROp = IR->GetOp<IROp_Header>(Old);
|
||||
Ref InsertFill(Ref Node) {
|
||||
IROp_Header* IROp = IR->GetOp<IROp_Header>(Node);
|
||||
|
||||
// Remat if we can
|
||||
if (Rematerializable(IROp)) {
|
||||
@@ -159,22 +89,17 @@ private:
|
||||
}
|
||||
|
||||
// Otherwise fill from stack
|
||||
uint32_t SlotPlusOne = SpillSlots[IR->GetID(Old).Value];
|
||||
LOGMAN_THROW_A_FMT(SlotPlusOne >= 1, "Old must have been spilled");
|
||||
uint32_t SlotPlusOne = SpillSlots[IR->GetID(Node).Value];
|
||||
LOGMAN_THROW_A_FMT(SlotPlusOne >= 1, "Node must have been spilled");
|
||||
|
||||
RegisterClassType RegClass = GetRegClassFromNode(IR, IROp);
|
||||
|
||||
auto Fill = IREmit->_FillRegister(Old, SlotPlusOne - 1, RegClass);
|
||||
Fill.first->Header.Size = IROp->Size;
|
||||
Fill.first->Header.ElementSize = IROp->ElementSize;
|
||||
return Fill;
|
||||
return IREmit->_FillRegister(IROp->Size, IROp->ElementSize, SlotPlusOne - 1, RegClass);
|
||||
};
|
||||
|
||||
// IP of next-use of each Old source. IPs are measured from the end of the
|
||||
// IP of next-use of each source. IPs are measured from the end of the
|
||||
// block, so we don't need to size the block up-front.
|
||||
fextl::vector<uint32_t> NextUses;
|
||||
|
||||
unsigned SpillSlotCount;
|
||||
bool AnySpilled;
|
||||
|
||||
bool IsValidArg(OrderedNodeWrapper Arg) {
|
||||
@@ -182,15 +107,8 @@ private:
|
||||
return false;
|
||||
}
|
||||
|
||||
switch (IR->GetOp<IROp_Header>(Arg)->Op) {
|
||||
case OP_INLINECONSTANT:
|
||||
case OP_INLINEENTRYPOINTOFFSET:
|
||||
case OP_IRHEADER: return false;
|
||||
|
||||
case OP_SPILLREGISTER: LOGMAN_MSG_A_FMT("should not be seen"); return false;
|
||||
|
||||
default: return true;
|
||||
}
|
||||
auto Op = IR->GetOp<IROp_Header>(Arg)->Op;
|
||||
return Op != OP_INLINECONSTANT && Op != OP_INLINEENTRYPOINTOFFSET;
|
||||
};
|
||||
|
||||
RegisterClass* GetClass(PhysicalRegister Reg) {
|
||||
@@ -201,13 +119,14 @@ private:
|
||||
return 1 << Reg.Reg;
|
||||
};
|
||||
|
||||
bool IsInRegisterFile(Ref Old) {
|
||||
LOGMAN_THROW_A_FMT(IsOld(Old), "Precondition");
|
||||
bool IsInRegisterFile(Ref Node) {
|
||||
auto ID = IR->GetID(Node).Value;
|
||||
LOGMAN_THROW_A_FMT(ID < SSAToReg.size(), "Only old nodes looked up");
|
||||
|
||||
PhysicalRegister Reg = SSAToReg[IR->GetID(Map(Old)).Value];
|
||||
PhysicalRegister Reg = SSAToReg[ID];
|
||||
RegisterClass* Class = GetClass(Reg);
|
||||
|
||||
return (Class->Available & GetRegBits(Reg)) == 0 && Class->RegToSSA[Reg.Reg] == Old;
|
||||
return (Class->Available & GetRegBits(Reg)) == 0 && Class->RegToSSA[Reg.Reg] == Node;
|
||||
};
|
||||
|
||||
void FreeReg(PhysicalRegister Reg) {
|
||||
@@ -219,14 +138,9 @@ private:
|
||||
Class->Available |= RegBits;
|
||||
};
|
||||
|
||||
bool HasSource(IROp_Header* I, Ref Old) {
|
||||
LOGMAN_THROW_A_FMT(IsOld(Old), "Invariant2");
|
||||
|
||||
bool HasSource(IROp_Header* I, PhysicalRegister Reg) {
|
||||
for (auto s = 0; s < IR::GetRAArgs(I->Op); ++s) {
|
||||
Ref Node = IR->GetNode(I->Args[s]);
|
||||
LOGMAN_THROW_A_FMT(IsOld(Node), "not yet mapped");
|
||||
|
||||
if (Node == Old) {
|
||||
if (I->Args[s].IsImmediate() && PhysicalRegister(I->Args[s]) == Reg) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
@@ -248,36 +162,39 @@ private:
|
||||
return nullptr;
|
||||
};
|
||||
|
||||
PhysicalRegister DecodeSRAReg(const IROp_Header* IROp) {
|
||||
RegisterClassType Class {};
|
||||
uint8_t Reg {};
|
||||
|
||||
PhysicalRegister DecodeSRAReg(const IROp_Header* IROp, Ref Node) {
|
||||
uint8_t FlagOffset = Classes[GPRFixedClass.Val].Count - 2;
|
||||
|
||||
if (IROp->Op == OP_LOADREGISTER) {
|
||||
const IROp_LoadRegister* Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
|
||||
Class = Op->Class;
|
||||
Reg = Op->Reg;
|
||||
} else if (IROp->Op == OP_STOREREGISTER) {
|
||||
const IROp_StoreRegister* Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
|
||||
Class = Op->Class;
|
||||
Reg = Op->Reg;
|
||||
if (IROp->Op == OP_STOREREGISTER) {
|
||||
return PhysicalRegister(Node);
|
||||
} else if (IROp->Op == OP_LOADPF || IROp->Op == OP_STOREPF) {
|
||||
return PhysicalRegister {GPRFixedClass, FlagOffset};
|
||||
} else if (IROp->Op == OP_LOADAF || IROp->Op == OP_STOREAF) {
|
||||
return PhysicalRegister {GPRFixedClass, (uint8_t)(FlagOffset + 1)};
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Class == GPRClass || Class == FPRClass, "SRA classes");
|
||||
if (Class == FPRClass) {
|
||||
return PhysicalRegister {FPRFixedClass, Reg};
|
||||
} else {
|
||||
return PhysicalRegister {GPRFixedClass, Reg};
|
||||
const IROp_LoadRegister* Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
|
||||
LOGMAN_THROW_A_FMT(Op->Class == GPRClass || Op->Class == FPRClass, "SRA classes");
|
||||
if (Op->Class == FPRClass) {
|
||||
return PhysicalRegister {FPRFixedClass, (uint8_t)Op->Reg};
|
||||
} else {
|
||||
return PhysicalRegister {GPRFixedClass, (uint8_t)Op->Reg};
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
bool IsTrivial(Ref Node, const IROp_Header* Header) {
|
||||
switch (Header->Op) {
|
||||
case OP_ALLOCATEGPR: return true;
|
||||
case OP_ALLOCATEGPRAFTER: return true;
|
||||
case OP_ALLOCATEFPR: return true;
|
||||
case OP_RMWHANDLE: return PhysicalRegister(Node) == PhysicalRegister(Header->Args[0]);
|
||||
case OP_LOADREGISTER: return PhysicalRegister(Node) == DecodeSRAReg(Header, Node);
|
||||
case OP_STOREREGISTER: return PhysicalRegister(Header->Args[0]) == DecodeSRAReg(Header, Node);
|
||||
default: return false;
|
||||
}
|
||||
}
|
||||
|
||||
// Helper macro to walk the set bits b in a 32-bit word x, using ffs to get
|
||||
// the next set bit and then clearing on each iteration.
|
||||
#define foreach_bit(b, x) for (uint32_t __x = (x), b; ((b) = __builtin_ffs(__x) - 1, __x); __x &= ~(1 << (b)))
|
||||
@@ -292,33 +209,33 @@ private:
|
||||
uint32_t Allocated = ((1u << Class->Count) - 1) & ~Class->Available;
|
||||
|
||||
foreach_bit(i, Allocated) {
|
||||
Ref Old = Class->RegToSSA[i];
|
||||
Ref Node = Class->RegToSSA[i];
|
||||
auto Reg = SSAToReg[IR->GetID(Node).Value];
|
||||
|
||||
LOGMAN_THROW_A_FMT(Old != nullptr, "Invariant3");
|
||||
LOGMAN_THROW_A_FMT(SSAToReg[IR->GetID(Map(Old)).Value].Reg == i, "Invariant4");
|
||||
LOGMAN_THROW_A_FMT(Node != nullptr, "Invariant3");
|
||||
LOGMAN_THROW_A_FMT(Reg.Reg == i, "Invariant4");
|
||||
|
||||
// Skip any source used by the current instruction, it is unspillable.
|
||||
if (!HasSource(Exclude, Old)) {
|
||||
uint32_t NextUse = NextUses[IR->GetID(Old).Value];
|
||||
if (!HasSource(Exclude, Reg)) {
|
||||
uint32_t NextUse = NextUses[IR->GetID(Node).Value];
|
||||
|
||||
// Prioritize remat over spilling. It is typically cheaper to remat a
|
||||
// constant multiple times than to spill a single value.
|
||||
if (!Rematerializable(IR->GetOp<IROp_Header>(Old))) {
|
||||
if (!Rematerializable(IR->GetOp<IROp_Header>(Node))) {
|
||||
NextUse += 100000;
|
||||
}
|
||||
|
||||
if (NextUse < BestDistance) {
|
||||
BestDistance = NextUse;
|
||||
BestReg = i;
|
||||
Candidate = Old;
|
||||
Candidate = Node;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Candidate != nullptr, "must've found something..");
|
||||
LOGMAN_THROW_A_FMT(IsOld(Candidate), "Invariant5");
|
||||
|
||||
PhysicalRegister Reg = SSAToReg[IR->GetID(Map(Candidate)).Value];
|
||||
PhysicalRegister Reg = SSAToReg[IR->GetID(Candidate).Value];
|
||||
LOGMAN_THROW_A_FMT(Reg.Reg == BestReg, "Invariant6");
|
||||
|
||||
IROp_Header* Header = IR->GetOp<IROp_Header>(Candidate);
|
||||
@@ -336,10 +253,10 @@ private:
|
||||
}
|
||||
|
||||
// TODO: we should colour spill slots
|
||||
uint32_t Slot = SpillSlotCount++;
|
||||
uint32_t Slot = IR->GetHeader()->SpillSlots++;
|
||||
|
||||
// We must map here in case we're spilling something we shuffled.
|
||||
auto SpillOp = IREmit->_SpillRegister(Map(Candidate), Slot, RegisterClassType {Reg.Class});
|
||||
auto SpillOp = IREmit->_SpillRegister(OrderedNodeWrapper::FromImmediate(Reg.Raw), Slot, RegisterClassType {Reg.Class});
|
||||
SpillOp.first->Header.Size = Header->Size;
|
||||
SpillOp.first->Header.ElementSize = Header->ElementSize;
|
||||
SpillSlots[Value] = Slot + 1;
|
||||
@@ -350,22 +267,27 @@ private:
|
||||
AnySpilled = true;
|
||||
};
|
||||
|
||||
void RemapReg(Ref Node, PhysicalRegister Reg) {
|
||||
RegisterClass* Class = GetClass(Reg);
|
||||
Class->RegToSSA[Reg.Reg] = Node;
|
||||
|
||||
uint32_t Index = IR->GetID(Node).Value;
|
||||
if (Index < SSAToReg.size()) {
|
||||
SSAToReg[Index] = Reg;
|
||||
}
|
||||
};
|
||||
|
||||
// Record a given assignment of register Reg to Node.
|
||||
void SetReg(Ref Node, PhysicalRegister Reg) {
|
||||
uint32_t Index = IR->GetID(Node).Value;
|
||||
RegisterClass* Class = GetClass(Reg);
|
||||
uint32_t RegBits = GetRegBits(Reg);
|
||||
|
||||
LOGMAN_THROW_A_FMT((Class->Available & RegBits) == RegBits, "Precondition");
|
||||
|
||||
Class->Available &= ~RegBits;
|
||||
Class->RegToSSA[Reg.Reg] = Unmap(Node);
|
||||
|
||||
if (Index >= SSAToReg.size()) {
|
||||
SSAToReg.resize(Index + 1, PhysicalRegister::Invalid());
|
||||
}
|
||||
|
||||
SSAToReg[Index] = Reg;
|
||||
RemapReg(Node, Reg);
|
||||
Node->Reg = Reg.Raw;
|
||||
};
|
||||
|
||||
// Assign a register for a given Node, spilling if necessary.
|
||||
@@ -387,7 +309,7 @@ private:
|
||||
|
||||
// Try to handle tied registers. This can fail, the JIT will insert moves.
|
||||
if (int TiedIdx = IR::TiedSource(IROp->Op); TiedIdx >= 0) {
|
||||
PhysicalRegister Reg = SSAToReg[IROp->Args[TiedIdx].ID().Value];
|
||||
auto Reg = PhysicalRegister(IROp->Args[TiedIdx]);
|
||||
RegisterClass* Class = GetClass(Reg);
|
||||
uint32_t RegBits = GetRegBits(Reg);
|
||||
|
||||
@@ -417,7 +339,7 @@ private:
|
||||
}
|
||||
} else if (IROp->Op == OP_ALLOCATEGPRAFTER) {
|
||||
uint32_t Available = Classes[GPRClass].Available;
|
||||
auto After = SSAToReg[IR->GetID(IR->GetNode(IROp->Args[0])).Value];
|
||||
auto After = PhysicalRegister(IROp->Args[0]);
|
||||
if ((After.Reg & 1) == 0 && Available & (1ull << (After.Reg + 1))) {
|
||||
SetReg(CodeNode, PhysicalRegister(GPRClass, After.Reg + 1));
|
||||
return;
|
||||
@@ -438,24 +360,134 @@ private:
|
||||
unsigned Reg = std::countr_zero(Class->Available);
|
||||
SetReg(CodeNode, PhysicalRegister(ClassType, Reg));
|
||||
};
|
||||
|
||||
bool IsRAOp(IROps Op) {
|
||||
return Op == OP_SPILLREGISTER || Op == OP_FILLREGISTER || Op == OP_COPY;
|
||||
};
|
||||
};
|
||||
|
||||
void ConstrainedRAPass::AddRegisters(IR::RegisterClassType Class, uint32_t RegisterCount) {
|
||||
LOGMAN_THROW_A_FMT(RegisterCount <= INVALID_REG, "Up to {} regs supported", INVALID_REG);
|
||||
LOGMAN_THROW_A_FMT(RegisterCount <= 31, "Up to 31 regs supported");
|
||||
|
||||
Classes[Class].Count = RegisterCount;
|
||||
}
|
||||
|
||||
RegisterAllocationData* ConstrainedRAPass::GetAllocationData() {
|
||||
return AllocData.get();
|
||||
inline bool KillMove(IROp_Header* LastOp, IROp_Header* IROp, Ref LastNode, Ref CodeNode) {
|
||||
// 32-bit moves in x86_64 are represented as a Bfe, detect them.
|
||||
if (LastOp->Op == OP_BFE && LastOp->C<IR::IROp_Bfe>()->lsb == 0 && LastOp->C<IR::IROp_Bfe>()->Width == 32) {
|
||||
auto Op = IROp->Op;
|
||||
|
||||
if (Op == OP_AND) {
|
||||
// Rewrite "mov wA, wB; and xA, xA, xC" into "and wA, wB, wC", since
|
||||
// ((b & 0xffffffff) & c) == (b & c) & 0xffffffff.
|
||||
IROp->Size = OpSize::i32Bit;
|
||||
return true;
|
||||
} else if (IROp->Size == OpSize::i32Bit) {
|
||||
return Op == OP_OR || Op == OP_XOR || Op == OP_AND || Op == OP_SUB || Op == OP_LSHL || Op == OP_LSHR || Op == OP_ASHR;
|
||||
}
|
||||
}
|
||||
|
||||
return LastOp->Op == OP_STOREREGISTER;
|
||||
}
|
||||
|
||||
RegisterAllocationData::UniquePtr ConstrainedRAPass::PullAllocationData() {
|
||||
return std::move(AllocData);
|
||||
inline bool IsSignext(const IROp_Header* IROp, OrderedNodeWrapper Src, OpSize Size) {
|
||||
if (IROp->Op == OP_SBFE) {
|
||||
auto Sbfe = IROp->C<IR::IROp_Sbfe>();
|
||||
return Sbfe->Width == 1 && Sbfe->lsb == (IR::OpSizeAsBits(Size) - 1) && Sbfe->Src == Src;
|
||||
} else {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
inline bool IsZero(const IROp_Header* IROp) {
|
||||
return IROp->Op == OP_CONSTANT && IROp->C<IROp_Constant>()->Constant == 0;
|
||||
}
|
||||
|
||||
bool ConstrainedRAPass::TryPostRAMerge(Ref LastNode, Ref CodeNode, IROp_Header* IROp) {
|
||||
auto LastOp = IR->GetOp<IROp_Header>(LastNode);
|
||||
|
||||
if (IROp->Op == OP_PUSH && LastOp->Op == OP_PUSH) {
|
||||
auto SP = PhysicalRegister(CodeNode);
|
||||
auto Push = IR->GetOp<IROp_Push>(CodeNode);
|
||||
auto LastPush = IR->GetOp<IROp_Push>(LastNode);
|
||||
|
||||
if (LastOp->Size == IROp->Size && LastPush->ValueSize == Push->ValueSize && SP == PhysicalRegister(LastNode) &&
|
||||
SP == PhysicalRegister(IROp->Args[1]) && SP == PhysicalRegister(LastOp->Args[1]) && SP != PhysicalRegister(IROp->Args[0]) &&
|
||||
SP != PhysicalRegister(LastOp->Args[0]) && Push->ValueSize >= OpSize::i32Bit) {
|
||||
|
||||
IREmit->SetWriteCursorBefore(LastNode);
|
||||
IREmit->_PushTwo(IROp->Size, Push->ValueSize, IROp->Args[0], LastOp->Args[0], IROp->Args[1]);
|
||||
IREmit->RemovePostRA(CodeNode);
|
||||
return true;
|
||||
}
|
||||
} else if (IROp->Op == OP_POP) {
|
||||
auto SP = PhysicalRegister(IROp->Args[0]);
|
||||
|
||||
if (LastOp->Op == OP_POP && LastOp->Size == IROp->Size && IROp->Size >= OpSize::i32Bit && SP == PhysicalRegister(LastOp->Args[0])) {
|
||||
IREmit->SetWriteCursorBefore(LastNode);
|
||||
IREmit->_PopTwo(IROp->Size, IROp->Args[0], LastOp->Args[1], IROp->Args[1]);
|
||||
IREmit->RemovePostRA(CodeNode);
|
||||
return true;
|
||||
}
|
||||
} else if ((IROp->Op == OP_DIV || IROp->Op == OP_UDIV) && IROp->Size >= OpSize::i32Bit) {
|
||||
// If Upper came from a sign/zero extension, we only need a 64-bit division.
|
||||
auto Op = IROp->CW<IR::IROp_Div>();
|
||||
if (!Op->Upper.IsInvalid() && PhysicalRegister(Op->Upper) == PhysicalRegister(LastNode)) {
|
||||
if (IROp->Op == OP_DIV ? IsSignext(LastOp, Op->Lower, IROp->Size) : IsZero(LastOp)) {
|
||||
Op->Upper.SetInvalid();
|
||||
return PhysicalRegister(LastNode) == PhysicalRegister(Op->OutRemainder);
|
||||
}
|
||||
}
|
||||
} else if (IROp->Op == OP_XGETBV && PhysicalRegister(IROp->Args[0]) == PhysicalRegister(LastNode) && LastOp->Op == OP_CONSTANT) {
|
||||
// Try to constant fold
|
||||
uint64_t ConstantFunction = LastOp->C<IROp_Constant>()->Constant;
|
||||
auto Op = IROp->CW<IR::IROp_XGetBV>();
|
||||
if (CPUID->DoesXCRFunctionReportConstantData(ConstantFunction)) {
|
||||
const auto Result = CPUID->RunXCRFunction(ConstantFunction);
|
||||
IREmit->SetWriteCursorBefore(CodeNode);
|
||||
IREmit->_Constant(Result.eax).Node->Reg = PhysicalRegister(Op->OutEAX).Raw;
|
||||
IREmit->_Constant(Result.edx).Node->Reg = PhysicalRegister(Op->OutEDX).Raw;
|
||||
IREmit->RemovePostRA(CodeNode);
|
||||
return false;
|
||||
}
|
||||
} else if (IROp->Op == OP_CPUID && PhysicalRegister(IROp->Args[0]) == PhysicalRegister(LastNode) && LastOp->Op == OP_CONSTANT) {
|
||||
// Try to constant fold. As a limitation of merging only 2 instructions, we
|
||||
// can only handle constant functions, not constant leafs. This could be
|
||||
// lifted if we generalized at a (significant) complexity cost.
|
||||
uint64_t ConstantFunction = LastOp->C<IROp_Constant>()->Constant;
|
||||
auto Op = IROp->CW<IR::IROp_CPUID>();
|
||||
|
||||
const auto SupportsConstant = CPUID->DoesFunctionReportConstantData(ConstantFunction);
|
||||
if (SupportsConstant.SupportsConstantFunction == CPUIDEmu::SupportsConstant::CONSTANT &&
|
||||
SupportsConstant.NeedsLeaf != CPUIDEmu::NeedsLeafConstant::NEEDSLEAFCONSTANT) {
|
||||
const auto Result = CPUID->RunFunction(ConstantFunction, 0 /* leaf */);
|
||||
|
||||
IREmit->SetWriteCursorBefore(CodeNode);
|
||||
IREmit->_Fence({FEXCore::IR::Fence_Inst});
|
||||
IREmit->_Constant(Result.eax).Node->Reg = PhysicalRegister(Op->OutEAX).Raw;
|
||||
IREmit->_Constant(Result.ebx).Node->Reg = PhysicalRegister(Op->OutEBX).Raw;
|
||||
IREmit->_Constant(Result.ecx).Node->Reg = PhysicalRegister(Op->OutECX).Raw;
|
||||
IREmit->_Constant(Result.edx).Node->Reg = PhysicalRegister(Op->OutEDX).Raw;
|
||||
IREmit->RemovePostRA(CodeNode);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// Merge moves that are immediately consumed.
|
||||
//
|
||||
// x86 code inserts such moves to workaround x86's 2-address code. Because
|
||||
// arm64 is 3-address code, we can optimize these out.
|
||||
//
|
||||
// Note we rely on the short-circuiting here.
|
||||
if (PhysicalRegister(LastNode) == PhysicalRegister(CodeNode) && KillMove(LastOp, IROp, LastNode, CodeNode)) {
|
||||
LOGMAN_THROW_A_FMT(!PhysicalRegister(CodeNode).IsInvalid(), "invariant");
|
||||
|
||||
for (auto s = 0; s < IR::GetRAArgs(IROp->Op); ++s) {
|
||||
if (IROp->Args[s].IsImmediate() && PhysicalRegister(IROp->Args[s]) == PhysicalRegister(LastNode)) {
|
||||
IROp->Args[s].SetImmediate(PhysicalRegister(LastOp->Args[0]).Raw);
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
@@ -465,11 +497,9 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
auto IR_ = IREmit->ViewIR();
|
||||
IR = &IR_;
|
||||
|
||||
// SSAToNewSSA, NewSSAToSSA allocated on first-use
|
||||
PreferredReg.resize(IR->GetSSACount(), PhysicalRegister::Invalid());
|
||||
SSAToReg.resize(IR->GetSSACount(), PhysicalRegister::Invalid());
|
||||
NextUses.resize(IR->GetSSACount(), 0);
|
||||
SpillSlotCount = 0;
|
||||
AnySpilled = false;
|
||||
|
||||
// Next-use distance relative to the block end of each source, last first.
|
||||
@@ -519,7 +549,7 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
// each register, used below. Since we initialized Class->Available,
|
||||
// RegToSSA is otherwise undefined so we can stash our temps there.
|
||||
if (auto Node = DecodeSRANode(IROp, CodeNode); Node != nullptr) {
|
||||
auto Reg = DecodeSRAReg(IROp);
|
||||
auto Reg = DecodeSRAReg(IROp, CodeNode);
|
||||
|
||||
PreferredReg[IR->GetID(Node).Value] = Reg;
|
||||
GetClass(Reg)->RegToSSA[Reg.Reg] = CodeNode;
|
||||
@@ -560,35 +590,46 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
// SourcesNextUses is read backwards, this tracks the index
|
||||
int64_t SourceIndex = SourcesNextUses.size();
|
||||
|
||||
// Forward pass: Assign registers, spilling as we go.
|
||||
// Last nontrivial instruction, for merging as we go.
|
||||
Ref LastNode = nullptr;
|
||||
|
||||
// Forward pass: Assign registers, spilling & optimizing as we go.
|
||||
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
|
||||
LOGMAN_THROW_A_FMT(!IsRAOp(IROp->Op), "RA ops inserted before, so not seen iterating forward");
|
||||
// These do not read or write registers, and must be skipped for merging.
|
||||
// Since we'd be doing this check anyway for merging, do the check now so
|
||||
// we can skip the rest of the logic too.
|
||||
if (IROp->Op == OP_GUESTOPCODE || IROp->Op == OP_INLINECONSTANT) {
|
||||
continue;
|
||||
}
|
||||
|
||||
// Static registers must be consistent at SRA load/store. Evict to ensure.
|
||||
if (auto Node = DecodeSRANode(IROp, CodeNode); Node != nullptr) {
|
||||
auto Reg = DecodeSRAReg(IROp);
|
||||
auto Reg = DecodeSRAReg(IROp, CodeNode);
|
||||
RegisterClass* Class = &Classes[Reg.Class];
|
||||
|
||||
if (!(Class->Available & (1u << Reg.Reg))) {
|
||||
Ref Old = Class->RegToSSA[Reg.Reg];
|
||||
|
||||
LOGMAN_THROW_A_FMT(IsOld(Old), "RegToSSA invariant");
|
||||
LOGMAN_THROW_A_FMT(IsOld(Node), "Haven't remapped this instruction");
|
||||
|
||||
if (Old != Node) {
|
||||
// Before inserting instructions, we need to set the cursor and
|
||||
// reset LastNode so we don't merge across an inserted copy.
|
||||
// Otherwise, we would erroneously miss the copy when determining if
|
||||
// we can merge, and end up unsoundly merging a mov+xchg sequence.
|
||||
IREmit->SetWriteCursorBefore(CodeNode);
|
||||
LastNode = nullptr;
|
||||
|
||||
Ref Copy;
|
||||
|
||||
if (Reg.Class == FPRFixedClass) {
|
||||
IROp_Header* Header = IR->GetOp<IROp_Header>(Old);
|
||||
Copy = IREmit->_VMov(Header->Size, Map(Old));
|
||||
Copy = IREmit->_VMov(Header->Size, OrderedNodeWrapper::FromImmediate(Reg.Raw));
|
||||
} else {
|
||||
Copy = IREmit->_Copy(Map(Old));
|
||||
Copy = IREmit->_Copy(OrderedNodeWrapper::FromImmediate(Reg.Raw));
|
||||
}
|
||||
|
||||
Remap(Old, Copy);
|
||||
FreeReg(Reg);
|
||||
AssignReg(IR->GetOp<IROp_Header>(Copy), Copy, IROp);
|
||||
RemapReg(Old, PhysicalRegister(Copy));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -604,14 +645,15 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
}
|
||||
|
||||
Ref Old = IR->GetNode(IROp->Args[s]);
|
||||
LOGMAN_THROW_A_FMT(IsOld(Old), "before remapping");
|
||||
|
||||
if (!IsInRegisterFile(Old)) {
|
||||
IREmit->SetWriteCursorBefore(CodeNode);
|
||||
LastNode = nullptr;
|
||||
|
||||
Ref Fill = InsertFill(Old);
|
||||
|
||||
Remap(Old, Fill);
|
||||
AssignReg(IR->GetOp<IROp_Header>(Fill), Fill, IROp);
|
||||
RemapReg(Old, PhysicalRegister(Fill));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -621,62 +663,54 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
continue;
|
||||
}
|
||||
|
||||
Ref Node = IR->GetNode(IROp->Args[s]);
|
||||
auto ID = IR->GetID(Node).Value;
|
||||
auto Reg = SSAToReg[ID];
|
||||
|
||||
SourceIndex--;
|
||||
LOGMAN_THROW_A_FMT(SourceIndex >= 0, "Consistent source count");
|
||||
|
||||
if (!SourcesNextUses[SourceIndex]) {
|
||||
Ref Old = IR->GetNode(IROp->Args[s]);
|
||||
auto Reg = SSAToReg[IR->GetID(Map(Old)).Value];
|
||||
if (!Reg.IsInvalid()) {
|
||||
IROp->Args[s].SetImmediate(Reg.Raw);
|
||||
|
||||
if (!Reg.IsInvalid()) {
|
||||
LOGMAN_THROW_A_FMT(IsInRegisterFile(Old), "sources in file");
|
||||
if (!SourcesNextUses[SourceIndex]) {
|
||||
LOGMAN_THROW_A_FMT(IsInRegisterFile(Node), "sources in file");
|
||||
FreeReg(Reg);
|
||||
}
|
||||
}
|
||||
|
||||
NextUses[IROp->Args[s].ID().Value] = SourcesNextUses[SourceIndex];
|
||||
NextUses[ID] = SourcesNextUses[SourceIndex];
|
||||
}
|
||||
|
||||
// Assign destinations.
|
||||
if (GetHasDest(IROp->Op)) {
|
||||
if (GetHasDest(IROp->Op) && PhysicalRegister(CodeNode).IsInvalid()) {
|
||||
AssignReg(IROp, CodeNode, IROp);
|
||||
}
|
||||
|
||||
// Remap sources last, since AssignReg can shuffle.
|
||||
if (!SSAToNewSSA.empty()) {
|
||||
for (auto s = 0; s < IR::GetRAArgs(IROp->Op); ++s) {
|
||||
Ref Remapped = SSAToNewSSA[IROp->Args[s].ID().Value];
|
||||
|
||||
if (Remapped != nullptr) {
|
||||
IREmit->ReplaceNodeArgument(CodeNode, s, Remapped);
|
||||
}
|
||||
}
|
||||
if (IsTrivial(CodeNode, IROp)) {
|
||||
// Delete instructions that only exist for RA
|
||||
IREmit->RemovePostRA(CodeNode);
|
||||
} else if (LastNode && TryPostRAMerge(LastNode, CodeNode, IROp)) {
|
||||
// Merge adjacent instructions
|
||||
IREmit->RemovePostRA(LastNode);
|
||||
LastNode = nullptr;
|
||||
} else {
|
||||
LastNode = CodeNode;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(IP >= 1, "IP relative to end of block, iterating forward");
|
||||
--IP;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(SourceIndex == 0, "Consistent source count in block");
|
||||
}
|
||||
|
||||
/* Now that we're done growing things, we can finalize our results.
|
||||
*
|
||||
* TODO: Rework RegisterAllocationData to remove this memcpy, it's pointless.
|
||||
*/
|
||||
AllocData = RegisterAllocationData::Create(SSAToReg.size());
|
||||
AllocData->SpillSlotCount = SpillSlotCount;
|
||||
memcpy(AllocData->Map, SSAToReg.data(), sizeof(PhysicalRegister) * SSAToReg.size());
|
||||
|
||||
PreferredReg.clear();
|
||||
SSAToNewSSA.clear();
|
||||
NewSSAToSSA.clear();
|
||||
SSAToReg.clear();
|
||||
SpillSlots.clear();
|
||||
NextUses.clear();
|
||||
|
||||
IR->GetHeader()->PostRA = true;
|
||||
}
|
||||
|
||||
fextl::unique_ptr<IR::RegisterAllocationPass> CreateRegisterAllocationPass() {
|
||||
return fextl::make_unique<ConstrainedRAPass>();
|
||||
fextl::unique_ptr<IR::RegisterAllocationPass> CreateRegisterAllocationPass(const FEXCore::CPUIDEmu* CPUID) {
|
||||
return fextl::make_unique<ConstrainedRAPass>(CPUID);
|
||||
}
|
||||
} // namespace FEXCore::IR
|
||||
@@ -12,8 +12,6 @@ $end_info$
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class RegisterAllocationData;
|
||||
struct RegisterAllocationDataDeleter;
|
||||
struct RegisterClassType;
|
||||
|
||||
class RegisterAllocationPass : public FEXCore::IR::Pass {
|
||||
@@ -22,16 +20,6 @@ public:
|
||||
|
||||
// Number of GPRs usable for pairs at start of GPR set. Must be even.
|
||||
uint32_t PairRegs;
|
||||
|
||||
/**
|
||||
* @brief Returns the register and class map array
|
||||
*/
|
||||
virtual RegisterAllocationData* GetAllocationData() = 0;
|
||||
|
||||
/**
|
||||
* @brief Returns and transfers ownership of the register and class map array
|
||||
*/
|
||||
virtual std::unique_ptr<RegisterAllocationData, RegisterAllocationDataDeleter> PullAllocationData() = 0;
|
||||
};
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -161,6 +161,7 @@ private:
|
||||
const FEXCore::HostFeatures& Features;
|
||||
const OpSize GPROpSize;
|
||||
bool ReducedPrecisionMode;
|
||||
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
|
||||
|
||||
// Helpers
|
||||
Ref RotateRight8(uint32_t V, Ref Amount);
|
||||
@@ -774,12 +775,26 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
|
||||
Ref SinValue {};
|
||||
Ref CosValue {};
|
||||
if (ReducedPrecisionMode) {
|
||||
SinValue = IREmit->_F64SIN(St0);
|
||||
CosValue = IREmit->_F64COS(St0);
|
||||
} else {
|
||||
SinValue = IREmit->_F80SIN(St0);
|
||||
CosValue = IREmit->_F80COS(St0);
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
if (DisableVixlIndirectCalls() == 0) {
|
||||
if (ReducedPrecisionMode) {
|
||||
SinValue = IREmit->_F64SIN(St0);
|
||||
CosValue = IREmit->_F64COS(St0);
|
||||
} else {
|
||||
SinValue = IREmit->_F80SIN(St0);
|
||||
CosValue = IREmit->_F80COS(St0);
|
||||
}
|
||||
} else
|
||||
#endif
|
||||
{
|
||||
SinValue = IREmit->_AllocateFPR(OpSize::i128Bit, OpSize::i128Bit);
|
||||
CosValue = IREmit->_AllocateFPR(OpSize::i128Bit, OpSize::i128Bit);
|
||||
if (ReducedPrecisionMode) {
|
||||
IREmit->_F64SINCOS(St0, SinValue, CosValue);
|
||||
} else {
|
||||
IREmit->_F80SINCOS(St0, SinValue, CosValue);
|
||||
}
|
||||
}
|
||||
|
||||
// Push values
|
||||
|
||||
@@ -24,72 +24,22 @@ union PhysicalRegister {
|
||||
: Reg(Reg)
|
||||
, Class(Class.Val) {}
|
||||
|
||||
PhysicalRegister(OrderedNodeWrapper Arg)
|
||||
: Raw(Arg.GetImmediate()) {}
|
||||
|
||||
PhysicalRegister(Ref Node)
|
||||
: Raw(Node->Reg) {}
|
||||
|
||||
static const PhysicalRegister Invalid() {
|
||||
return PhysicalRegister(InvalidClass, InvalidReg);
|
||||
return PhysicalRegister(InvalidClass, 0);
|
||||
}
|
||||
|
||||
bool IsInvalid() const {
|
||||
return *this == Invalid();
|
||||
static_assert(InvalidClass == 0);
|
||||
return Raw == 0;
|
||||
}
|
||||
};
|
||||
|
||||
static_assert(sizeof(PhysicalRegister) == 1);
|
||||
|
||||
struct RegisterAllocationDataDeleter;
|
||||
|
||||
// This class is serialized, can't have any holes in the structure
|
||||
// otherwise ASAN complains about reading uninitialized memory
|
||||
class FEX_PACKED RegisterAllocationData {
|
||||
public:
|
||||
uint32_t SpillSlotCount {};
|
||||
uint32_t MapCount {};
|
||||
PhysicalRegister Map[0];
|
||||
|
||||
PhysicalRegister GetNodeRegister(NodeID Node) const {
|
||||
return Map[Node.Value];
|
||||
}
|
||||
uint32_t SpillSlots() const {
|
||||
return SpillSlotCount;
|
||||
}
|
||||
|
||||
static size_t Size(uint32_t NodeCount) {
|
||||
return sizeof(RegisterAllocationData) + NodeCount * sizeof(Map[0]);
|
||||
}
|
||||
|
||||
using UniquePtr = std::unique_ptr<FEXCore::IR::RegisterAllocationData, RegisterAllocationDataDeleter>;
|
||||
|
||||
static UniquePtr Create(uint32_t NodeCount);
|
||||
|
||||
UniquePtr CreateCopy() const;
|
||||
|
||||
void Serialize(FEXCore::Context::AOTIRWriter& stream) const {
|
||||
stream.Write((const char*)&SpillSlotCount, sizeof(SpillSlotCount));
|
||||
stream.Write((const char*)&MapCount, sizeof(MapCount));
|
||||
// RAData (inline)
|
||||
stream.Write((const char*)&Map[0], sizeof(Map[0]) * MapCount);
|
||||
}
|
||||
};
|
||||
|
||||
struct RegisterAllocationDataDeleter {
|
||||
void operator()(RegisterAllocationData* r) const {
|
||||
FEXCore::Allocator::free(r);
|
||||
}
|
||||
};
|
||||
|
||||
inline auto RegisterAllocationData::Create(uint32_t NodeCount) -> UniquePtr {
|
||||
auto Ret = (RegisterAllocationData*)FEXCore::Allocator::malloc(Size(NodeCount));
|
||||
memset(&Ret->Map[0], PhysicalRegister::Invalid().Raw, NodeCount);
|
||||
Ret->SpillSlotCount = 0;
|
||||
Ret->MapCount = NodeCount;
|
||||
return UniquePtr {Ret};
|
||||
}
|
||||
|
||||
inline auto RegisterAllocationData::CreateCopy() const -> UniquePtr {
|
||||
auto copy = (RegisterAllocationData*)FEXCore::Allocator::malloc(Size(MapCount));
|
||||
memcpy((void*)©->Map[0], (void*)&Map[0], MapCount * sizeof(Map[0]));
|
||||
copy->SpillSlotCount = SpillSlotCount;
|
||||
copy->MapCount = MapCount;
|
||||
return UniquePtr {copy};
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -265,34 +265,29 @@ public:
|
||||
return ValueData;
|
||||
}
|
||||
|
||||
template<typename TT = T>
|
||||
requires (std::is_fundamental_v<TT>)
|
||||
T operator()() const {
|
||||
T operator()() const requires (std::is_fundamental_v<T>)
|
||||
{
|
||||
return ValueData;
|
||||
}
|
||||
|
||||
template<typename TT = T>
|
||||
requires (std::is_same_v<TT, fextl::string>)
|
||||
const T& operator()() const {
|
||||
const fextl::string& operator()() const requires (std::is_same_v<T, fextl::string>)
|
||||
{
|
||||
return ValueData;
|
||||
}
|
||||
|
||||
template<typename TT = T>
|
||||
requires (!std::is_same_v<TT, DefaultValues::Type::StringArrayType>)
|
||||
Value<T>(T Value) {
|
||||
Value(T Value) requires (!std::is_same_v<T, DefaultValues::Type::StringArrayType>)
|
||||
{
|
||||
ValueData = std::move(Value);
|
||||
}
|
||||
|
||||
// Array value types.
|
||||
template<typename TT = T>
|
||||
requires (std::is_same_v<TT, DefaultValues::Type::StringArrayType>)
|
||||
Value(FEXCore::Config::ConfigOption Option, std::string_view) {
|
||||
Value(FEXCore::Config::ConfigOption Option, std::string_view) requires (std::is_same_v<T, DefaultValues::Type::StringArrayType>)
|
||||
{
|
||||
GetListIfExists(Option, &ValueData);
|
||||
}
|
||||
|
||||
template<typename TT = T>
|
||||
requires (std::is_same_v<TT, DefaultValues::Type::StringArrayType>)
|
||||
DefaultValues::Type::StringArrayType& All() {
|
||||
DefaultValues::Type::StringArrayType& All() requires (std::is_same_v<T, DefaultValues::Type::StringArrayType>)
|
||||
{
|
||||
return ValueData;
|
||||
}
|
||||
|
||||
|
||||
@@ -171,7 +171,7 @@ public:
|
||||
FEX_DEFAULT_VISIBILITY virtual void FinalizeAOTIRCache() = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void WriteFilesWithCode(AOTIRCodeFileWriterFn Writer) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer = true) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::ForkableSharedMutex& GetCodeInvalidationMutex() = 0;
|
||||
|
||||
@@ -181,7 +181,7 @@ public:
|
||||
ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) = 0;
|
||||
|
||||
/**
|
||||
* @brief Checks if a PC is inside of a thread's JIT code buffer.
|
||||
* @brief Checks if a PC is inside any code buffer used by the thread's JIT.
|
||||
*
|
||||
* @param Thread Which thread's code buffers to check inside of.
|
||||
* @param Address The PC to check against.
|
||||
|
||||
@@ -209,6 +209,7 @@ enum FallbackHandlerIndex {
|
||||
OPINDEX_F80SQRT,
|
||||
OPINDEX_F80SIN,
|
||||
OPINDEX_F80COS,
|
||||
OPINDEX_F80SINCOS,
|
||||
OPINDEX_F80XTRACT_EXP,
|
||||
OPINDEX_F80XTRACT_SIG,
|
||||
OPINDEX_F80BCDSTORE,
|
||||
@@ -228,6 +229,7 @@ enum FallbackHandlerIndex {
|
||||
// Double Precision
|
||||
OPINDEX_F64SIN,
|
||||
OPINDEX_F64COS,
|
||||
OPINDEX_F64SINCOS,
|
||||
OPINDEX_F64TAN,
|
||||
OPINDEX_F64ATAN,
|
||||
OPINDEX_F64F2XM1,
|
||||
|
||||
@@ -17,7 +17,6 @@ namespace FEXCore::IR {
|
||||
|
||||
class OrderedNode;
|
||||
class RegisterAllocationPass;
|
||||
class RegisterAllocationData;
|
||||
|
||||
enum class SyscallFlags : uint8_t {
|
||||
DEFAULT = 0,
|
||||
|
||||
@@ -6,6 +6,7 @@
|
||||
#include <cstdarg>
|
||||
|
||||
#include <fmt/format.h>
|
||||
#include <fmt/color.h>
|
||||
|
||||
namespace LogMan {
|
||||
enum DebugLevels {
|
||||
@@ -14,23 +15,29 @@ enum DebugLevels {
|
||||
ERROR = 2, ///< Only Errors printed
|
||||
DEBUG = 3, ///< Debug messages added
|
||||
INFO = 4, ///< Info messages added
|
||||
STDOUT = 5, ///< Meant to go to STDOUT
|
||||
STDERR = 6, ///< Meant to go to STDERR
|
||||
};
|
||||
|
||||
static inline const char* DebugLevelStr(uint32_t Level) {
|
||||
switch (Level) {
|
||||
case NONE: return "NONE";
|
||||
case ASSERT: return "ASSERT";
|
||||
case ERROR: return "ERROR";
|
||||
case DEBUG: return "DEBUG";
|
||||
case INFO: return "INFO";
|
||||
case STDOUT: return "STDOUT";
|
||||
case STDERR: return "STDERR";
|
||||
case ASSERT: return "A";
|
||||
case ERROR: return "E";
|
||||
case DEBUG: return "D";
|
||||
case INFO: return "I";
|
||||
default: return "???"; break;
|
||||
}
|
||||
}
|
||||
|
||||
static inline fmt::text_style DebugLevelStyle(uint32_t Level) {
|
||||
switch (Level) {
|
||||
case LogMan::ASSERT: return fmt::bg(fmt::color::red) | fmt::emphasis::bold | fmt::fg(fmt::color::white);
|
||||
case LogMan::ERROR: return fmt::fg(fmt::color::red);
|
||||
case LogMan::DEBUG: return fmt::fg(fmt::color::gray);
|
||||
case LogMan::INFO: return fmt::fg(fmt::color::green);
|
||||
default: return {}; break;
|
||||
}
|
||||
}
|
||||
|
||||
constexpr DebugLevels MSG_LEVEL = INFO;
|
||||
|
||||
// Note that all logging functions with the Fmt or _FMT suffix on them expect
|
||||
@@ -53,9 +60,9 @@ namespace Throw {
|
||||
MFmt(fmt, fmt::make_format_args(args...));
|
||||
}
|
||||
|
||||
#define LOGMAN_THROW_A_FMT(pred, ...) \
|
||||
do { \
|
||||
LogMan::Throw::AFmt(pred, __VA_ARGS__); \
|
||||
#define LOGMAN_THROW_A_FMT(pred, format, ...) \
|
||||
do { \
|
||||
LogMan::Throw::AFmt((pred), "{}:{}, {}: " format, __FILE_NAME__, __LINE__, __FUNCTION__ __VA_OPT__(, ) __VA_ARGS__); \
|
||||
} while (0)
|
||||
#else
|
||||
static inline void AFmt(bool, const char*, ...) {}
|
||||
@@ -104,16 +111,6 @@ namespace Msg {
|
||||
MFmtImpl(INFO, fmt, fmt::make_format_args(args...));
|
||||
}
|
||||
|
||||
template<typename... Args>
|
||||
static inline void OutFmt(const char* fmt, const Args&... args) {
|
||||
MFmtImpl(STDOUT, fmt, fmt::make_format_args(args...));
|
||||
}
|
||||
|
||||
template<typename... Args>
|
||||
static inline void ErrFmt(const char* fmt, const Args&... args) {
|
||||
MFmtImpl(STDERR, fmt, fmt::make_format_args(args...));
|
||||
}
|
||||
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
template<typename... Args>
|
||||
static inline void AFmt(const char* fmt, const Args&... args) {
|
||||
|
||||
@@ -44,6 +44,13 @@ public:
|
||||
void StealAndDropActiveLocks() {
|
||||
Mutex = PTHREAD_MUTEX_INITIALIZER;
|
||||
}
|
||||
|
||||
// Asserts that the mutex isn't exclusively owned by the calling thread.
|
||||
void check_lock_owned_by_self() {
|
||||
[[maybe_unused]] const auto Result = pthread_mutex_lock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == EDEADLK, "User of unique lock must have already locked mutex as write!");
|
||||
}
|
||||
|
||||
private:
|
||||
pthread_mutex_t Mutex;
|
||||
};
|
||||
@@ -85,6 +92,13 @@ public:
|
||||
const auto Result = pthread_rwlock_tryrdlock(&Mutex);
|
||||
return Result == 0;
|
||||
}
|
||||
|
||||
// Asserts that the rwlock isn't exclusively owned by the calling thread.
|
||||
void check_lock_owned_by_self_as_write() {
|
||||
[[maybe_unused]] const auto Result = pthread_rwlock_wrlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == EDEADLK, "User of rwlock must have already locked mutex as write!");
|
||||
}
|
||||
|
||||
// Initialize the internal pthread object to its default initializer state.
|
||||
// Should only ever be used in the child process when a Linux fork() has occured.
|
||||
void StealAndDropActiveLocks() {
|
||||
|
||||
@@ -24,7 +24,7 @@ namespace FEXCore::Utils {
|
||||
* - This is relatively cheap.
|
||||
* - `Unclaim` when the buffer won't be used again for an extended period.
|
||||
* - This is expensive and requires a mutex shared between threads
|
||||
* - `FixedSizePooledAllocation` helper class provided to help with this.
|
||||
* - `PoolBufferWithTimedRetirement` helper class provided to help with this.
|
||||
*
|
||||
* Once the client has disowned a buffer then the allocator is free to reclaim the buffer when another thread is trying to `Claim` a new buffer.
|
||||
* The buffer getting claimed from a disowned client must have had its last use greater than the defined `DURATION` before it has a chance to get
|
||||
@@ -124,7 +124,7 @@ public:
|
||||
*
|
||||
* @param Buffer - The iterator that was previously given with ClaimBuffer
|
||||
*/
|
||||
void UnclaimBuffer(ContainerType::iterator Buffer, BufferOwnedFlag* ClientFlag) {
|
||||
void UnclaimBuffer(const ContainerType::iterator& Buffer, BufferOwnedFlag* ClientFlag) {
|
||||
// Transition the buffer to free, unclaiming if it wasn't free prior.
|
||||
if (ClientFlag->exchange(ClientFlags::FLAG_FREE) != ClientFlags::FLAG_FREE) {
|
||||
std::unique_lock lk {AllocationMutex};
|
||||
@@ -149,24 +149,45 @@ public:
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Try to reown a buffer that we have previous disowned, failing that, claim a new buffer
|
||||
* @brief Try to reown a buffer that was previously disowned
|
||||
*
|
||||
* @param Buffer - The buffer we previously disowned
|
||||
* @param Size - The size of the buffer
|
||||
* @param CurrentClientFlag - The client tracked flag
|
||||
*
|
||||
* Once a DisownBuffer has been called, it is unsafe to use the buffer until it has been reowned
|
||||
* Always Reown a buffer after disowning it before use!
|
||||
* Once DisownBuffer has been called, it is unsafe to use the buffer until it has been reowned
|
||||
* Always reown a buffer before use!
|
||||
*
|
||||
* @return Either the original buffer passed in if we managed to reclaim, or a new buffer if we couldn't
|
||||
* @return The original buffer passed in on successful reown, otherwise std::nullopt
|
||||
*/
|
||||
ContainerType::iterator ReownOrClaimBuffer(ContainerType::iterator Buffer, size_t Size, BufferOwnedFlag* CurrentClientFlag) {
|
||||
std::optional<ContainerType::iterator> TryToReownBuffer(const ContainerType::iterator& Buffer, size_t Size, BufferOwnedFlag* CurrentClientFlag) {
|
||||
ClientFlags Expected = ClientFlags::FLAG_DISOWNED;
|
||||
if (CurrentClientFlag->compare_exchange_strong(Expected, ClientFlags::FLAG_OWNED)) {
|
||||
// If we managed to change the flag from DISOWNED to OWNED then we have successfully reclaimed
|
||||
// Finish setting up state
|
||||
(*Buffer)->LastUsed.store(ClockType::now(), std::memory_order_relaxed);
|
||||
return Buffer;
|
||||
if (!CurrentClientFlag->compare_exchange_strong(Expected, ClientFlags::FLAG_OWNED)) {
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
// If we managed to change the flag from DISOWNED to OWNED then we have successfully reclaimed
|
||||
// Finish setting up state
|
||||
(*Buffer)->LastUsed.store(ClockType::now(), std::memory_order_relaxed);
|
||||
return Buffer;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Try to reown a buffer that was previously disowned, failing that, claim a new buffer
|
||||
*
|
||||
* @param Buffer - The buffer we previously disowned
|
||||
* @param Size - The size of the buffer
|
||||
* @param CurrentClientFlag - The client tracked flag
|
||||
*
|
||||
* Once DisownBuffer has been called, it is unsafe to use the buffer until it has been reowned
|
||||
* Always reown a buffer before use!
|
||||
*
|
||||
* @return The original buffer passed in on successful reown, otherwise a new buffer
|
||||
*/
|
||||
ContainerType::iterator ReownOrClaimBuffer(const ContainerType::iterator& Buffer, size_t Size, BufferOwnedFlag* CurrentClientFlag) {
|
||||
auto Reowned = TryToReownBuffer(Buffer, Size, CurrentClientFlag);
|
||||
if (Reowned) {
|
||||
return Reowned.value();
|
||||
}
|
||||
|
||||
// Couldn't reclaim, just get a new buffer
|
||||
@@ -410,7 +431,7 @@ private:
|
||||
* - Frees stale buffers opportunistically
|
||||
*/
|
||||
template<typename Type, size_t PeriodMS, size_t PeriodFrequency>
|
||||
class FixedSizePooledAllocation final {
|
||||
class PoolBufferWithTimedRetirement final {
|
||||
// If the delayed object reclaimer is more than the thread pool allocator's duration then the pool allocator would always need to reclaim
|
||||
// the buffer rather than giving it back.
|
||||
static_assert(std::chrono::duration(std::chrono::milliseconds(PeriodMS)) <= IntrusivePooledAllocator::DURATION, "DeplayedObjectReclaimer "
|
||||
@@ -420,24 +441,39 @@ class FixedSizePooledAllocation final {
|
||||
"duration");
|
||||
|
||||
public:
|
||||
FixedSizePooledAllocation(IntrusivePooledAllocator& Allocator, size_t Size)
|
||||
PoolBufferWithTimedRetirement(IntrusivePooledAllocator& Allocator, size_t Size)
|
||||
: ThreadAllocator {Allocator}
|
||||
, Size {Size} {}
|
||||
|
||||
/**
|
||||
* @brief Return the owned buffer or allocate another one from the `Allocator`
|
||||
*
|
||||
* The buffer returned isn't guaranteed to be the exact size of `Size` but it will be at least `Size`.
|
||||
* The contents of the memory returned isn't guaranteed to be zero initialized or not.
|
||||
* Not even guaranteed to contain the previous data from the previous reowning if the pointer is the same.
|
||||
* The buffer is guaranteed to have at least `Size` bytes of data.
|
||||
* The initial data in the buffer is undefined, even when the buffer is just reowned.
|
||||
*
|
||||
* @return object of type `Type` allocated with at least the size of `Size` from the constructor
|
||||
* @param NewSize Optional new size for managed data
|
||||
*
|
||||
* @return object of type `Type` allocated within the selected buffer
|
||||
*/
|
||||
Type ReownOrClaimBuffer() {
|
||||
if (!FEXCore::Utils::IntrusivePooledAllocator::IsClientBufferOwned(ClientOwnedFlag)) {
|
||||
Info = ThreadAllocator.ReownOrClaimBuffer(Info, Size, &ClientOwnedFlag);
|
||||
Type ReownOrClaimBuffer(std::optional<size_t> NewSize = std::nullopt) {
|
||||
// Check if we can cheaply re-own a previous buffer
|
||||
std::optional Buffer =
|
||||
IntrusivePooledAllocator::IsClientBufferOwned(ClientOwnedFlag) ? Info : ThreadAllocator.TryToReownBuffer(Info, Size, &ClientOwnedFlag);
|
||||
|
||||
// Ensure the now owned buffer has enough space. If not, unclaim it and proceed to claim a new one
|
||||
if (NewSize && Buffer && (**Buffer)->Size < NewSize.value()) {
|
||||
UnclaimBuffer();
|
||||
Buffer.reset();
|
||||
}
|
||||
|
||||
// Claim a new buffer if needed
|
||||
Size = NewSize.value_or(Size);
|
||||
if (!Buffer) {
|
||||
Buffer = ThreadAllocator.ClaimBuffer(Size, &ClientOwnedFlag);
|
||||
}
|
||||
|
||||
Info = *Buffer;
|
||||
|
||||
// Putting a memset here is very handy for using thread sanitizer to find buffer usage races
|
||||
// Leaving this here for future excavation that will definitely occur here
|
||||
// memset((*Info)->Ptr, 0, Size);
|
||||
|
||||
@@ -25,6 +25,9 @@ struct default_delete : public std::default_delete<T> {
|
||||
template<class T, class Deleter = fextl::default_delete<T>>
|
||||
using unique_ptr = std::unique_ptr<T, Deleter>;
|
||||
|
||||
template<class T>
|
||||
using shared_ptr = std::shared_ptr<T>;
|
||||
|
||||
template<class T, class... Args>
|
||||
requires (!std::is_array_v<T>)
|
||||
fextl::unique_ptr<T> make_unique(Args&&... args) {
|
||||
@@ -32,4 +35,10 @@ fextl::unique_ptr<T> make_unique(Args&&... args) {
|
||||
auto Result = ::new (ptr) T(std::forward<Args>(args)...);
|
||||
return fextl::unique_ptr<T>(Result);
|
||||
}
|
||||
|
||||
template<class T, class... Args>
|
||||
requires (!std::is_array_v<T>)
|
||||
fextl::shared_ptr<T> make_shared(Args&&... args) {
|
||||
return std::allocate_shared<T>(fextl::FEXAlloc<T> {}, std::forward<Args>(args)...);
|
||||
}
|
||||
} // namespace fextl
|
||||
@@ -1,45 +1,36 @@
|
||||
[中文](https://github.com/FEX-Emu/FEX/blob/main/docs/Readme_CN.md)
|
||||
# FEX - Fast x86 emulation frontend
|
||||
FEX allows you to run x86 and x86-64 binaries on an AArch64 host, similar to qemu-user and box86.
|
||||
It has native support for a rootfs overlay, so you don't need to chroot, as well as some thunklibs so it can forward things like GL to the host.
|
||||
FEX presents a Linux 5.15+ interface to the guest, and supports only AArch64 as a host.
|
||||
FEX is very much work in progress, so expect things to change.
|
||||
# FEX: Emulate x86 Programs on ARM64
|
||||
FEX allows you to run x86 applications on ARM64 Linux devices, similar to qemu-user and box64.
|
||||
It offers broad compatibility with both 32-bit and 64-bit binaries, and it can be used alongside Wine/Proton to play Windows games.
|
||||
|
||||
It supports forwarding API calls to host system libraries like OpenGL or Vulkan to reduce emulation overhead.
|
||||
An experimental code cache helps minimize in-game stuttering as much as possible.
|
||||
Furthermore, a per-app configuration system allows tweaking performance per game, e.g. by skipping costly memory model emulation.
|
||||
We also provide a user-friendly FEXConfig GUI to explore and change these settings.
|
||||
|
||||
## Quick start guide
|
||||
## Prerequisites
|
||||
FEX requires ARMv8.0+ hardware. It has been tested with the following Linux distributions, though others are likely to work as well:
|
||||
|
||||
- Arch Linux
|
||||
- Fedora Linux
|
||||
- openSUSE
|
||||
- Ubuntu 22.04/24.04/24.10
|
||||
|
||||
An x86-64 RootFS is required and can be downloaded using our `FEXRootFSFetcher` tool for many distributions.
|
||||
For other distributions you will need to generate your own RootFS (our [wiki page](https://wiki.fex-emu.com/index.php/Development:Setting_up_RootFS) might help).
|
||||
|
||||
## Quick Start
|
||||
### For Ubuntu 22.04, 24.04 and 24.10
|
||||
Execute the following command in the terminal to install FEX through a PPA.
|
||||
|
||||
`curl --silent https://raw.githubusercontent.com/FEX-Emu/FEX/main/Scripts/InstallFEX.py --output /tmp/InstallFEX.py && python3 /tmp/InstallFEX.py && rm /tmp/InstallFEX.py`
|
||||
```sh
|
||||
curl --silent https://raw.githubusercontent.com/FEX-Emu/FEX/main/Scripts/InstallFEX.py | python3
|
||||
```
|
||||
|
||||
This command will walk you through installing FEX through a PPA, and downloading a RootFS for use with FEX.
|
||||
|
||||
Ubuntu PPA is updated with our monthly releases.
|
||||
|
||||
### For everyone else
|
||||
Please see [Building FEX](#building-fex).
|
||||
|
||||
## Getting Started
|
||||
FEX has been tested to build and run on ARMv8.0+ hardware.
|
||||
ARMv7 hardware will not work.
|
||||
Expected operating system usage is Linux. FEX has been tested with the following Linux OSes:
|
||||
|
||||
- Ubuntu 22.04
|
||||
- Ubuntu 24.04
|
||||
- Ubuntu 24.10
|
||||
- Arch Linux
|
||||
|
||||
On AArch64 hosts the user **MUST** have an x86-64 RootFS [Creating a RootFS](#RootFS-Generation).
|
||||
### For other Distributions
|
||||
Follow the guide on the official FEX-Emu Wiki [here](https://wiki.fex-emu.com/index.php/Development:Setting_up_FEX).
|
||||
|
||||
### Navigating the Source
|
||||
See the [Source Outline](docs/SourceOutline.md) for more information.
|
||||
|
||||
### Building FEX
|
||||
Follow the guide on the official FEX-Emu Wiki [here](https://wiki.fex-emu.com/index.php/Development:Setting_up_FEX).
|
||||
|
||||
### RootFS generation
|
||||
AArch64 hosts require a rootfs for running applications.
|
||||
Follow the guide on the wiki page for seeing how to set up the rootfs from scratch
|
||||
https://wiki.fex-emu.com/index.php/Development:Setting_up_RootFS
|
||||
|
||||

|
||||
@@ -84,7 +84,8 @@ def IsSupportedDistro():
|
||||
# We only support what is available in ppa:fex-emu/fex
|
||||
return Distro[1] == "22.04" or \
|
||||
Distro[1] == "24.04" or \
|
||||
Distro[1] == "24.10"
|
||||
Distro[1] == "24.10" or \
|
||||
Distro[1] == "25.04"
|
||||
|
||||
return False
|
||||
|
||||
|
||||
+30
-9
@@ -54,6 +54,7 @@ private:
|
||||
std::vector<pollfd> PollFDs;
|
||||
std::optional<int> CurrentFD; // FD that is currently being processed
|
||||
|
||||
bool is_stopped = false;
|
||||
int AsyncStopRequest[2] = {-1, -1};
|
||||
|
||||
// Maps FD to callback
|
||||
@@ -91,12 +92,22 @@ public:
|
||||
::close(AsyncStopRequest[1]);
|
||||
}
|
||||
|
||||
error run(std::optional<std::chrono::nanoseconds> Timeout = std::nullopt) {
|
||||
void cleanup() {
|
||||
callbacks.clear();
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool stopped() const {
|
||||
return is_stopped;
|
||||
}
|
||||
|
||||
error run_one(std::optional<std::chrono::nanoseconds> Timeout = std::nullopt) {
|
||||
// Process events queued before entering wait loop
|
||||
update_fd_list();
|
||||
|
||||
timespec ts = to_timespec(Timeout.value_or(std::chrono::nanoseconds {0}));
|
||||
|
||||
// ppoll may return EINTR/EAGAIN, so a loop is used here. Normally, we return in the first iteration.
|
||||
while (true) {
|
||||
int Result = ::ppoll(PollFDs.data(), PollFDs.size(), Timeout ? &ts : nullptr, nullptr);
|
||||
|
||||
@@ -104,14 +115,10 @@ public:
|
||||
if (errno == EINTR || errno == EAGAIN) {
|
||||
continue;
|
||||
}
|
||||
callbacks.clear();
|
||||
return error::generic_errno;
|
||||
} else if (Result == 0) {
|
||||
callbacks.clear();
|
||||
return error::timeout;
|
||||
} else {
|
||||
bool exit_requested = false;
|
||||
|
||||
// Walk the FDs and see if we got any results
|
||||
for (auto& ActiveFD : PollFDs) {
|
||||
if (ActiveFD.revents == 0) {
|
||||
@@ -139,14 +146,14 @@ public:
|
||||
ActiveFD.revents = 0;
|
||||
}
|
||||
} else if (Ret == post_callback::stop_reactor) {
|
||||
exit_requested = true;
|
||||
is_stopped = true;
|
||||
}
|
||||
CurrentFD.reset();
|
||||
}
|
||||
if (ActiveFD.revents & (POLLHUP | POLLERR | POLLNVAL | POLLRDHUP)) {
|
||||
auto Callback = std::move(callbacks[ActiveFD.fd]);
|
||||
if (Callback) {
|
||||
exit_requested |= (Callback(error::eof) == post_callback::stop_reactor);
|
||||
is_stopped |= (Callback(error::eof) == post_callback::stop_reactor);
|
||||
}
|
||||
// Error or hangup, erase the socket from our list
|
||||
QueuedEvents.push_back(Event {.FD = {.fd = ActiveFD.fd}, .Erase = true});
|
||||
@@ -155,12 +162,23 @@ public:
|
||||
ActiveFD.revents = 0;
|
||||
}
|
||||
|
||||
if (exit_requested) {
|
||||
callbacks.clear();
|
||||
if (is_stopped) {
|
||||
cleanup();
|
||||
return error::success;
|
||||
}
|
||||
|
||||
update_fd_list();
|
||||
return error::success;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
error run(std::optional<std::chrono::nanoseconds> Timeout = std::nullopt) {
|
||||
while (true) {
|
||||
auto Result = run_one(Timeout);
|
||||
if (Result != error::success || is_stopped) {
|
||||
cleanup();
|
||||
return Result;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -404,6 +422,9 @@ struct posix_descriptor {
|
||||
, FD(std::exchange(Other.FD, -1)) {}
|
||||
|
||||
posix_descriptor& operator=(posix_descriptor&& Other) {
|
||||
if (&Other == this) {
|
||||
return *this;
|
||||
}
|
||||
posix_descriptor::~posix_descriptor();
|
||||
Reactor = Other.Reactor;
|
||||
FD = std::exchange(Other.FD, -1);
|
||||
|
||||
@@ -152,6 +152,10 @@ public:
|
||||
|
||||
protected:
|
||||
void MapNameToOption(const char* ConfigName, const char* ConfigString);
|
||||
void SetCurrentConfigFile(const fextl::string& Filename) {
|
||||
CurrentConfigFile = Filename;
|
||||
}
|
||||
fextl::string CurrentConfigFile;
|
||||
};
|
||||
|
||||
class MainLoader final : public OptionMapper {
|
||||
@@ -199,6 +203,7 @@ void OptionMapper::MapNameToOption(const char* ConfigName, const char* ConfigStr
|
||||
}
|
||||
|
||||
if (!KeyOptionValue.has_value()) {
|
||||
LogMan::Msg::IFmt("Unknown configuration option '{}' in JSON config file '{}'", ConfigName, CurrentConfigFile);
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -223,6 +228,7 @@ MainLoader::MainLoader(FEXCore::Config::LayerType Type, std::string_view ConfigF
|
||||
, Config {ConfigFile} {}
|
||||
|
||||
void MainLoader::Load() {
|
||||
SetCurrentConfigFile(Config);
|
||||
JSON::LoadJSonConfig(Config, [this](const char* Name, const char* ConfigString) { MapNameToOption(Name, ConfigString); });
|
||||
}
|
||||
|
||||
@@ -236,6 +242,7 @@ AppLoader::AppLoader(const fextl::string& Filename, FEXCore::Config::LayerType T
|
||||
}
|
||||
|
||||
void AppLoader::Load() {
|
||||
SetCurrentConfigFile(Config);
|
||||
JSON::LoadJSonConfig(Config, [this](const char* Name, const char* ConfigString) { MapNameToOption(Name, ConfigString); });
|
||||
}
|
||||
|
||||
|
||||
@@ -11,6 +11,7 @@
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXHeaderUtils/Filesystem.h>
|
||||
|
||||
#include <cstdlib>
|
||||
#include <fcntl.h>
|
||||
#include <linux/limits.h>
|
||||
#include <unistd.h>
|
||||
@@ -23,6 +24,7 @@
|
||||
#include <sys/un.h>
|
||||
#include <sys/uio.h>
|
||||
#include <thread>
|
||||
#include <cstring>
|
||||
|
||||
namespace FEXServerClient {
|
||||
int RequestPIDFDPacket(int ServerSocket, PacketType Type) {
|
||||
@@ -222,7 +224,14 @@ int ConnectToAndStartServer(std::string_view InterpreterPath) {
|
||||
return -1;
|
||||
}
|
||||
|
||||
fextl::string FEXServerPath = fextl::fmt::format("{}/FEXServer", InterpreterPath);
|
||||
// Extract directory from InterpreterPath
|
||||
fextl::string InterpreterDir {InterpreterPath};
|
||||
size_t LastSlash = InterpreterDir.rfind('/');
|
||||
if (LastSlash != fextl::string::npos) {
|
||||
InterpreterDir = InterpreterDir.substr(0, LastSlash);
|
||||
}
|
||||
|
||||
fextl::string FEXServerPath = fextl::fmt::format("{}/FEXServer", InterpreterDir);
|
||||
// Check if a local FEXServer next to FEXInterpreter exists
|
||||
// If it does then it takes priority over the installed one
|
||||
if (!FHU::Filesystem::Exists(FEXServerPath)) {
|
||||
|
||||
@@ -186,7 +186,7 @@ void CodeSizeValidation::CalculateBaseStats(FEXCore::Context::Context* CTX, FEXC
|
||||
SetupInfoDisabled = false;
|
||||
}
|
||||
|
||||
static CodeSizeValidation Validation {};
|
||||
static CodeSizeValidation* Validation {};
|
||||
} // namespace CodeSize
|
||||
|
||||
void MsgHandler(LogMan::DebugLevels Level, const char* Message) {
|
||||
@@ -194,19 +194,19 @@ void MsgHandler(LogMan::DebugLevels Level, const char* Message) {
|
||||
|
||||
if (Level == LogMan::INFO) {
|
||||
// Disassemble information is sent through the Info log level.
|
||||
if (!CodeSize::Validation.ParseMessage(Message)) {
|
||||
if (!CodeSize::Validation->ParseMessage(Message)) {
|
||||
return;
|
||||
}
|
||||
if (CodeSize::Validation.InfoPrintingDisabled()) {
|
||||
if (CodeSize::Validation->InfoPrintingDisabled()) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
fextl::fmt::print("[{}] {}\n", CharLevel, Message);
|
||||
fextl::fmt::print("{} {}\n", CharLevel, Message);
|
||||
}
|
||||
|
||||
void AssertHandler(const char* Message) {
|
||||
fextl::fmt::print("[ASSERT] {}\n", Message);
|
||||
fextl::fmt::print("A {}\n", Message);
|
||||
|
||||
// make sure buffers are flushed
|
||||
fflush(nullptr);
|
||||
@@ -248,7 +248,7 @@ static bool TestInstructions(FEXCore::Context::Context* CTX, FEXCore::Core::Inte
|
||||
LogMan::Msg::IFmt("Compiling instruction '{}'", CurrentTest->TestInst);
|
||||
|
||||
TestData[i] =
|
||||
CodeSize::Validation.CompileAndGetStats(CTX, Thread, reinterpret_cast<void*>(CodeRIP), CurrentTest->CodeSize, CurrentTest->x86InstCount);
|
||||
CodeSize::Validation->CompileAndGetStats(CTX, Thread, reinterpret_cast<void*>(CodeRIP), CurrentTest->CodeSize, CurrentTest->x86InstCount);
|
||||
|
||||
// Go to the next test.
|
||||
CurrentTest = reinterpret_cast<const TestInfo*>(&CurrentTest->Code[CurrentTest->CodeSize]);
|
||||
@@ -434,10 +434,44 @@ public:
|
||||
private:
|
||||
fextl::vector<std::pair<std::string_view, std::string_view>> Env;
|
||||
};
|
||||
|
||||
class SimpleSyscallHandler : public FEXCore::HLE::SyscallHandler, public FEXCore::Allocator::FEXAllocOperators {
|
||||
public:
|
||||
SimpleSyscallHandler() {
|
||||
// Just claim to be linux 64-bit for simplicity.
|
||||
OSABI = FEXCore::HLE::SyscallOSABI::OS_LINUX64;
|
||||
}
|
||||
uint64_t HandleSyscall(FEXCore::Core::CpuStateFrame* Frame, FEXCore::HLE::SyscallArguments* Args) override {
|
||||
// Don't do anything
|
||||
return 0;
|
||||
}
|
||||
|
||||
FEXCore::HLE::SyscallABI GetSyscallABI(uint64_t Syscall) override {
|
||||
if (Syscall == 0) {
|
||||
// Claim syscall 0 is simple for instcountci inline tests.
|
||||
return FEXCore::HLE::SyscallABI {
|
||||
.NumArgs = 0,
|
||||
.HasReturn = true,
|
||||
.HostSyscallNumber = 0, // Just map to host syscall zero, it isn't going to get called.
|
||||
};
|
||||
}
|
||||
return {0, false, -1};
|
||||
}
|
||||
|
||||
// These are no-ops implementations of the SyscallHandler API
|
||||
FEXCore::HLE::AOTIRCacheEntryLookupResult LookupAOTIRCacheEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestAddr) override {
|
||||
return {0, 0};
|
||||
}
|
||||
};
|
||||
} // namespace
|
||||
|
||||
int main(int argc, char** argv, char** const envp) {
|
||||
FEXCore::Allocator::GLIBCScopedFault GLIBFaultScope;
|
||||
|
||||
// Initialize early as the message handlers use it.
|
||||
CodeSize::CodeSizeValidation Validation {};
|
||||
CodeSize::Validation = &Validation;
|
||||
|
||||
LogMan::Throw::InstallHandler(AssertHandler);
|
||||
LogMan::Msg::InstallHandler(MsgHandler);
|
||||
FEXCore::Config::Initialize();
|
||||
@@ -620,7 +654,7 @@ int main(int argc, char** argv, char** const envp) {
|
||||
}
|
||||
|
||||
auto SignalDelegation = FEX::DummyHandlers::CreateSignalDelegator();
|
||||
auto SyscallHandler = FEX::DummyHandlers::CreateSyscallHandler();
|
||||
auto SyscallHandler = fextl::make_unique<SimpleSyscallHandler>();
|
||||
|
||||
CTX->SetSignalDelegator(SignalDelegation.get());
|
||||
CTX->SetSyscallHandler(SyscallHandler.get());
|
||||
@@ -630,7 +664,7 @@ int main(int argc, char** argv, char** const envp) {
|
||||
auto ParentThread = CTX->CreateThread(0, 0);
|
||||
|
||||
// Calculate the base stats for instruction testing.
|
||||
CodeSize::Validation.CalculateBaseStats(CTX.get(), ParentThread);
|
||||
CodeSize::Validation->CalculateBaseStats(CTX.get(), ParentThread);
|
||||
|
||||
// Test all the instructions.
|
||||
auto Result = TestInstructions(CTX.get(), ParentThread, argc >= 2 ? argv[2] : nullptr) ? 0 : 1;
|
||||
|
||||
@@ -406,13 +406,8 @@ public:
|
||||
}
|
||||
|
||||
uint64_t GetStackPointer() override {
|
||||
if (Config.Is64BitMode()) {
|
||||
return reinterpret_cast<uint64_t>(FEXCore::Allocator::VirtualAlloc(StackSize())) + StackSize();
|
||||
} else {
|
||||
uint64_t Result = reinterpret_cast<uint64_t>(FEXCore::Allocator::VirtualAlloc(reinterpret_cast<void*>(STACK_OFFSET), StackSize()));
|
||||
LOGMAN_THROW_A_FMT(Result != ~0ULL, "Stack Pointer mmap failed");
|
||||
return Result + StackSize();
|
||||
}
|
||||
LOGMAN_MSG_A_FMT("This should be unused.");
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
uint64_t DefaultRIP() const override {
|
||||
@@ -460,6 +455,11 @@ public:
|
||||
DoMMap(region, size);
|
||||
}
|
||||
|
||||
if (!Config.Is64BitMode()) {
|
||||
// 32-bit gets a fixed page allocated for stack.
|
||||
DoMMap(STACK_OFFSET, StackSize());
|
||||
}
|
||||
|
||||
LoadMemory();
|
||||
|
||||
return true;
|
||||
|
||||
@@ -1,17 +1,11 @@
|
||||
add_executable(FEXBash FEXBash.cpp)
|
||||
target_include_directories(FEXBash
|
||||
PRIVATE
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/Source/
|
||||
${CMAKE_BINARY_DIR}/generated
|
||||
)
|
||||
target_include_directories(FEXBash PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/Source/)
|
||||
|
||||
target_link_libraries(FEXBash
|
||||
PRIVATE
|
||||
FEXCore
|
||||
Common
|
||||
JemallocLibs
|
||||
LinuxEmulation
|
||||
${PTHREAD_LIB}
|
||||
)
|
||||
|
||||
if (CMAKE_BUILD_TYPE MATCHES "RELEASE")
|
||||
|
||||
Loaded 100 of 208 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user