mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 19:00:17 +02:00
Compare commits
255
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
320c5f1847 | ||
|
|
3b88947cbe | ||
|
|
f05be0120e | ||
|
|
10c3b2fb5d | ||
|
|
74ae89865c | ||
|
|
a2445904e7 | ||
|
|
bd5188e7bc | ||
|
|
d89c119d9d | ||
|
|
802f3bed47 | ||
|
|
9c606a278d | ||
|
|
7aa5bc0503 | ||
|
|
d94b9fae94 | ||
|
|
05fcaaa758 | ||
|
|
9626a64340 | ||
|
|
c1cfd4db83 | ||
|
|
bfda91ac16 | ||
|
|
b7117b86ea | ||
|
|
a94059ad81 | ||
|
|
2cab87fbd0 | ||
|
|
2017100ff3 | ||
|
|
72cb29d36b | ||
|
|
a2b0d594fb | ||
|
|
a16d4ff1f3 | ||
|
|
305b1ecf2d | ||
|
|
3cc0cae249 | ||
|
|
137aa59254 | ||
|
|
5f8cb0dc54 | ||
|
|
45978474f3 | ||
|
|
75206c7a51 | ||
|
|
06ff0a45a7 | ||
|
|
c948d532a7 | ||
|
|
3825483f79 | ||
|
|
0b238d942e | ||
|
|
16b09a9dd3 | ||
|
|
4f6800b768 | ||
|
|
c84801adb3 | ||
|
|
69f990c1f8 | ||
|
|
8a83564678 | ||
|
|
43d93f836f | ||
|
|
7c2c0f7fe5 | ||
|
|
3ff47cf677 | ||
|
|
3165dce89e | ||
|
|
d4fad0a3b0 | ||
|
|
15fcaa794f | ||
|
|
6a1a505336 | ||
|
|
85937a7bfb | ||
|
|
3baa598b9e | ||
|
|
6f83b10af6 | ||
|
|
f52bcb49ab | ||
|
|
9cef8ff7ce | ||
|
|
a499ad6404 | ||
|
|
7c8767ab32 | ||
|
|
38a8559943 | ||
|
|
48c6acbc87 | ||
|
|
b5d93bc4a0 | ||
|
|
a9fffe771b | ||
|
|
69f3ab14b9 | ||
|
|
a7b9a34ec9 | ||
|
|
f45be1f59e | ||
|
|
012c2ba851 | ||
|
|
a0c2ce0068 | ||
|
|
86898035da | ||
|
|
05eeffe0f7 | ||
|
|
70cffce5eb | ||
|
|
0258e914e0 | ||
|
|
c35b81d2c5 | ||
|
|
fc63dcedb6 | ||
|
|
c879650c4b | ||
|
|
c4d019939b | ||
|
|
3841cc4aa5 | ||
|
|
468f2f2d2d | ||
|
|
be5db91c13 | ||
|
|
6e7d0a520a | ||
|
|
e37996873a | ||
|
|
25e851b570 | ||
|
|
d41b21a626 | ||
|
|
d65c54ba21 | ||
|
|
a5d3bf0f48 | ||
|
|
f76e8c7185 | ||
|
|
96145e8e3f | ||
|
|
530821fc4a | ||
|
|
e4a4a529dd | ||
|
|
863d1e0007 | ||
|
|
c64d6f2b36 | ||
|
|
a798880ac8 | ||
|
|
9db9be9612 | ||
|
|
6a19c77184 | ||
|
|
c17b9dea30 | ||
|
|
7564b6c69a | ||
|
|
5858c5041f | ||
|
|
18d0dec416 | ||
|
|
752046108c | ||
|
|
773ddbf5c4 | ||
|
|
88e90a6f3a | ||
|
|
84325d6b5c | ||
|
|
1b8e03e1de | ||
|
|
569d7297f0 | ||
|
|
d9520c9498 | ||
|
|
6b61093037 | ||
|
|
74b09d5581 | ||
|
|
08da8f6f5c | ||
|
|
7f8fffbb24 | ||
|
|
b08e5f9821 | ||
|
|
1758648f56 | ||
|
|
418ce0aed9 | ||
|
|
4e926076c9 | ||
|
|
3f0aaeba68 | ||
|
|
6ecec77a74 | ||
|
|
79917dc9bb | ||
|
|
b0b60fb761 | ||
|
|
7463152f50 | ||
|
|
cbd8217c60 | ||
|
|
4de1762c21 | ||
|
|
73962402df | ||
|
|
d90ccceec7 | ||
|
|
0c5ef1ce1d | ||
|
|
e5680e031d | ||
|
|
eee0cffa57 | ||
|
|
c480b0ba41 | ||
|
|
e7a47a647c | ||
|
|
500c82bbf8 | ||
|
|
008528af91 | ||
|
|
13199ad203 | ||
|
|
533dd1a54b | ||
|
|
b6f2e10416 | ||
|
|
c58db36a7c | ||
|
|
4593099882 | ||
|
|
21e29af48b | ||
|
|
892e44a900 | ||
|
|
bca29c2549 | ||
|
|
330e7f628c | ||
|
|
efe401c7a4 | ||
|
|
b3d88c043d | ||
|
|
edd36dac91 | ||
|
|
6de5fb2885 | ||
|
|
3b2ebafd83 | ||
|
|
dd4f508b77 | ||
|
|
3058a825b7 | ||
|
|
bdbeef3b9a | ||
|
|
8ead4a3c34 | ||
|
|
2d53a9b0dc | ||
|
|
8f47b46219 | ||
|
|
4f0d35bae0 | ||
|
|
de6d907e27 | ||
|
|
c9fc347de1 | ||
|
|
85e67f92c5 | ||
|
|
034a27ce5a | ||
|
|
056abb5901 | ||
|
|
af28406dbc | ||
|
|
376d6ba72c | ||
|
|
736a73453f | ||
|
|
6ced309c5c | ||
|
|
a30593efd6 | ||
|
|
3fa400bc55 | ||
|
|
90ea16325a | ||
|
|
be84a4332a | ||
|
|
cfeba859ac | ||
|
|
38ed7c9bb1 | ||
|
|
315ee90837 | ||
|
|
278574ce91 | ||
|
|
6c225d5469 | ||
|
|
d6a466ac0b | ||
|
|
58526d4faa | ||
|
|
3c90a82e75 | ||
|
|
06a27deebb | ||
|
|
f607f877c9 | ||
|
|
7273041314 | ||
|
|
997d041a84 | ||
|
|
226233ddce | ||
|
|
6c8edcbea5 | ||
|
|
e0f27bb855 | ||
|
|
4a14003f34 | ||
|
|
216a37348b | ||
|
|
ef7b2a9d4f | ||
|
|
ffb1cf4c7c | ||
|
|
964165eabe | ||
|
|
dca2d74eb6 | ||
|
|
e136ff52ab | ||
|
|
0df4944a7b | ||
|
|
2bb296df3b | ||
|
|
52380a3927 | ||
|
|
732f725a24 | ||
|
|
8887b16299 | ||
|
|
75c8b08d9b | ||
|
|
9741dc214d | ||
|
|
8b4b80c725 | ||
|
|
e3b6cccca1 | ||
|
|
d28ef1859d | ||
|
|
74b59f458d | ||
|
|
4ae04c2a9a | ||
|
|
7bd0789402 | ||
|
|
eb9fb9e834 | ||
|
|
c0ef0a7503 | ||
|
|
61719115e5 | ||
|
|
4a8cbe2ab8 | ||
|
|
c966f44189 | ||
|
|
193ef8b232 | ||
|
|
ad70a60d05 | ||
|
|
9dfd5b5526 | ||
|
|
09ec374eec | ||
|
|
8c1e9eda12 | ||
|
|
0eedd55dfd | ||
|
|
58c86d16f3 | ||
|
|
4a9170e981 | ||
|
|
b5b8ff01d5 | ||
|
|
6c86ffdeb7 | ||
|
|
70a1d92d9f | ||
|
|
0749477eb9 | ||
|
|
fb54341a1f | ||
|
|
db601d333b | ||
|
|
af1c2cccac | ||
|
|
bf1b76dcd2 | ||
|
|
48787ab460 | ||
|
|
7d4bf84304 | ||
|
|
5cf5d3e3a2 | ||
|
|
8ddb229447 | ||
|
|
9fdd96af61 | ||
|
|
7fbdd0607f | ||
|
|
0c0a1d8f12 | ||
|
|
30dd9ed267 | ||
|
|
5f0a1d55e5 | ||
|
|
17bef26708 | ||
|
|
58b5620f97 | ||
|
|
327b62ea45 | ||
|
|
91c6d693df | ||
|
|
aa3df3914b | ||
|
|
ad3a024b69 | ||
|
|
b9ec94264e | ||
|
|
fee9e91c4f | ||
|
|
1dc560d45a | ||
|
|
258e2b80f4 | ||
|
|
dcb889704d | ||
|
|
93096c27f9 | ||
|
|
3c6eba561f | ||
|
|
d2714f3338 | ||
|
|
b917db34c7 | ||
|
|
c45abaaa6b | ||
|
|
a2286cb00a | ||
|
|
481787d45c | ||
|
|
756002eef8 | ||
|
|
6789919fca | ||
|
|
d8c615ec65 | ||
|
|
6a1eb0e508 | ||
|
|
984b0260d5 | ||
|
|
785e59e889 | ||
|
|
671be5b191 | ||
|
|
b4e556e63a | ||
|
|
2128943b57 | ||
|
|
6ddb1094b6 | ||
|
|
b9d93b19d4 | ||
|
|
97a5da232d | ||
|
|
0d77af5366 | ||
|
|
a3435f2d22 | ||
|
|
154dd46b6d | ||
|
|
782952d55d |
No files matched your search
@@ -7,3 +7,6 @@ FEXCore/Source/Interface/Core/X86Tables/*
|
||||
# Inline headers with list-like content that can't be processed individually
|
||||
Source/Tools/LinuxEmulation/LinuxSyscalls/x*/SyscallsNames.inl
|
||||
Source/Tools/LinuxEmulation/LinuxSyscalls/x*/Ioctl/*.inl
|
||||
|
||||
# Include files in unittests
|
||||
unittests/*ASM/Includes/*.inc
|
||||
@@ -20,3 +20,5 @@
|
||||
# Whole-tree reformat with clang-format-19
|
||||
5267cde60e7642852d18f20ae8568643bb5293d5
|
||||
|
||||
# Minor reformat with clang-format-19
|
||||
9fdd96af61c969cb5732471223f00eda64b7a069
|
||||
@@ -34,7 +34,6 @@ jobs:
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
|
||||
@@ -41,7 +41,6 @@ jobs:
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
|
||||
@@ -34,7 +34,6 @@ jobs:
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
|
||||
@@ -33,7 +33,6 @@ jobs:
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
|
||||
@@ -48,7 +48,6 @@ jobs:
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
@@ -78,7 +77,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_TESTS=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
|
||||
@@ -35,7 +35,6 @@ jobs:
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
|
||||
@@ -46,12 +46,12 @@ jobs:
|
||||
- name: Configure CMake arm64ec
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build_arm64ec
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=arm64ec-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=/usr -DBUILD_TESTS=False -DCMAKE_INSTALL_PREFIX=/usr
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=arm64ec-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=/usr -DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=/usr
|
||||
|
||||
- name: Configure CMake wow64
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build_wow64
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=aarch64-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=/usr -DBUILD_TESTS=False -DCMAKE_INSTALL_PREFIX=/usr
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=aarch64-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=/usr -DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=/usr
|
||||
|
||||
- name: Build arm64ec
|
||||
working-directory: ${{runner.workspace}}/build_arm64ec
|
||||
|
||||
+13
-55
@@ -4,7 +4,6 @@ project(FEX C CXX ASM)
|
||||
INCLUDE (CheckIncludeFiles)
|
||||
CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
|
||||
|
||||
option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
|
||||
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
|
||||
option(BUILD_THUNKS "Build thunks" FALSE)
|
||||
option(BUILD_FEXCONFIG "Build FEXConfig" TRUE)
|
||||
@@ -304,7 +303,8 @@ set (CMAKE_LINKER_FLAGS_RELEASE "${CMAKE_LINKER_FLAGS_RELEASE} -fomit-frame-poin
|
||||
|
||||
include_directories(External/robin-map/include/)
|
||||
|
||||
if (BUILD_TESTS OR ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
|
||||
include(CTest)
|
||||
if (BUILD_TESTING OR ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
|
||||
add_subdirectory(External/vixl/)
|
||||
include_directories(SYSTEM External/vixl/src/)
|
||||
endif()
|
||||
@@ -319,7 +319,7 @@ if (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
|
||||
endif()
|
||||
|
||||
find_package(PkgConfig REQUIRED)
|
||||
find_package(Python 3.0 REQUIRED COMPONENTS Interpreter)
|
||||
find_package(Python 3.9 REQUIRED COMPONENTS Interpreter)
|
||||
|
||||
set(BUILD_SHARED_LIBS OFF)
|
||||
|
||||
@@ -335,7 +335,7 @@ endif()
|
||||
add_definitions(-Wno-trigraphs)
|
||||
add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
|
||||
|
||||
if (BUILD_TESTS)
|
||||
if (BUILD_TESTING)
|
||||
find_package(Catch2 3 QUIET)
|
||||
if (NOT Catch2_FOUND)
|
||||
add_subdirectory(External/Catch2/)
|
||||
@@ -345,6 +345,9 @@ if (BUILD_TESTS)
|
||||
endif()
|
||||
|
||||
include(Catch)
|
||||
else ()
|
||||
# Override any previously generated test list to avoid running stale test binaries
|
||||
file(GENERATE OUTPUT CTestTestfile.cmake CONTENT "# No tests since BUILD_TESTING is disabled")
|
||||
endif()
|
||||
|
||||
find_package(fmt QUIET)
|
||||
@@ -455,13 +458,8 @@ endif()
|
||||
|
||||
add_compile_options(-Wall)
|
||||
|
||||
include(CTest)
|
||||
if (BUILD_TESTS)
|
||||
if (BUILD_TESTING)
|
||||
message(STATUS "Unit tests are enabled")
|
||||
if (NOT BUILD_TESTING)
|
||||
# CMake checks this variable before generating CTestTestfile.cmake
|
||||
message(SEND_ERROR "Unit tests require BUILD_TESTING to be enabled")
|
||||
endif()
|
||||
|
||||
set (TEST_JOB_COUNT "" CACHE STRING "Override number of parallel jobs to use while running tests")
|
||||
if (TEST_JOB_COUNT)
|
||||
@@ -492,10 +490,11 @@ file(GLOB CONFIG_SOURCES CONFIGURE_DEPENDS ${CMAKE_CURRENT_SOURCE_DIR}/Data/*.js
|
||||
# Any application configuration json file gets installed
|
||||
foreach(CONFIG_SRC ${CONFIG_SOURCES})
|
||||
install(FILES ${CONFIG_SRC}
|
||||
DESTINATION ${DATA_DIRECTORY}/)
|
||||
DESTINATION ${DATA_DIRECTORY}/
|
||||
COMPONENT Runtime)
|
||||
endforeach()
|
||||
|
||||
if (BUILD_TESTS)
|
||||
if (BUILD_TESTING)
|
||||
add_subdirectory(unittests/)
|
||||
endif()
|
||||
|
||||
@@ -556,6 +555,7 @@ if (BUILD_THUNKS)
|
||||
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest
|
||||
)"
|
||||
DEPENDS guest-libs
|
||||
COMPONENT Runtime
|
||||
)
|
||||
|
||||
install(
|
||||
@@ -565,6 +565,7 @@ if (BUILD_THUNKS)
|
||||
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest_32
|
||||
)"
|
||||
DEPENDS guest-libs-32
|
||||
COMPONENT Runtime
|
||||
)
|
||||
|
||||
add_custom_target(uninstall_guest-libs
|
||||
@@ -606,46 +607,3 @@ if (OVERRIDE_VERSION STREQUAL "detect")
|
||||
else()
|
||||
set(GIT_DESCRIBE_STRING "FEX-${OVERRIDE_VERSION}")
|
||||
endif()
|
||||
|
||||
# Parse the version here
|
||||
# Change something like `FEX-2106.1-76-<hash>` in to a list
|
||||
string(REPLACE "-" ";" DESCRIBE_LIST ${GIT_DESCRIBE_STRING})
|
||||
|
||||
# Extract the `2106.1` element
|
||||
list(GET DESCRIBE_LIST 1 DESCRIBE_LIST)
|
||||
|
||||
# Change `2106.1` in to a list
|
||||
string(REPLACE "." ";" DESCRIBE_LIST ${DESCRIBE_LIST})
|
||||
|
||||
# Calculate list size
|
||||
list(LENGTH DESCRIBE_LIST LIST_SIZE)
|
||||
|
||||
# Pull out the major version
|
||||
list(GET DESCRIBE_LIST 0 FEX_VERSION_MAJOR)
|
||||
|
||||
# Minor version only exists if there is a .1 at the end
|
||||
# eg: 2106 versus 2106.1
|
||||
if (LIST_SIZE GREATER 1)
|
||||
list(GET DESCRIBE_LIST 1 FEX_VERSION_MINOR)
|
||||
endif()
|
||||
|
||||
# Package creation
|
||||
set (CPACK_GENERATOR "DEB")
|
||||
set (CPACK_PACKAGE_NAME fex-emu)
|
||||
set (CPACK_PACKAGE_FILE_NAME "${CPACK_PACKAGE_NAME}-${GIT_DESCRIBE_STRING}_${CMAKE_SYSTEM_PROCESSOR}")
|
||||
set (CPACK_PACKAGE_CONTACT "FEX-Emu Maintainers <team@fex-emu.com>")
|
||||
set (CPACK_PACKAGE_VERSION_MAJOR "${FEX_VERSION_MAJOR}")
|
||||
set (CPACK_PACKAGE_VERSION_MINOR "${FEX_VERSION_MINOR}")
|
||||
set (CPACK_PACKAGE_VERSION_PATCH "${FEX_VERSION_PATCH}")
|
||||
set (CPACK_PACKAGE_DESCRIPTION_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/Description.txt")
|
||||
|
||||
# Debian defines
|
||||
set (CPACK_DEBIAN_PACKAGE_DEPENDS "libc6, libstdc++6, libepoxy0, libsdl2-2.0-0, libegl1, libx11-6, squashfuse")
|
||||
set (CPACK_DEBIAN_PACKAGE_CONTROL_EXTRA
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/postinst;${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/prerm;${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/triggers")
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
|
||||
# binfmt_misc conflicts with qemu-user-static
|
||||
# We also only install binfmt_misc on aarch64 hosts
|
||||
set (CPACK_DEBIAN_PACKAGE_CONFLICTS "${CPACK_DEBIAN_PACKAGE_CONFLICTS}, qemu-user-static")
|
||||
endif()
|
||||
include (CPack)
|
||||
@@ -2244,8 +2244,7 @@ public:
|
||||
|
||||
template<IsQOrDRegister T>
|
||||
void movi(SubRegSize size, T rd, uint64_t Imm, uint16_t Shift = 0) {
|
||||
LOGMAN_THROW_A_FMT(size == SubRegSize::i8Bit || size == SubRegSize::i16Bit || size == SubRegSize::i32Bit ||
|
||||
size == SubRegSize::i64Bit,
|
||||
LOGMAN_THROW_A_FMT(size == SubRegSize::i8Bit || size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit,
|
||||
"Unsupported movi size");
|
||||
|
||||
uint32_t cmode;
|
||||
|
||||
@@ -5125,7 +5125,7 @@ private:
|
||||
requires (std::is_same_v<T, float> || std::is_same_v<T, double>)
|
||||
[[nodiscard]]
|
||||
static bool IsValidFPValueForImm8(T value) {
|
||||
const uint64_t bits = FEXCore::BitCast<FloatToEquivalentUInt<T>>(value);
|
||||
const uint64_t bits = std::bit_cast<FloatToEquivalentUInt<T>>(value);
|
||||
const uint64_t datasize_idx = FEXCore::ilog2(sizeof(T)) - 1;
|
||||
|
||||
static constexpr std::array mantissa_masks {
|
||||
@@ -5171,7 +5171,7 @@ protected:
|
||||
LOGMAN_THROW_A_FMT(IsValidFPValueForImm8(value), "Value ({}) cannot be encoded into an 8-bit immediate", value);
|
||||
#endif
|
||||
|
||||
const auto bits = FEXCore::BitCast<uint32_t>(value);
|
||||
const auto bits = std::bit_cast<uint32_t>(value);
|
||||
const auto sign = (bits & 0x80000000) >> 24;
|
||||
const auto expb2 = (bits & 0x20000000) >> 23;
|
||||
const auto b5_to_0 = (bits >> 19) & 0x3F;
|
||||
@@ -5184,7 +5184,7 @@ protected:
|
||||
LOGMAN_THROW_A_FMT(IsValidFPValueForImm8(value), "Value ({}) cannot be encoded into an 8-bit immediate", value);
|
||||
#endif
|
||||
|
||||
const auto bits = FEXCore::BitCast<uint64_t>(value);
|
||||
const auto bits = std::bit_cast<uint64_t>(value);
|
||||
const auto sign = (bits & 0x80000000'00000000) >> 56;
|
||||
const auto expb2 = (bits & 0x20000000'00000000) >> 55;
|
||||
const auto b5_to_0 = (bits >> 48) & 0x3F;
|
||||
|
||||
@@ -4,7 +4,8 @@ file(GLOB GEN_CONFIG_SOURCES CONFIGURE_DEPENDS *.json.in)
|
||||
# Any application configuration json file gets installed
|
||||
foreach(CONFIG_SRC ${CONFIG_SOURCES})
|
||||
install(FILES ${CONFIG_SRC}
|
||||
DESTINATION ${DATA_DIRECTORY}/AppConfig/)
|
||||
DESTINATION ${DATA_DIRECTORY}/AppConfig/
|
||||
COMPONENT Runtime)
|
||||
endforeach()
|
||||
|
||||
# Any configuration file json file that needs to be generated
|
||||
@@ -21,5 +22,6 @@ foreach(GEN_CONFIG_SRC ${GEN_CONFIG_SOURCES})
|
||||
# Then install the configured json
|
||||
install(
|
||||
FILES ${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}
|
||||
DESTINATION ${DATA_DIRECTORY}/AppConfig/)
|
||||
DESTINATION ${DATA_DIRECTORY}/AppConfig/
|
||||
COMPONENT Runtime)
|
||||
endforeach()
|
||||
@@ -1,3 +0,0 @@
|
||||
x86 and x86-64 Linux emulator
|
||||
|
||||
FEX allows you to run x86 applications on ARM64 Linux devices. It offers broad compatibility with both 32-bit and 64-bit binaries, and it can be used alongside Wine/Proton to play Windows games.
|
||||
@@ -1,18 +0,0 @@
|
||||
#!/bin/sh
|
||||
set -e
|
||||
update_binfmt() {
|
||||
# Check for update-binfmts
|
||||
command -v update-binfmts >/dev/null || return 0
|
||||
|
||||
# Setup binfmt_misc
|
||||
update-binfmts --import FEX-x86
|
||||
update-binfmts --import FEX-x86_64
|
||||
}
|
||||
|
||||
# Install FEXInterpreter hardlink
|
||||
# Needs to be done before setting up binfmt_misc
|
||||
ln -f /usr/bin/FEXLoader /usr/bin/FEXInterpreter
|
||||
|
||||
if [ $(uname -m) = 'aarch64' ]; then
|
||||
update_binfmt
|
||||
fi
|
||||
@@ -1,17 +0,0 @@
|
||||
#!/bin/sh
|
||||
set -e
|
||||
update_binfmt() {
|
||||
# Check for update-binfmts
|
||||
command -v update-binfmts >/dev/null || return 0
|
||||
|
||||
# Uninstall
|
||||
update-binfmts --unimport FEX-x86
|
||||
update-binfmts --unimport FEX-x86_64
|
||||
}
|
||||
|
||||
if [ $(uname -m) = 'aarch64' ]; then
|
||||
update_binfmt
|
||||
fi
|
||||
|
||||
# Remove FEXInterpreter hardlink
|
||||
unlink /usr/bin/FEXInterpreter
|
||||
@@ -1 +0,0 @@
|
||||
activate-noawait ldconfig
|
||||
+1
-1
@@ -14,7 +14,7 @@ RUN mkdir build
|
||||
|
||||
ARG CC=clang-13
|
||||
ARG CXX=clang++-13
|
||||
RUN cmake -DCMAKE_INSTALL_PREFIX=/usr -DCMAKE_BUILD_TYPE=Release -DUSE_LINKER=lld -DENABLE_LTO=True -DBUILD_TESTS=False -DENABLE_ASSERTIONS=False -G Ninja .
|
||||
RUN cmake -DCMAKE_INSTALL_PREFIX=/usr -DCMAKE_BUILD_TYPE=Release -DUSE_LINKER=lld -DENABLE_LTO=True -DBUILD_TESTING=False -DENABLE_ASSERTIONS=False -G Ninja .
|
||||
RUN ninja
|
||||
|
||||
WORKDIR /FEX/build
|
||||
|
||||
@@ -10,7 +10,8 @@ function(GenBinFmt Name)
|
||||
# Then install the configured binfmt
|
||||
install(
|
||||
FILES ${CMAKE_BINARY_DIR}/Data/binfmts/${FMT_NAME}
|
||||
DESTINATION ${CMAKE_INSTALL_PREFIX}/share/binfmts/)
|
||||
DESTINATION ${CMAKE_INSTALL_PREFIX}/share/binfmts/
|
||||
COMPONENT Runtime)
|
||||
endfunction()
|
||||
|
||||
if (NOT USE_LEGACY_BINFMTMISC)
|
||||
@@ -19,7 +20,8 @@ if (NOT USE_LEGACY_BINFMTMISC)
|
||||
|
||||
install(
|
||||
FILES ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86.conf ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86_64.conf
|
||||
DESTINATION ${CMAKE_INSTALL_PREFIX}/lib/binfmt.d/)
|
||||
DESTINATION ${CMAKE_INSTALL_PREFIX}/lib/binfmt.d/
|
||||
COMPONENT Runtime)
|
||||
else()
|
||||
GenBinFmt(FEX-x86.in)
|
||||
GenBinFmt(FEX-x86_64.in)
|
||||
|
||||
@@ -1 +1 @@
|
||||
:FEX-x86:M:0:\x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter:POCF
|
||||
:FEX-x86:M:0:\x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEX:POCF
|
||||
@@ -1,5 +1,5 @@
|
||||
package fex
|
||||
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter
|
||||
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEX
|
||||
magic \x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00
|
||||
offset 0
|
||||
mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff
|
||||
|
||||
@@ -1 +1 @@
|
||||
:FEX-x86_64:M:0:\x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter:POCF
|
||||
:FEX-x86_64:M:0:\x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEX:POCF
|
||||
@@ -1,5 +1,5 @@
|
||||
package fex
|
||||
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter
|
||||
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEX
|
||||
magic \x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00
|
||||
offset 0
|
||||
mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff
|
||||
|
||||
@@ -2,8 +2,8 @@
|
||||
|
||||
let
|
||||
toolchain = pkgs.fetchzip {
|
||||
url = "https://github.com/bylaws/llvm-mingw/releases/download/20250305/llvm-mingw-20250305-ucrt-ubuntu-20.04-aarch64.tar.xz";
|
||||
sha256 = "sha256-cA03/ab9O61eO9+S2JzIXD4V0HzTXK5/AYyxW2d73Po=";
|
||||
url = "https://github.com/bylaws/llvm-mingw/releases/download/20250920/llvm-mingw-20250920-ucrt-ubuntu-22.04-aarch64.tar.xz";
|
||||
sha256 = "sha256-LaojKjC8KzY+soW5u6eoDoXE3qtYk9Ejr7M3enTqRAE=";
|
||||
};
|
||||
|
||||
cmakeToolchainFile = pkgs.substitute {
|
||||
@@ -45,7 +45,7 @@ pkgs.mkShell {
|
||||
fi
|
||||
'';
|
||||
|
||||
# E.g. cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTS=False
|
||||
# E.g. cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTING=False
|
||||
FEX_CMAKE_TOOLCHAIN_ARM64EC = "--toolchain ${cmakeToolchainFile} -DMINGW_TRIPLE=arm64ec-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows";
|
||||
FEX_CMAKE_TOOLCHAIN_WOW64 = "--toolchain ${cmakeToolchainFile} -DMINGW_TRIPLE=aarch64-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows";
|
||||
FEX_MESON_CROSSFILE = "--cross-file ${mesonCrossFile}";
|
||||
|
||||
@@ -18,4 +18,4 @@ then
|
||||
fi
|
||||
|
||||
set -o xtrace
|
||||
cmake $FEX_CMAKE_TOOLCHAIN_WOW64 -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTS=False $@
|
||||
cmake $FEX_CMAKE_TOOLCHAIN_WOW64 -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTING=False $@
|
||||
@@ -18,4 +18,4 @@ then
|
||||
fi
|
||||
|
||||
set -o xtrace
|
||||
cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTS=False $@
|
||||
cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTING=False $@
|
||||
@@ -14,4 +14,4 @@ fi
|
||||
rm -rf unittests/FEXLinuxTests
|
||||
|
||||
set -o xtrace
|
||||
cmake . $FEX_CMAKE_TOOLCHAINS -DBUILD_TESTS=ON -DBUILD_FEX_LINUX_TESTS=ON
|
||||
cmake . $FEX_CMAKE_TOOLCHAINS -DBUILD_TESTING=ON -DBUILD_FEX_LINUX_TESTS=ON
|
||||
Vendored
+1
-1
Submodule External/fmt updated: 20c8fdad06...e424e3f2e6.
Vendored
+1
-1
Submodule External/vixl updated: 84bc10c107...ed690c9eca.
@@ -78,6 +78,6 @@ install (DIRECTORY include/FEXCore ${CMAKE_BINARY_DIR}/include/FEXCore
|
||||
DESTINATION include
|
||||
COMPONENT Development)
|
||||
|
||||
if (BUILD_TESTS)
|
||||
if (BUILD_TESTING)
|
||||
add_subdirectory(unittests/)
|
||||
endif()
|
||||
@@ -118,41 +118,6 @@ def print_man_env_option(name, desc, default, no_json_key):
|
||||
output_man.write("\\fBdefault:\\fR {0}\n".format(default))
|
||||
output_man.write(".Pp\n\n")
|
||||
|
||||
def print_man_options(options):
|
||||
output_man.write(".Sh OPTIONS\n")
|
||||
output_man.write(".Bl -tag -width -indent\n")
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
short = None
|
||||
long = op_key.lower()
|
||||
|
||||
if ("ShortArg" in op_vals):
|
||||
short = op_vals["ShortArg"]
|
||||
|
||||
default = op_vals["Default"]
|
||||
value_type = op_vals["Type"]
|
||||
|
||||
# Textual default rather than enum based
|
||||
if ("TextDefault" in op_vals):
|
||||
default = op_vals["TextDefault"]
|
||||
|
||||
if (value_type == "str" or value_type == "strarray" or value_type == "strenum"):
|
||||
# Wrap the string argument in quotes
|
||||
default = "'" + default + "'"
|
||||
print_man_option(
|
||||
short,
|
||||
long,
|
||||
op_vals["Desc"],
|
||||
default
|
||||
)
|
||||
if (value_type == "strenum"):
|
||||
Enums = op_vals["Enums"]
|
||||
output_man.write("\\fBAvailable Options:\\fR\n")
|
||||
output_man.write(", ".join(f"{enum_op_val}" for [_, enum_op_val] in Enums.items()))
|
||||
output_man.write("\n.sp\n")
|
||||
|
||||
output_man.write(".El\n")
|
||||
|
||||
def print_man_environment(options):
|
||||
output_man.write(".Sh ENVIRONMENT\n")
|
||||
output_man.write(".Bl -tag -width -indent\n")
|
||||
@@ -194,7 +159,7 @@ def print_man_environment_tail():
|
||||
"By default FEX will look in {$HOME, $XDG_CONFIG_HOME}/.fex-emu/",
|
||||
"This will override the full path",
|
||||
"If FEX_PORTABLE is declared then relative paths are also supported",
|
||||
"For FEXInterpreter: Relative to the FEXInterpreter binary",
|
||||
"For FEX: Relative to the FEX binary",
|
||||
"For WINE: Relative to %LOCALAPPDATA%"
|
||||
],
|
||||
"''", True)
|
||||
@@ -208,7 +173,7 @@ def print_man_environment_tail():
|
||||
"One must be careful with this option as it will override any applications that load with execve as well"
|
||||
"If you need to support applications that execve then use FEX_APP_CONFIG_LOCATION instead"
|
||||
"If FEX_PORTABLE is declared then relative paths are also supported",
|
||||
"For FEXInterpreter: Relative to the FEXInterpreter binary",
|
||||
"For FEX: Relative to the FEX binary",
|
||||
"For WINE: Relative to %LOCALAPPDATA%"
|
||||
],
|
||||
"''", True)
|
||||
@@ -227,8 +192,8 @@ def print_man_environment_tail():
|
||||
"PORTABLE",
|
||||
[
|
||||
"Allows FEX to run without installation. Global locations for configuration and binfmt_misc are ignored.",
|
||||
"For FEXInterpreter on Linux:",
|
||||
"These files are instead read from <FEXInterpreterPath>/fex-emu/ by default.",
|
||||
"For FEX on Linux:",
|
||||
"These files are instead read from <FEXPath>/fex-emu/ by default.",
|
||||
"For Arm64ec/Wow64 WINE builds:",
|
||||
"These files are instead read from $LOCALAPPDATA/fex-emu/ by default.",
|
||||
"For further customization, see FEX_APP_CONFIG_LOCATION and FEX_APP_DATA_LOCATION."
|
||||
@@ -240,20 +205,12 @@ def print_man_header():
|
||||
.Dt FEX
|
||||
.Os Linux
|
||||
.Sh NAME
|
||||
.Nm FEXLoader
|
||||
.Nm FEXInterpreter
|
||||
.Nm FEX
|
||||
.Nm FEXBash
|
||||
.Nd Fast x86-64 and x86 emulation.
|
||||
.Sh SYNOPSIS
|
||||
.Nm
|
||||
.Op options
|
||||
.Op Ar --
|
||||
.Ar Application
|
||||
<args> ...
|
||||
.Pp
|
||||
.Nm FEXInterpreter
|
||||
.Ar Application
|
||||
<args> ...
|
||||
.Ar <args> ...
|
||||
.Pp
|
||||
.Nm FEXBash
|
||||
.Ar <args> ...
|
||||
@@ -361,82 +318,6 @@ def print_config_option(type, group_name, json_name, default_value, short, choic
|
||||
|
||||
output_argloader.write("\n");
|
||||
|
||||
def print_argloader_options(options):
|
||||
output_argloader.write("#ifdef BEFORE_PARSE\n")
|
||||
output_argloader.write("#undef BEFORE_PARSE\n")
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
default = op_vals["Default"]
|
||||
|
||||
if (op_vals["Type"] == "str" or op_vals["Type"] == "strarray" or op_vals["Type"] == "strenum"):
|
||||
# Wrap the string argument in quotes
|
||||
default = "\"" + default + "\""
|
||||
|
||||
# Textual default rather than enum based
|
||||
if ("TextDefault" in op_vals):
|
||||
default = "\"" + op_vals["TextDefault"] + "\""
|
||||
|
||||
short = None
|
||||
choices = None
|
||||
|
||||
if ("ShortArg" in op_vals):
|
||||
short = op_vals["ShortArg"]
|
||||
if ("Choices" in op_vals):
|
||||
choices = op_vals["Choices"]
|
||||
|
||||
print_config_option(
|
||||
op_vals["Type"],
|
||||
op_group,
|
||||
op_key,
|
||||
default,
|
||||
short,
|
||||
choices,
|
||||
op_vals["Desc"])
|
||||
|
||||
output_argloader.write("\n")
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
def print_parse_argloader_options(options):
|
||||
output_argloader.write("#ifdef AFTER_PARSE\n")
|
||||
output_argloader.write("#undef AFTER_PARSE\n")
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
output_argloader.write("if (Options.is_set_by_user(\"{0}\")) {{\n".format(op_key))
|
||||
|
||||
value_type = op_vals["Type"]
|
||||
NeedsString = False
|
||||
conversion_func = "fextl::fmt::format(\"{}\", "
|
||||
if ("ArgumentHandler" in op_vals):
|
||||
NeedsString = True
|
||||
conversion_func = "FEXCore::Config::Handler::{0}(".format(op_vals["ArgumentHandler"])
|
||||
if (value_type == "str"):
|
||||
NeedsString = True
|
||||
conversion_func = "std::move("
|
||||
if (value_type == "bool"):
|
||||
# boolean values need a decimal specifier. Otherwise fmt prints strings.
|
||||
conversion_func = "fextl::fmt::format(\"{:d}\", "
|
||||
|
||||
if (value_type == "strenum"):
|
||||
output_argloader.write("\tfextl::string UserValue = Options[\"{0}\"];\n".format(op_key))
|
||||
output_argloader.write("\tSet(FEXCore::Config::ConfigOption::CONFIG_{}, FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, UserValue));\n".format(op_key.upper(), op_key, op_key))
|
||||
elif (value_type == "strarray"):
|
||||
# these need a bit more help
|
||||
output_argloader.write("\tauto Array = Options.all(\"{0}\");\n".format(op_key))
|
||||
output_argloader.write("\tfor (auto iter = Array.begin(); iter != Array.end(); ++iter) {\n")
|
||||
output_argloader.write("\t\tAppendStrArrayValue(FEXCore::Config::ConfigOption::CONFIG_{0}, *iter);\n".format(op_key.upper()))
|
||||
output_argloader.write("\t}\n")
|
||||
else:
|
||||
if (NeedsString):
|
||||
output_argloader.write("\tfextl::string UserValue = Options[\"{0}\"];\n".format(op_key))
|
||||
else:
|
||||
output_argloader.write("\t{0} UserValue = Options.get(\"{1}\");\n".format(value_type, op_key))
|
||||
|
||||
output_argloader.write("\tSet(FEXCore::Config::ConfigOption::CONFIG_{0}, {1}UserValue));\n".format(op_key.upper(), conversion_func))
|
||||
output_argloader.write("}\n")
|
||||
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
|
||||
def print_parse_envloader_options(options):
|
||||
output_argloader.write("#ifdef ENVLOADER\n")
|
||||
output_argloader.write("#undef ENVLOADER\n")
|
||||
@@ -517,41 +398,6 @@ def print_parse_enum_options(options):
|
||||
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
def check_for_duplicate_options(options):
|
||||
short_map = []
|
||||
long_map = []
|
||||
|
||||
# Spin through all the items and see if we have a duplicate option
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
short = None
|
||||
long = op_key.lower()
|
||||
long_invert = None
|
||||
if ("ShortArg" in op_vals):
|
||||
short = op_vals["ShortArg"]
|
||||
if (op_vals["Type"] == "bool"):
|
||||
long_invert = "no-" + long
|
||||
|
||||
# Check for short key duplication
|
||||
if (short != None):
|
||||
if (short in short_map):
|
||||
raise Exception("Short config '{0}' for option '{1}' has duplicate entry!".format(short, op_key))
|
||||
else:
|
||||
short_map.append(short)
|
||||
|
||||
# Check for long key duplication
|
||||
if (long in long_map):
|
||||
raise Exception("Long config '{0}' has duplicate entry!".format(long))
|
||||
else:
|
||||
long_map.append(long)
|
||||
|
||||
# Check for long key duplication
|
||||
if (long_invert != None):
|
||||
if (long_invert in long_map):
|
||||
raise Exception("Long config '{0}' has duplicate entry!".format(long_invert))
|
||||
else:
|
||||
long_map.append(long_invert)
|
||||
|
||||
if (len(sys.argv) < 5):
|
||||
sys.exit()
|
||||
|
||||
@@ -568,8 +414,6 @@ json_object = json.loads(json_text)
|
||||
options = json_object["Options"]
|
||||
unnamed_options = json_object["UnnamedOptions"]
|
||||
|
||||
check_for_duplicate_options(options)
|
||||
|
||||
# Generate config include file
|
||||
output_file = open(output_filename, "w")
|
||||
print_header()
|
||||
@@ -581,7 +425,6 @@ output_file.close()
|
||||
# Generate man file
|
||||
output_man = open(output_man_page, "w")
|
||||
print_man_header()
|
||||
print_man_options(options)
|
||||
print_man_environment(options)
|
||||
print_man_tail()
|
||||
|
||||
@@ -589,8 +432,6 @@ output_man.close()
|
||||
|
||||
# Generate argument loader code
|
||||
output_argloader = open(output_argumentloader_filename, "w")
|
||||
print_argloader_options(options);
|
||||
print_parse_argloader_options(options);
|
||||
|
||||
# Generate environment loader code
|
||||
print_parse_envloader_options(options);
|
||||
|
||||
@@ -58,10 +58,10 @@ class OpDefinition:
|
||||
JITDispatch: bool
|
||||
JITDispatchOverride: str
|
||||
TiedSource: int
|
||||
Inline: list
|
||||
Arguments: list
|
||||
EmitValidation: list
|
||||
Desc: list
|
||||
Inline: list[str]
|
||||
Arguments: list[OpArgument]
|
||||
EmitValidation: list[str]
|
||||
Desc: list[str]
|
||||
|
||||
def __init__(self):
|
||||
self.Name = None
|
||||
@@ -92,19 +92,14 @@ class OpDefinition:
|
||||
attrs = vars(self)
|
||||
print(", ".join("%s: %s" % item for item in attrs.items()))
|
||||
|
||||
IRTypesToCXX = {}
|
||||
CXXTypeToIR = {}
|
||||
IROps = []
|
||||
IRTypesToCXX: dict[str, IRType] = {}
|
||||
CXXTypeToIR: dict[str, IRType] = {}
|
||||
IROps: list[OpDefinition] = []
|
||||
|
||||
IROpNameMap = {}
|
||||
IROpNameSet: set[str] = set()
|
||||
|
||||
def is_ssa_type(type):
|
||||
if (type == "SSA" or
|
||||
type == "GPR" or
|
||||
type == "GPRPair" or
|
||||
type == "FPR"):
|
||||
return True
|
||||
return False
|
||||
def is_ssa_type(op_type: str):
|
||||
return op_type in {"SSA", "GPR", "GPRPair", "FPR"}
|
||||
|
||||
def parse_irtypes(irtypes):
|
||||
for op_key, op_val in irtypes.items():
|
||||
@@ -219,11 +214,8 @@ def parse_ops(ops):
|
||||
OpArg.DefaultInitializer = DefaultInit[1][:-1]
|
||||
|
||||
# If SSA type then we can generate validation for this op
|
||||
if (OpArg.IsSSA and
|
||||
(OpArg.Type == "GPR" or
|
||||
OpArg.Type == "GPRPair" or
|
||||
OpArg.Type == "FPR")):
|
||||
OpDef.EmitValidation.append(f"GetOpRegClass({ArgName}) == InvalidClass || WalkFindRegClass({ArgName}) == {OpArg.Type}Class")
|
||||
if OpArg.IsSSA and OpArg.Type in {"GPR", "GPRPair", "FPR"}:
|
||||
OpDef.EmitValidation.append(f"GetOpRegClass({ArgName}) == RegClass::Invalid || WalkFindRegClass({ArgName}) == RegClass::{OpArg.Type}")
|
||||
|
||||
OpArg.Name = ArgName
|
||||
OpArg.NameWithPrefix = NameWithPrefix
|
||||
@@ -296,22 +288,29 @@ def parse_ops(ops):
|
||||
#OpDef.print()
|
||||
|
||||
# Error on duplicate op
|
||||
if OpDef.Name in IROpNameMap:
|
||||
if OpDef.Name in IROpNameSet:
|
||||
ExitError("Duplicate Op defined! {}".format(OpDef.Name))
|
||||
|
||||
IROps.append(OpDef)
|
||||
IROpNameMap[OpDef.Name] = 1
|
||||
IROpNameSet.add(OpDef.Name)
|
||||
|
||||
# Print out enum values
|
||||
def print_enums():
|
||||
def print_enums(enums):
|
||||
output_file.write("#ifdef IROP_ENUM\n")
|
||||
output_file.write("enum IROps : uint16_t {\n")
|
||||
|
||||
for op in IROps:
|
||||
output_file.write("\tOP_{},\n" .format(op.Name.upper()))
|
||||
|
||||
output_file.write("};\n")
|
||||
|
||||
for name, members in enums.items():
|
||||
output_file.write(f"enum {name} {{\n")
|
||||
for member in members:
|
||||
if member:
|
||||
output_file.write(f"\t{member}\n")
|
||||
else:
|
||||
output_file.write("\n")
|
||||
output_file.write("};\n\n")
|
||||
|
||||
output_file.write("#undef IROP_ENUM\n")
|
||||
output_file.write("#endif\n\n")
|
||||
|
||||
@@ -408,7 +407,7 @@ def print_ir_sizes():
|
||||
[[nodiscard, gnu::const]] std::string_view const& GetName(IROps Op);
|
||||
[[nodiscard, gnu::const]] uint8_t GetArgs(IROps Op);
|
||||
[[nodiscard, gnu::const]] uint8_t GetRAArgs(IROps Op);
|
||||
[[nodiscard, gnu::const]] FEXCore::IR::RegisterClassType GetRegClass(IROps Op);
|
||||
[[nodiscard, gnu::const]] FEXCore::IR::RegClass GetRegClass(IROps Op);
|
||||
[[nodiscard, gnu::const]] bool HasSideEffects(IROps Op);
|
||||
[[nodiscard, gnu::const]] bool ImplicitFlagClobber(IROps Op);
|
||||
[[nodiscard, gnu::const]] bool GetHasDest(IROps Op);
|
||||
@@ -422,30 +421,29 @@ def print_ir_sizes():
|
||||
def print_ir_reg_classes():
|
||||
output_file.write("#ifdef IROP_REG_CLASSES_IMPL\n")
|
||||
|
||||
output_file.write("constexpr std::array<FEXCore::IR::RegisterClassType, IROps::OP_LAST + 1> IRRegClasses = {\n")
|
||||
output_file.write("constexpr std::array<FEXCore::IR::RegClass, IROps::OP_LAST + 1> IRRegClasses = {\n")
|
||||
for op in IROps:
|
||||
if op.Name == "Last":
|
||||
output_file.write("\tFEXCore::IR::InvalidClass,\n")
|
||||
output_file.write("\tRegClass::Invalid,\n")
|
||||
else:
|
||||
Class = "Invalid"
|
||||
if op.HasDest and op.DestType == None:
|
||||
if op.HasDest and op.DestType is None:
|
||||
ExitError("IR op {} has destination with no destination class".format(op.Name))
|
||||
|
||||
if op.HasDest and op.DestType == "SSA": # Special case SSA type
|
||||
output_file.write("\tFEXCore::IR::ComplexClass,\n")
|
||||
output_file.write("\tRegClass::Complex,\n")
|
||||
elif op.HasDest:
|
||||
output_file.write("\tFEXCore::IR::{}Class,\n".format(op.DestType))
|
||||
output_file.write("\tRegClass::{},\n".format(op.DestType))
|
||||
else:
|
||||
# No destination so it has an invalid destination class
|
||||
output_file.write("\tFEXCore::IR::InvalidClass, // No destination\n")
|
||||
output_file.write("\tRegClass::Invalid, // No destination\n")
|
||||
|
||||
|
||||
output_file.write("};\n\n")
|
||||
|
||||
output_file.write("// Make sure our array maps directly to the IROps enum\n")
|
||||
output_file.write("static_assert(IRRegClasses[IROps::OP_LAST] == FEXCore::IR::InvalidClass);\n\n")
|
||||
output_file.write("static_assert(IRRegClasses[IROps::OP_LAST] == RegClass::Invalid);\n\n")
|
||||
|
||||
output_file.write("FEXCore::IR::RegisterClassType GetRegClass(IROps Op) { return IRRegClasses[Op]; }\n\n")
|
||||
output_file.write("FEXCore::IR::RegClass GetRegClass(IROps Op) { return IRRegClasses[Op]; }\n\n")
|
||||
|
||||
output_file.write("#undef IROP_REG_CLASSES_IMPL\n")
|
||||
output_file.write("#endif\n\n")
|
||||
@@ -568,9 +566,7 @@ def print_ir_arg_printer():
|
||||
|
||||
SSAArgNum = 0
|
||||
FirstArg = True
|
||||
for i in range(0, len(op.Arguments)):
|
||||
arg = op.Arguments[i]
|
||||
|
||||
for arg in op.Arguments:
|
||||
# No point printing temporaries that we can't recover
|
||||
if arg.Temporary:
|
||||
continue
|
||||
@@ -671,7 +667,7 @@ def print_ir_allocator_helpers():
|
||||
output_file.write("\t\treturn HeaderOp->Op;\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("\tFEXCore::IR::RegisterClassType GetOpRegClass(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\tFEXCore::IR::RegClass GetOpRegClass(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\t\treturn GetRegClass(GetOpType(Op));\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
@@ -685,22 +681,21 @@ def print_ir_allocator_helpers():
|
||||
output_file.write("\tIRPair<IROp_{}> _{}(" .format(op.Name, op.Name))
|
||||
|
||||
# Output SSA args first
|
||||
for i in range(0, len(op.Arguments)):
|
||||
arg = op.Arguments[i]
|
||||
LastArg = len(op.Arguments) - i - 1 == 0
|
||||
for i, arg in enumerate(op.Arguments):
|
||||
LastArg = i == len(op.Arguments) - 1
|
||||
|
||||
if arg.Temporary:
|
||||
CType = IRTypesToCXX[arg.Type].CXXName
|
||||
output_file.write("{} {}".format(CType, arg.Name));
|
||||
output_file.write("{} {}".format(CType, arg.Name))
|
||||
elif arg.IsSSA:
|
||||
# SSA value
|
||||
output_file.write("OrderedNodeWrapper {}".format(arg.Name))
|
||||
else:
|
||||
# User defined op that is stored
|
||||
CType = IRTypesToCXX[arg.Type].CXXName
|
||||
output_file.write("{} {}".format(CType, arg.Name));
|
||||
output_file.write("{} {}".format(CType, arg.Name))
|
||||
|
||||
if arg.DefaultInitializer != None:
|
||||
if arg.DefaultInitializer:
|
||||
output_file.write(" = {}".format(arg.DefaultInitializer))
|
||||
|
||||
if not LastArg:
|
||||
@@ -758,20 +753,19 @@ def print_ir_allocator_helpers():
|
||||
if op.SSAArgNum:
|
||||
output_file.write("\tIRPair<IROp_{}> _{}(" .format(op.Name, op.Name))
|
||||
|
||||
for i in range(0, len(op.Arguments)):
|
||||
arg = op.Arguments[i]
|
||||
LastArg = len(op.Arguments) - i - 1 == 0
|
||||
for i, arg in enumerate(op.Arguments):
|
||||
LastArg = i == len(op.Arguments) - 1
|
||||
|
||||
if arg.Temporary:
|
||||
CType = IRTypesToCXX[arg.Type].CXXName
|
||||
output_file.write("{} {}".format(CType, arg.Name));
|
||||
output_file.write("{} {}".format(CType, arg.Name))
|
||||
elif arg.IsSSA:
|
||||
output_file.write("OrderedNode *{}".format(arg.Name))
|
||||
else:
|
||||
CType = IRTypesToCXX[arg.Type].CXXName
|
||||
output_file.write("{} {}".format(CType, arg.Name));
|
||||
output_file.write("{} {}".format(CType, arg.Name))
|
||||
|
||||
if arg.DefaultInitializer != None:
|
||||
if arg.DefaultInitializer:
|
||||
output_file.write(" = {}".format(arg.DefaultInitializer))
|
||||
|
||||
if not LastArg:
|
||||
@@ -812,16 +806,15 @@ def print_ir_allocator_helpers():
|
||||
print_validation(op)
|
||||
|
||||
output_file.write(f"\t\treturn _{op.Name}(")
|
||||
for i in range(0, len(op.Arguments)):
|
||||
arg = op.Arguments[i]
|
||||
LastArg = len(op.Arguments) - i - 1 == 0
|
||||
for i, arg in enumerate(op.Arguments):
|
||||
LastArg = i == len(op.Arguments) - 1
|
||||
output_file.write(arg.Name)
|
||||
if arg.IsSSA:
|
||||
output_file.write("->Wrapped(ListDataBegin)")
|
||||
if not LastArg:
|
||||
output_file.write(", ")
|
||||
output_file.write(");\n");
|
||||
output_file.write("\t}\n\n");
|
||||
output_file.write(");\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("#undef IROP_ALLOCATE_HELPERS\n")
|
||||
output_file.write("#endif\n")
|
||||
@@ -852,8 +845,8 @@ def print_ir_dispatcher_dispatch():
|
||||
output_dispatch_file.write("#endif\n")
|
||||
|
||||
|
||||
if (len(sys.argv) < 4):
|
||||
ExitError()
|
||||
if len(sys.argv) < 4:
|
||||
ExitError("Insufficient parameters passed to script")
|
||||
|
||||
output_filename = sys.argv[2]
|
||||
output_dispatcher_filename = sys.argv[3]
|
||||
@@ -865,6 +858,7 @@ json_file.close()
|
||||
json_object = json.loads(json_text)
|
||||
json_object = {k.upper(): v for k, v in json_object.items()}
|
||||
|
||||
enums = json_object["ENUMS"]
|
||||
ops = json_object["OPS"]
|
||||
irtypes = json_object["IRTYPES"]
|
||||
defines = json_object["DEFINES"]
|
||||
@@ -874,7 +868,7 @@ parse_ops(ops)
|
||||
|
||||
output_file = open(output_filename, "w")
|
||||
|
||||
print_enums()
|
||||
print_enums(enums)
|
||||
print_ir_structs(defines)
|
||||
print_ir_sizes()
|
||||
print_ir_reg_classes()
|
||||
|
||||
@@ -18,6 +18,7 @@ set (SRCS
|
||||
Common/JitSymbols.cpp
|
||||
Interface/Context/Context.cpp
|
||||
Interface/Core/LookupCache.cpp
|
||||
Interface/Core/CodeCache.cpp
|
||||
Interface/Core/Core.cpp
|
||||
Interface/Core/CPUBackend.cpp
|
||||
Interface/Core/Addressing.cpp
|
||||
@@ -57,7 +58,6 @@ set (SRCS
|
||||
Interface/Core/X86Tables/VEXTables.cpp
|
||||
Interface/Core/X86Tables/X87Tables.cpp
|
||||
Interface/GDBJIT/GDBJIT.cpp
|
||||
Interface/IR/AOTIR.cpp
|
||||
Interface/IR/IRDumper.cpp
|
||||
Interface/IR/IREmitter.cpp
|
||||
Interface/IR/PassManager.cpp
|
||||
@@ -202,7 +202,7 @@ add_custom_target(CONFIG_INC
|
||||
DEPENDS "${OUTPUT_MAN_NAME_COMPRESS}")
|
||||
|
||||
# Install the compressed man page
|
||||
install(FILES ${OUTPUT_MAN_NAME_COMPRESS} DESTINATION ${MAN_DIR}/man1)
|
||||
install(FILES ${OUTPUT_MAN_NAME_COMPRESS} COMPONENT Runtime DESTINATION ${MAN_DIR}/man1)
|
||||
|
||||
# Add in diagnostic colours if the option is available.
|
||||
# Ninja code generator will kill colours if this isn't here
|
||||
|
||||
@@ -4,9 +4,9 @@
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXHeaderUtils/BitUtils.h>
|
||||
#include "cephes_128bit.h"
|
||||
|
||||
#include <bit>
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
#include <stdint.h>
|
||||
@@ -501,12 +501,53 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
float ToF32(softfloat_state* state) const {
|
||||
const float32_t Result = extF80_to_f32(state, *this);
|
||||
return FEXCore::BitCast<float>(Result);
|
||||
return std::bit_cast<float>(Result);
|
||||
}
|
||||
|
||||
bool IsSignalingNaN() const {
|
||||
return (Exponent == 0x7FFF) && (Significand & 0x8000000000000000ULL) && !(Significand & 0x4000000000000000ULL) && // Bit 62 clear (signaling)
|
||||
(Significand & 0x3FFFFFFFFFFFFFFFULL);
|
||||
}
|
||||
|
||||
bool IsQuietNaN() const {
|
||||
return (Exponent == 0x7FFF) && (Significand & 0x8000000000000000ULL) && (Significand & 0x4000000000000000ULL); // Bit 62 set (quiet)
|
||||
}
|
||||
|
||||
// Helper to detect if this is any NaN
|
||||
bool IsNaN() const {
|
||||
return IsSignalingNaN() || IsQuietNaN();
|
||||
}
|
||||
|
||||
// X87 value to F64 while preserving signaling nan property
|
||||
double ToF64_PreserveNan(softfloat_state* state) const {
|
||||
if (IsSignalingNaN()) {
|
||||
// we keep it as a signaling nan in ieee754 in 64bits
|
||||
uint64_t sign_bit = Sign ? 0x8000000000000000ULL : 0;
|
||||
uint64_t exp_bits = 0x7FF0000000000000ULL;
|
||||
uint64_t x87_frac = Significand & 0x3FFFFFFFFFFFFFFFULL;
|
||||
uint64_t ieee_frac = (x87_frac >> 11) & 0x0007FFFFFFFFFFFFULL;
|
||||
|
||||
if (ieee_frac == 0) {
|
||||
ieee_frac = 1;
|
||||
}
|
||||
ieee_frac &= ~0x0008000000000000ULL;
|
||||
|
||||
uint64_t result_bits = sign_bit | exp_bits | ieee_frac;
|
||||
return std::bit_cast<double>(result_bits);
|
||||
} else if (IsQuietNaN()) {
|
||||
const float64_t Result = extF80_to_f64(state, *this);
|
||||
uint64_t result_bits = std::bit_cast<uint64_t>(Result);
|
||||
result_bits |= 0x0008000000000000ULL;
|
||||
return std::bit_cast<double>(result_bits);
|
||||
} else {
|
||||
const float64_t Result = extF80_to_f64(state, *this);
|
||||
return std::bit_cast<double>(Result);
|
||||
}
|
||||
}
|
||||
|
||||
double ToF64(softfloat_state* state) const {
|
||||
const float64_t Result = extF80_to_f64(state, *this);
|
||||
return FEXCore::BitCast<double>(Result);
|
||||
return std::bit_cast<double>(Result);
|
||||
}
|
||||
|
||||
FEXCore::VectorRegType ToVector() const {
|
||||
@@ -518,7 +559,7 @@ struct FEX_PACKED X80SoftFloat {
|
||||
BIGFLOAT ToFMax(softfloat_state* state) const {
|
||||
#if BIGFLOATSIZE == 16
|
||||
const float128_t Result = extF80_to_f128(state, *this);
|
||||
return FEXCore::BitCast<BIGFLOAT>(Result);
|
||||
return std::bit_cast<BIGFLOAT>(Result);
|
||||
#else
|
||||
BIGFLOAT result {};
|
||||
memcpy(&result, this, sizeof(result));
|
||||
@@ -577,18 +618,51 @@ struct FEX_PACKED X80SoftFloat {
|
||||
}
|
||||
|
||||
X80SoftFloat(softfloat_state* state, const float rhs) {
|
||||
*this = f32_to_extF80(state, FEXCore::BitCast<float32_t>(rhs));
|
||||
*this = f32_to_extF80(state, std::bit_cast<float32_t>(rhs));
|
||||
}
|
||||
|
||||
X80SoftFloat(softfloat_state* state, const double rhs) {
|
||||
*this = f64_to_extF80(state, FEXCore::BitCast<float64_t>(rhs));
|
||||
*this = f64_to_extF80(state, std::bit_cast<float64_t>(rhs));
|
||||
}
|
||||
|
||||
// Create X80SoftFloat from double while preserving NaN signaling properties
|
||||
static X80SoftFloat FromF64_PreserveNaN(softfloat_state* state, double value) {
|
||||
uint64_t bits = std::bit_cast<uint64_t>(value);
|
||||
|
||||
// Check if it's a nan
|
||||
if ((bits & 0x7FF0000000000000ULL) == 0x7FF0000000000000ULL && (bits & 0x000FFFFFFFFFFFFFULL) != 0) {
|
||||
|
||||
X80SoftFloat result;
|
||||
result.Sign = (bits >> 63) & 1;
|
||||
result.Exponent = 0x7FFF;
|
||||
|
||||
bool is_signaling = !(bits & 0x0008000000000000ULL);
|
||||
uint64_t ieee_payload = bits & 0x0007FFFFFFFFFFFFULL;
|
||||
|
||||
// set bit 63 required for x87
|
||||
result.Significand = 0x8000000000000000ULL;
|
||||
|
||||
if (is_signaling) { // clear bit 62 for signaling nan
|
||||
result.Significand &= ~0x4000000000000000ULL;
|
||||
} else { // clear bit 62 for quiet nan
|
||||
result.Significand |= 0x4000000000000000ULL;
|
||||
}
|
||||
|
||||
// ieee754 51-bit payload -> x87 62-bit payload
|
||||
result.Significand |= (ieee_payload << 11) & 0x3FFFFFFFFFFFFFFFULL;
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
// For non-NaN values, use standard conversion
|
||||
return X80SoftFloat(state, value);
|
||||
}
|
||||
|
||||
X80SoftFloat(softfloat_state* state, BIGFLOAT rhs) {
|
||||
#if BIGFLOATSIZE == 16
|
||||
*this = f128_to_extF80(state, FEXCore::BitCast<float128_t>(rhs));
|
||||
*this = f128_to_extF80(state, std::bit_cast<float128_t>(rhs));
|
||||
#else
|
||||
*this = FEXCore::BitCast<long double>(rhs);
|
||||
*this = std::bit_cast<long double>(rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
|
||||
@@ -2,59 +2,23 @@
|
||||
#pragma once
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <concepts>
|
||||
#include <string_view>
|
||||
#include <optional>
|
||||
|
||||
namespace FEXCore::StrConv {
|
||||
inline bool Conv(std::string_view Value, bool* Result) {
|
||||
*Result = std::strtoull(Value.data(), nullptr, 0);
|
||||
template<std::integral T>
|
||||
bool Conv(std::string_view Value, T* Result) {
|
||||
if constexpr (std::is_signed_v<T>) {
|
||||
*Result = static_cast<T>(std::strtoll(Value.data(), nullptr, 0));
|
||||
} else {
|
||||
*Result = static_cast<T>(std::strtoull(Value.data(), nullptr, 0));
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
inline bool Conv(std::string_view Value, uint8_t* Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
inline bool Conv(std::string_view Value, int8_t* Result) {
|
||||
*Result = std::strtol(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
inline bool Conv(std::string_view Value, uint16_t* Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
inline bool Conv(std::string_view Value, int16_t* Result) {
|
||||
*Result = std::strtol(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
inline bool Conv(std::string_view Value, uint32_t* Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
inline bool Conv(std::string_view Value, int32_t* Result) {
|
||||
*Result = std::strtol(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
inline bool Conv(std::string_view Value, uint64_t* Result) {
|
||||
*Result = std::strtoull(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
inline bool Conv(std::string_view Value, int64_t* Result) {
|
||||
*Result = std::strtoll(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
template<typename T, typename = std::enable_if<std::is_enum<T>::value, T>>
|
||||
inline bool Conv(std::string_view Value, T* Result) {
|
||||
*Result = static_cast<T>(std::stoull(Value.data(), nullptr, 0));
|
||||
template<typename T, typename = std::enable_if_t<std::is_enum_v<T>, T>>
|
||||
bool Conv(std::string_view Value, T* Result) {
|
||||
*Result = static_cast<T>(std::strtoull(Value.data(), nullptr, 0));
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
@@ -4,7 +4,6 @@
|
||||
"Multiblock": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"ShortArg": "m",
|
||||
"Desc": [
|
||||
"Controls multiblock code compilation",
|
||||
"Can cause long JIT compilation times and stutter"
|
||||
@@ -13,7 +12,6 @@
|
||||
"MaxInst": {
|
||||
"Type": "int32",
|
||||
"Default": "5000",
|
||||
"ShortArg": "n",
|
||||
"Desc": [
|
||||
"Maximum number of instruction to store in a block"
|
||||
]
|
||||
@@ -61,7 +59,9 @@
|
||||
"ENABLEWFXT": "enablewfxt",
|
||||
"DISABLEWFXT": "disablewfxt",
|
||||
"ENABLE3DNOW": "enable3dnow",
|
||||
"DISABLE3DNOW": "disable3dnow"
|
||||
"DISABLE3DNOW": "disable3dnow",
|
||||
"ENABLESSE4A": "enablesse4a",
|
||||
"DISABLESSE4A": "disablesse4a"
|
||||
},
|
||||
"Desc": [
|
||||
"Allows controlling of the CPU features in the JIT.",
|
||||
@@ -84,7 +84,8 @@
|
||||
"\t{enable,disable}svebitperm: Will force enable or disable svebitperm even if the host doesn't support it",
|
||||
"\t{enable,disable}preserveallabi: Will force enable or disable preserve_all abi even if the host doesn't support it",
|
||||
"\t{enable,disable}wfxt: Will force enable or disable wfxt even if the host doesn't support it",
|
||||
"\t{enable,disable}3dnow: Will force enable or disable 3DNow even if the host doesn't support it"
|
||||
"\t{enable,disable}3dnow: Will force enable or disable 3DNow! even if the host doesn't support it",
|
||||
"\t{enable,disable}sse4a: Will force enable or disable SSE4a even if the host doesn't support it"
|
||||
]
|
||||
},
|
||||
"SmallTSCScale": {
|
||||
@@ -99,7 +100,6 @@
|
||||
"RootFS": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"ShortArg": "R",
|
||||
"Desc": [
|
||||
"Which Root filesystem prefix to use",
|
||||
"This can be a filesystem path",
|
||||
@@ -114,7 +114,6 @@
|
||||
"ThunkHostLibs": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_FULL_LIBDIR@/fex-emu/HostThunks",
|
||||
"ShortArg": "t",
|
||||
"Desc": [
|
||||
"Folder to find the host-side thunking libraries."
|
||||
]
|
||||
@@ -122,7 +121,6 @@
|
||||
"ThunkGuestLibs": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/share/fex-emu/GuestThunks",
|
||||
"ShortArg": "j",
|
||||
"Desc": [
|
||||
"Folder to find the guest-side thunking libraries."
|
||||
]
|
||||
@@ -130,7 +128,6 @@
|
||||
"ThunkConfig": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"ShortArg": "k",
|
||||
"Desc": [
|
||||
"A json file specifying where to overlay the thunks.",
|
||||
"This can be a filesystem path",
|
||||
@@ -145,7 +142,6 @@
|
||||
"Env": {
|
||||
"Type": "strarray",
|
||||
"Default": "",
|
||||
"ShortArg": "E",
|
||||
"Desc": [
|
||||
"Adds an environment variable to the emulated environment."
|
||||
]
|
||||
@@ -153,7 +149,6 @@
|
||||
"HostEnv": {
|
||||
"Type": "strarray",
|
||||
"Default": "",
|
||||
"ShortArg": "H",
|
||||
"Desc": [
|
||||
"Adds an environment variable to the host environment.",
|
||||
"This can be useful for setting environment variables that thunks can pick up.",
|
||||
@@ -172,7 +167,6 @@
|
||||
"SingleStep": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"ShortArg": "S",
|
||||
"Desc": [
|
||||
"Single stepping configuration."
|
||||
]
|
||||
@@ -180,7 +174,6 @@
|
||||
"GdbServer": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"ShortArg": "G",
|
||||
"Desc": [
|
||||
"Enables the GDB server."
|
||||
]
|
||||
@@ -214,7 +207,6 @@
|
||||
"DumpGPRs": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"ShortArg": "g",
|
||||
"Desc": [
|
||||
"When the test harness ends, print the GPR state."
|
||||
]
|
||||
@@ -222,7 +214,6 @@
|
||||
"O0": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"ShortArg": "O0",
|
||||
"Desc": [
|
||||
"Disables optimizations passes for debugging."
|
||||
]
|
||||
@@ -312,7 +303,6 @@
|
||||
"SilentLog": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"ShortArg": "s",
|
||||
"Desc": [
|
||||
"Disables logging"
|
||||
]
|
||||
@@ -320,7 +310,6 @@
|
||||
"OutputLog": {
|
||||
"Type": "str",
|
||||
"Default": "server",
|
||||
"ShortArg": "o",
|
||||
"Desc": [
|
||||
"File to write FEX output to.",
|
||||
"[stdout, stderr, server, <Filename>]"
|
||||
@@ -395,14 +384,6 @@
|
||||
"This is required to ensure a split-lock doesn't tear inside the process"
|
||||
]
|
||||
},
|
||||
"TSOAutoMigration": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"Automatically enables TSO when shared memory is used.",
|
||||
"Should work without issues in most cases."
|
||||
]
|
||||
},
|
||||
"VolatileMetadata": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
@@ -418,6 +399,14 @@
|
||||
"Emulates X87 floating point using 64-bit precision. This reduces emulation accuracy and may result in rendering bugs."
|
||||
]
|
||||
},
|
||||
"X87StrictReducedPrecision": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Enables stricter X87 floating point behavior when X87ReducedPrecision is enabled.",
|
||||
"Adds additional checks and implementations like NaN propagation for better compatibility."
|
||||
]
|
||||
},
|
||||
"ABILocalFlags": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
@@ -514,10 +503,6 @@
|
||||
},
|
||||
"UnnamedOptions": {
|
||||
"Misc": {
|
||||
"IS_INTERPRETER": {
|
||||
"Type": "bool",
|
||||
"Default": "false"
|
||||
},
|
||||
"INTERPRETER_INSTALLED": {
|
||||
"Type": "bool",
|
||||
"Default": "false"
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
|
||||
@@ -5,53 +5,46 @@
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/IR/AOTIR.h"
|
||||
#include <Interface/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/set.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
#include <stdint.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <mutex>
|
||||
#include <optional>
|
||||
#include <shared_mutex>
|
||||
|
||||
namespace FEXCore {
|
||||
class CodeLoader;
|
||||
class SignalDelegator;
|
||||
class ThunkHandler;
|
||||
|
||||
namespace Core {
|
||||
struct DebugData;
|
||||
struct InternalThreadState;
|
||||
} // namespace Core
|
||||
|
||||
namespace CPU {
|
||||
class Arm64JITCore;
|
||||
class Dispatcher;
|
||||
} // namespace CPU
|
||||
|
||||
namespace HLE {
|
||||
struct SyscallArguments;
|
||||
class SyscallHandler;
|
||||
class SourcecodeResolver;
|
||||
struct SourcecodeMap;
|
||||
class SyscallHandler;
|
||||
} // namespace HLE
|
||||
} // namespace FEXCore
|
||||
|
||||
namespace FEXCore::IR {
|
||||
namespace Validation {
|
||||
class IRValidation;
|
||||
}
|
||||
} // namespace FEXCore::IR
|
||||
|
||||
namespace FEXCore::Context {
|
||||
struct FEX_PACKED ExitFunctionLinkData {
|
||||
uint64_t HostCode;
|
||||
@@ -71,7 +64,23 @@ struct CustomIRResult {
|
||||
using BlockDelinkerFunc = void (*)(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record);
|
||||
constexpr uint32_t TSC_SCALE_MAXIMUM = 1'000'000'000; ///< 1Ghz
|
||||
|
||||
class ContextImpl final : public FEXCore::Context::Context, CPU::CodeBufferManager {
|
||||
class CodeCache : public AbstractCodeCache {
|
||||
public:
|
||||
CodeCache(ContextImpl&);
|
||||
~CodeCache();
|
||||
|
||||
ContextImpl& CTX;
|
||||
bool IsGeneratingCache = false;
|
||||
|
||||
void LoadData(Core::InternalThreadState&, std::byte* MappedCacheFile, const ExecutableFileSectionInfo&) override;
|
||||
bool SaveData(Core::InternalThreadState&, int TargetFD, const ExecutableFileSectionInfo&, uint64_t SerializedBaseAddress) override;
|
||||
|
||||
void InitiateCacheGeneration() override {
|
||||
IsGeneratingCache = true;
|
||||
}
|
||||
};
|
||||
|
||||
class ContextImpl final : public FEXCore::Context::Context, public CPU::CodeBufferManager {
|
||||
public:
|
||||
// Context base class implementation.
|
||||
bool InitCore() override;
|
||||
@@ -141,10 +150,9 @@ public:
|
||||
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) override;
|
||||
FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) override;
|
||||
|
||||
FEXCore::IR::AOTIRCacheEntry* LoadAOTIRCacheEntry(const fextl::string& Name) override;
|
||||
void UnloadAOTIRCacheEntry(FEXCore::IR::AOTIRCacheEntry* Entry) override;
|
||||
|
||||
void FinalizeAOTIRCache() override {}
|
||||
CodeCache& GetCodeCache() override {
|
||||
return CodeCache;
|
||||
}
|
||||
|
||||
void OnCodeBufferAllocated(CPU::CodeBuffer&) override;
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer = true) override;
|
||||
@@ -154,8 +162,6 @@ public:
|
||||
return CodeInvalidationMutex;
|
||||
}
|
||||
|
||||
void MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
|
||||
void ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) override;
|
||||
|
||||
bool IsAddressInCodeBuffer(FEXCore::Core::InternalThreadState* Thread, uintptr_t Address) const override;
|
||||
@@ -177,13 +183,6 @@ public:
|
||||
void MarkMonoBackpatcherBlock(uint64_t BlockEntry) override;
|
||||
|
||||
public:
|
||||
friend class FEXCore::HLE::SyscallHandler;
|
||||
#ifdef JIT_ARM64
|
||||
friend class FEXCore::CPU::Arm64JITCore;
|
||||
#endif
|
||||
|
||||
friend class FEXCore::IR::Validation::IRValidation;
|
||||
|
||||
struct {
|
||||
uint64_t VirtualMemSize {1ULL << 36};
|
||||
uint64_t TSCScale = 0;
|
||||
@@ -196,7 +195,6 @@ public:
|
||||
FEX_CONFIG_OPT(GdbServer, GDBSERVER);
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
|
||||
FEX_CONFIG_OPT(TSOAutoMigration, TSOAUTOMIGRATION);
|
||||
FEX_CONFIG_OPT(VectorTSOEnabled, VECTORTSOENABLED);
|
||||
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
|
||||
FEX_CONFIG_OPT(ABILocalFlags, ABILOCALFLAGS);
|
||||
@@ -209,6 +207,7 @@ public:
|
||||
FEX_CONFIG_OPT(GDBSymbols, GDBSYMBOLS);
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
|
||||
FEX_CONFIG_OPT(x87StrictReducedPrecision, X87STRICTREDUCEDPRECISION);
|
||||
FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY);
|
||||
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
|
||||
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
|
||||
@@ -227,6 +226,7 @@ public:
|
||||
FEXCore::HLE::SourcecodeResolver* SourcecodeResolver {};
|
||||
FEXCore::ThunkHandler* ThunkHandler {};
|
||||
fextl::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
CodeCache CodeCache;
|
||||
|
||||
SignalDelegator* SignalDelegation {};
|
||||
X86GeneratedCode X86CodeGen;
|
||||
@@ -317,12 +317,16 @@ protected:
|
||||
VectorAtomicTSOEmulationEnabled = true;
|
||||
MemcpyAtomicTSOEmulationEnabled = true;
|
||||
} else {
|
||||
// Atomic TSO emulation only enabled if the config option is enabled.
|
||||
AtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled;
|
||||
// Atomic vector TSO emulation only enabled if TSO emulation is enabled and also vector TSO is enabled.
|
||||
VectorAtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled && Config.VectorTSOEnabled;
|
||||
// Atomic memcpy TSO emulation only enabled if TSO emulation is enabled and also memcpy TSO is enabled.
|
||||
MemcpyAtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled && Config.MemcpySetTSOEnabled;
|
||||
AtomicTSOEmulationEnabled = Config.TSOEnabled;
|
||||
VectorAtomicTSOEmulationEnabled = Config.TSOEnabled && Config.VectorTSOEnabled;
|
||||
MemcpyAtomicTSOEmulationEnabled = Config.TSOEnabled && Config.MemcpySetTSOEnabled;
|
||||
}
|
||||
}
|
||||
|
||||
void UpdateX87PrecisionConfig() {
|
||||
// If strict reduced precision is enabled, automatically enable reduced precision
|
||||
if (Config.x87StrictReducedPrecision() && !Config.x87ReducedPrecision()) {
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_X87REDUCEDPRECISION, "1");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -336,9 +340,6 @@ private:
|
||||
*/
|
||||
void InitializeCompiler(FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
IR::AOTIRCaptureCache IRCaptureCache;
|
||||
|
||||
bool IsMemoryShared = false;
|
||||
bool SupportsHardwareTSO = false;
|
||||
bool AtomicTSOEmulationEnabled = true;
|
||||
bool VectorAtomicTSOEmulationEnabled = false;
|
||||
@@ -351,8 +352,8 @@ private:
|
||||
std::atomic<bool> HasCustomIRHandlers {};
|
||||
struct CustomIRHandlerEntry final {
|
||||
CustomIREntrypointHandler Handler;
|
||||
void *Creator;
|
||||
void *Data;
|
||||
void* Creator;
|
||||
void* Data;
|
||||
};
|
||||
fextl::unordered_map<uint64_t, CustomIRHandlerEntry> CustomIRHandlers;
|
||||
IntervalList<uint64_t> ForceTSOValidRanges; // The ranges for which ForceTSOInstructions has populated data
|
||||
|
||||
@@ -7,7 +7,7 @@
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage) {
|
||||
Ref LoadEffectiveAddress(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage) {
|
||||
Ref Tmp = A.Base;
|
||||
|
||||
if (A.Offset) {
|
||||
@@ -51,8 +51,8 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, b
|
||||
return Tmp ?: IREmit->Constant(0);
|
||||
}
|
||||
|
||||
AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO, bool Vector,
|
||||
IR::OpSize AccessSize) {
|
||||
AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO,
|
||||
bool Vector, IR::OpSize AccessSize) {
|
||||
const auto Is32Bit = GPRSize == OpSize::i32Bit;
|
||||
const auto GPRSizeMatchesAddrSize = A.AddrSize == GPRSize;
|
||||
const auto OffsetIndexToLargeFor32Bit = Is32Bit && (A.Offset <= -16384 || A.Offset >= 16384);
|
||||
@@ -103,7 +103,7 @@ AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSi
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(IREmit, B, GPRSize, true /* AddSegmentBase */, false),
|
||||
.Index = IREmit->Constant(A.Offset),
|
||||
.IndexType = MEM_OFFSET_SXTX,
|
||||
.IndexType = MemOffsetType::SXTX,
|
||||
.IndexScale = 1,
|
||||
};
|
||||
}
|
||||
@@ -111,15 +111,17 @@ AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSi
|
||||
if (AtomicTSO) {
|
||||
// TODO: LRCPC3 support for vector Imm9.
|
||||
} else if (!Is32Bit && A.Base && (A.Index || A.Segment) && !A.Offset && (A.IndexScale == 1 || A.IndexScale == AccessSizeAsImm)) {
|
||||
AddressMode B = A;
|
||||
|
||||
// ScaledRegisterLoadstore
|
||||
if (A.Index && A.Segment) {
|
||||
A.Base = IREmit->Add(GPRSize, A.Base, A.Segment);
|
||||
} else if (A.Segment) {
|
||||
A.Index = A.Segment;
|
||||
A.IndexScale = 1;
|
||||
if (B.Index && B.Segment) {
|
||||
B.Base = IREmit->Add(GPRSize, B.Base, B.Segment);
|
||||
} else if (B.Segment) {
|
||||
B.Index = B.Segment;
|
||||
B.IndexScale = 1;
|
||||
}
|
||||
|
||||
return A;
|
||||
return B;
|
||||
}
|
||||
|
||||
if (Vector || !AtomicTSO) {
|
||||
@@ -134,7 +136,7 @@ AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSi
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(IREmit, B, GPRSize, true /* AddSegmentBase */, false),
|
||||
.Index = IREmit->Constant(A.Offset),
|
||||
.IndexType = MEM_OFFSET_SXTX,
|
||||
.IndexType = MemOffsetType::SXTX,
|
||||
.IndexScale = 1,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -11,17 +11,18 @@ struct AddressMode {
|
||||
Ref Segment {nullptr};
|
||||
Ref Base {nullptr};
|
||||
Ref Index {nullptr};
|
||||
MemOffsetType IndexType = MEM_OFFSET_SXTX;
|
||||
uint8_t IndexScale = 1;
|
||||
int64_t Offset = 0;
|
||||
|
||||
MemOffsetType IndexType = MemOffsetType::SXTX;
|
||||
uint8_t IndexScale = 1;
|
||||
|
||||
// Size in bytes for the address calculation. 8 for an arm64 hardware mode.
|
||||
IR::OpSize AddrSize;
|
||||
bool NonTSO;
|
||||
};
|
||||
|
||||
Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage = false);
|
||||
AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO, bool Vector,
|
||||
IR::OpSize AccessSize);
|
||||
Ref LoadEffectiveAddress(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage = false);
|
||||
AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO,
|
||||
bool Vector, IR::OpSize AccessSize);
|
||||
|
||||
}; // namespace FEXCore::IR
|
||||
} // namespace FEXCore::IR
|
||||
@@ -1,10 +1,10 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "FEXCore/Core/X86Enums.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
|
||||
@@ -1,30 +1,31 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include "FEXCore/Utils/EnumUtils.h"
|
||||
#include "Interface/Core/JIT/Relocations.h"
|
||||
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
#include <aarch64/disasm-aarch64.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#endif
|
||||
#ifdef VIXL_SIMULATOR
|
||||
#include <aarch64/simulator-aarch64.h>
|
||||
#include <aarch64/simulator-constants-aarch64.h>
|
||||
#endif
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
#include <CodeEmitter/Registers.h>
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <optional>
|
||||
#include <span>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
namespace FEXCore::X86State {
|
||||
enum X86Reg : uint32_t;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
// Contains the address to the currently available CPU state
|
||||
|
||||
@@ -1,14 +1,17 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "FEXCore/Utils/AllocatorHooks.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/AllocatorHooks.h>
|
||||
#include <FEXCore/Utils/PrctlUtils.h>
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
#include "LookupCache.h"
|
||||
|
||||
#ifndef _WIN32
|
||||
#include <linux/prctl.h>
|
||||
#include <sys/prctl.h>
|
||||
#endif
|
||||
|
||||
@@ -357,6 +360,10 @@ namespace CPU {
|
||||
LogMan::Msg::EFmt("Failed to mprotect last page of code buffer.");
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
prctl(PR_SET_VMA, PR_SET_VMA_ANON_NAME, Ptr, Size, "FEXMemJIT");
|
||||
#endif
|
||||
|
||||
LookupCache = fextl::make_unique<GuestToHostMap>();
|
||||
}
|
||||
|
||||
|
||||
@@ -17,6 +17,10 @@ $end_info$
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
union Relocation;
|
||||
}
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
namespace IR {
|
||||
@@ -157,18 +161,7 @@ namespace CPU {
|
||||
virtual CompiledCode CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, bool CheckTF) = 0;
|
||||
|
||||
/**
|
||||
* @brief Relocates a block of code from the JIT code object cache
|
||||
*
|
||||
* @param Entry - RIP of the entry
|
||||
* @param SerializationData - Serialization data referring to the object cache for `Entry`
|
||||
*
|
||||
* @return An executable function pointer relocated from the cache object
|
||||
*/
|
||||
[[nodiscard]]
|
||||
virtual void* RelocateJITObjectCode(uint64_t /* Entry */, const CodeSerialize::CodeObjectFileSection* /* SerializationData */) {
|
||||
return nullptr;
|
||||
}
|
||||
virtual fextl::vector<FEXCore::CPU::Relocation> TakeRelocations() = 0;
|
||||
|
||||
virtual void ClearCache() {}
|
||||
|
||||
|
||||
@@ -43,12 +43,15 @@ namespace ProductNames {
|
||||
static const char ARM_A715[] = "Cortex-A715";
|
||||
static const char ARM_A720[] = "Cortex-A720";
|
||||
static const char ARM_A725[] = "Cortex-A725";
|
||||
static const char ARM_C1Pro[] = "C1-Pro";
|
||||
static const char ARM_C1Premium[] = "C1-Premium";
|
||||
static const char ARM_X1[] = "Cortex-X1";
|
||||
static const char ARM_X1C[] = "Cortex-X1C";
|
||||
static const char ARM_X2[] = "Cortex-X2";
|
||||
static const char ARM_X3[] = "Cortex-X3";
|
||||
static const char ARM_X4[] = "Cortex-X4";
|
||||
static const char ARM_X925[] = "Cortex-X925";
|
||||
static const char ARM_C1Ultra[] = "C1-Ultra";
|
||||
static const char ARM_N1[] = "Neoverse N1";
|
||||
static const char ARM_N2[] = "Neoverse N2";
|
||||
static const char ARM_N3[] = "Neoverse N3";
|
||||
@@ -59,6 +62,7 @@ namespace ProductNames {
|
||||
static const char ARM_A65[] = "Cortex-A65";
|
||||
static const char ARM_A510[] = "Cortex-A510";
|
||||
static const char ARM_A520[] = "Cortex-A520";
|
||||
static const char ARM_C1Nano[] = "C1-Nano";
|
||||
|
||||
static const char ARM_Kryo200[] = "Kryo 2xx";
|
||||
static const char ARM_Kryo300[] = "Kryo 3xx";
|
||||
@@ -70,6 +74,7 @@ namespace ProductNames {
|
||||
|
||||
static const char ARM_Denver[] = "Nvidia Denver";
|
||||
static const char ARM_Carmel[] = "Nvidia Carmel";
|
||||
static const char ARM_Olympus[] = "Nvidia Olympus";
|
||||
|
||||
static const char ARM_Firestorm_M1[] = "Apple Firestorm (M1)";
|
||||
static const char ARM_Icestorm_M1[] = "Apple Icestorm (M1)";
|
||||
@@ -85,6 +90,9 @@ namespace ProductNames {
|
||||
static const char ARM_Blizzard_M2Max[] = "Apple Blizzard (M2 Max)";
|
||||
|
||||
static const char ARM_ORYON_1[] = "Oryon-1";
|
||||
static const char ARM_Ampere_1[] = "AmpereOne";
|
||||
static const char ARM_Ampere_1A[] = "AmpereOneA";
|
||||
static const char ARM_Ampere_1B[] = "AmpereOneB";
|
||||
#else
|
||||
#endif
|
||||
} // namespace ProductNames
|
||||
@@ -170,7 +178,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
// CPU priority order
|
||||
// This is mostly arbitrary but will sort by some sort of CPU priority by performance
|
||||
// Relative list so things they will commonly end up in big.little configurations sort of relate
|
||||
static constexpr std::array<CPUMIDR, 58> CPUMIDRs = {{
|
||||
static constexpr std::array<CPUMIDR, 66> CPUMIDRs = {{
|
||||
// Typically big CPU cores
|
||||
{0x51, 0x001, 1, ProductNames::ARM_ORYON_1}, // Qualcomm Oryon-1
|
||||
|
||||
@@ -181,38 +189,46 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
{0x61, 0x025, 1, ProductNames::ARM_Firestorm_M1Pro}, // Apple Firestorm (M1 Pro)
|
||||
{0x61, 0x023, 1, ProductNames::ARM_Firestorm_M1}, // Apple Firestorm (M1)
|
||||
|
||||
{0x41, 0xd85, 1, ProductNames::ARM_X925}, // X925
|
||||
{0x41, 0xd87, 1, ProductNames::ARM_A725}, // A725
|
||||
{0x41, 0xd84, 1, ProductNames::ARM_V3}, // V3
|
||||
{0x41, 0xd83, 1, ProductNames::ARM_V3AE}, // V3AE
|
||||
{0x41, 0xd8e, 1, ProductNames::ARM_N3}, // N3
|
||||
{0x41, 0xd82, 1, ProductNames::ARM_X4}, // X4
|
||||
{0x41, 0xd81, 1, ProductNames::ARM_A720}, // A720
|
||||
{0x41, 0xd4e, 1, ProductNames::ARM_X3}, // X3
|
||||
{0x41, 0xd4d, 1, ProductNames::ARM_A715}, // A715
|
||||
{0x41, 0xd4f, 1, ProductNames::ARM_V2}, // V2
|
||||
{0x41, 0xd4b, 1, ProductNames::ARM_A78C}, // A78C
|
||||
{0x41, 0xd4a, 1, ProductNames::ARM_E1}, // E1
|
||||
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
|
||||
{0x41, 0xd48, 1, ProductNames::ARM_X2}, // X2
|
||||
{0x41, 0xd47, 1, ProductNames::ARM_A710}, // A710
|
||||
{0x41, 0xd4C, 1, ProductNames::ARM_X1C}, // X1C
|
||||
{0x41, 0xd44, 1, ProductNames::ARM_X1}, // X1
|
||||
{0x41, 0xd42, 1, ProductNames::ARM_A78AE}, // A78AE
|
||||
{0x41, 0xd41, 1, ProductNames::ARM_A78}, // A78
|
||||
{0x41, 0xd40, 1, ProductNames::ARM_V1}, // V1
|
||||
{0x41, 0xd0e, 1, ProductNames::ARM_A76AE}, // A76AE
|
||||
{0x41, 0xd0d, 1, ProductNames::ARM_A77}, // A77
|
||||
{0x41, 0xd0c, 1, ProductNames::ARM_N1}, // N1
|
||||
{0x41, 0xd0b, 1, ProductNames::ARM_A76}, // A76
|
||||
{0x51, 0x804, 1, ProductNames::ARM_Kryo400}, // Kryo 4xx Gold (A76 based)
|
||||
{0x41, 0xd0a, 1, ProductNames::ARM_A75}, // A75
|
||||
{0x51, 0x802, 1, ProductNames::ARM_Kryo300}, // Kryo 3xx Gold (A75 based)
|
||||
{0x41, 0xd09, 1, ProductNames::ARM_A73}, // A73
|
||||
{0x51, 0x800, 1, ProductNames::ARM_Kryo200}, // Kryo 2xx Gold (A73 based)
|
||||
{0x41, 0xd08, 1, ProductNames::ARM_A72}, // A72
|
||||
{0x41, 0xd8c, 1, ProductNames::ARM_C1Ultra}, // C1-Ultra
|
||||
{0x41, 0xd90, 1, ProductNames::ARM_C1Premium}, // C1-Premium
|
||||
{0x41, 0xd8b, 1, ProductNames::ARM_C1Pro}, // C1-Pro
|
||||
{0x41, 0xd85, 1, ProductNames::ARM_X925}, // X925
|
||||
{0x41, 0xd87, 1, ProductNames::ARM_A725}, // A725
|
||||
{0x41, 0xd84, 1, ProductNames::ARM_V3}, // V3
|
||||
{0x41, 0xd83, 1, ProductNames::ARM_V3AE}, // V3AE
|
||||
{0x41, 0xd8e, 1, ProductNames::ARM_N3}, // N3
|
||||
{0x41, 0xd82, 1, ProductNames::ARM_X4}, // X4
|
||||
{0x41, 0xd81, 1, ProductNames::ARM_A720}, // A720
|
||||
{0x41, 0xd4e, 1, ProductNames::ARM_X3}, // X3
|
||||
{0x41, 0xd4d, 1, ProductNames::ARM_A715}, // A715
|
||||
{0x41, 0xd4f, 1, ProductNames::ARM_V2}, // V2
|
||||
{0x41, 0xd4b, 1, ProductNames::ARM_A78C}, // A78C
|
||||
{0x41, 0xd4a, 1, ProductNames::ARM_E1}, // E1
|
||||
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
|
||||
{0x41, 0xd48, 1, ProductNames::ARM_X2}, // X2
|
||||
{0x41, 0xd47, 1, ProductNames::ARM_A710}, // A710
|
||||
{0x41, 0xd4C, 1, ProductNames::ARM_X1C}, // X1C
|
||||
{0x41, 0xd44, 1, ProductNames::ARM_X1}, // X1
|
||||
{0x41, 0xd42, 1, ProductNames::ARM_A78AE}, // A78AE
|
||||
{0x41, 0xd41, 1, ProductNames::ARM_A78}, // A78
|
||||
{0x41, 0xd40, 1, ProductNames::ARM_V1}, // V1
|
||||
{0x41, 0xd0e, 1, ProductNames::ARM_A76AE}, // A76AE
|
||||
{0x41, 0xd0d, 1, ProductNames::ARM_A77}, // A77
|
||||
{0x41, 0xd0c, 1, ProductNames::ARM_N1}, // N1
|
||||
{0x41, 0xd0b, 1, ProductNames::ARM_A76}, // A76
|
||||
{0x51, 0x804, 1, ProductNames::ARM_Kryo400}, // Kryo 4xx Gold (A76 based)
|
||||
{0x41, 0xd0a, 1, ProductNames::ARM_A75}, // A75
|
||||
{0x51, 0x802, 1, ProductNames::ARM_Kryo300}, // Kryo 3xx Gold (A75 based)
|
||||
{0x41, 0xd09, 1, ProductNames::ARM_A73}, // A73
|
||||
{0x51, 0x800, 1, ProductNames::ARM_Kryo200}, // Kryo 2xx Gold (A73 based)
|
||||
{0x41, 0xd08, 1, ProductNames::ARM_A72}, // A72
|
||||
|
||||
{0x4e, 0x004, 1, ProductNames::ARM_Carmel}, // Carmel
|
||||
{0xc0, 0xac3, 1, ProductNames::ARM_Ampere_1}, // AmpereOne
|
||||
{0xc0, 0xac4, 1, ProductNames::ARM_Ampere_1A}, // AmpereOneA
|
||||
{0xc0, 0xac5, 1, ProductNames::ARM_Ampere_1B}, // AmpereOneB
|
||||
|
||||
{0x4e, 0x010, 1, ProductNames::ARM_Olympus}, // Olympus
|
||||
{0x4e, 0x004, 1, ProductNames::ARM_Carmel}, // Carmel
|
||||
|
||||
// Denver rated above A57 to match TX2 weirdness
|
||||
{0x4e, 0x003, 1, ProductNames::ARM_Denver}, // Denver
|
||||
@@ -227,6 +243,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
{0x61, 0x024, 0, ProductNames::ARM_Icestorm_M1Pro}, // Apple Icestorm (M1 Pro)
|
||||
{0x61, 0x022, 0, ProductNames::ARM_Icestorm_M1}, // Apple Icestorm (M1)
|
||||
|
||||
{0x41, 0xd8a, 1, ProductNames::ARM_C1Nano}, // C1-Nano
|
||||
{0x41, 0xd80, 0, ProductNames::ARM_A520}, // A520
|
||||
{0x41, 0xd46, 0, ProductNames::ARM_A510}, // A510
|
||||
{0x41, 0xd06, 0, ProductNames::ARM_A65}, // A65
|
||||
@@ -892,38 +909,38 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) con
|
||||
|
||||
Res.eax = FAMILY_IDENTIFIER;
|
||||
|
||||
Res.ecx = (1 << 0) | // LAHF/SAHF
|
||||
(1 << 1) | // 0 = Single core product, 1 = multi core product
|
||||
(0 << 2) | // SVM
|
||||
(1 << 3) | // Extended APIC register space
|
||||
(0 << 4) | // LOCK MOV CR0 means MOV CR8
|
||||
(1 << 5) | // ABM instructions
|
||||
(0 << 6) | // SSE4a
|
||||
(0 << 7) | // Misaligned SSE mode
|
||||
(1 << 8) | // PREFETCHW
|
||||
(0 << 9) | // OS visible workaround support
|
||||
(0 << 10) | // Instruction based sampling support
|
||||
(0 << 11) | // XOP
|
||||
(0 << 12) | // SKINIT
|
||||
(0 << 13) | // Watchdog timer support
|
||||
(0 << 14) | // Reserved
|
||||
(0 << 15) | // Lightweight profiling support
|
||||
(0 << 16) | // FMA4
|
||||
(1 << 17) | // Translation cache extension
|
||||
(0 << 18) | // Reserved
|
||||
(0 << 19) | // Reserved
|
||||
(0 << 20) | // Reserved
|
||||
(0 << 21) | // XOP-TBM
|
||||
(0 << 22) | // Topology extensions support
|
||||
(0 << 23) | // Core performance counter extensions
|
||||
(0 << 24) | // NB performance counter extensions
|
||||
(0 << 25) | // Reserved
|
||||
(0 << 26) | // Data breakpoints extensions
|
||||
(0 << 27) | // Performance TSC
|
||||
(0 << 28) | // L2 perf counter extensions
|
||||
(0 << 29) | // MONITORX
|
||||
(0 << 30) | // Reserved
|
||||
(0 << 31); // Reserved
|
||||
Res.ecx = (1 << 0) | // LAHF/SAHF
|
||||
(1 << 1) | // 0 = Single core product, 1 = multi core product
|
||||
(0 << 2) | // SVM
|
||||
(1 << 3) | // Extended APIC register space
|
||||
(0 << 4) | // LOCK MOV CR0 means MOV CR8
|
||||
(1 << 5) | // ABM instructions
|
||||
(CTX->HostFeatures.SupportsSSE4a << 6) | // SSE4a
|
||||
(0 << 7) | // Misaligned SSE mode
|
||||
(1 << 8) | // PREFETCHW
|
||||
(0 << 9) | // OS visible workaround support
|
||||
(0 << 10) | // Instruction based sampling support
|
||||
(0 << 11) | // XOP
|
||||
(0 << 12) | // SKINIT
|
||||
(0 << 13) | // Watchdog timer support
|
||||
(0 << 14) | // Reserved
|
||||
(0 << 15) | // Lightweight profiling support
|
||||
(0 << 16) | // FMA4
|
||||
(1 << 17) | // Translation cache extension
|
||||
(0 << 18) | // Reserved
|
||||
(0 << 19) | // Reserved
|
||||
(0 << 20) | // Reserved
|
||||
(0 << 21) | // XOP-TBM
|
||||
(0 << 22) | // Topology extensions support
|
||||
(0 << 23) | // Core performance counter extensions
|
||||
(0 << 24) | // NB performance counter extensions
|
||||
(0 << 25) | // Reserved
|
||||
(0 << 26) | // Data breakpoints extensions
|
||||
(0 << 27) | // Performance TSC
|
||||
(0 << 28) | // L2 perf counter extensions
|
||||
(0 << 29) | // MONITORX
|
||||
(0 << 30) | // Reserved
|
||||
(0 << 31); // Reserved
|
||||
|
||||
Res.edx = (1 << 0) | // FPU
|
||||
(1 << 1) | // Virtual mode extensions
|
||||
|
||||
@@ -0,0 +1,27 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <Interface/Context/Context.h>
|
||||
|
||||
#include <FEXCore/HLE/SourcecodeResolver.h>
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
ExecutableFileInfo::~ExecutableFileInfo() = default;
|
||||
|
||||
} // namespace FEXCore
|
||||
|
||||
namespace FEXCore::Context {
|
||||
|
||||
CodeCache::CodeCache(ContextImpl& CTX_)
|
||||
: CTX(CTX_) {}
|
||||
CodeCache::~CodeCache() = default;
|
||||
|
||||
void CodeCache::LoadData(Core::InternalThreadState& Thread, std::byte* MappedCacheFile, const ExecutableFileSectionInfo& GuestRIPLookup) {
|
||||
// TODO
|
||||
}
|
||||
|
||||
bool CodeCache::SaveData(Core::InternalThreadState& Thread, int fd, const ExecutableFileSectionInfo& SourceBinary, uint64_t SerializedBaseAddress) {
|
||||
// TODO
|
||||
return true;
|
||||
}
|
||||
|
||||
} // namespace FEXCore::Context
|
||||
@@ -18,6 +18,7 @@ $end_info$
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include <Interface/GDBJIT/GDBJIT.h>
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
@@ -56,20 +57,13 @@ $end_info$
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <atomic>
|
||||
#include <chrono>
|
||||
#include <condition_variable>
|
||||
#include <fcntl.h>
|
||||
#include <functional>
|
||||
#include <mutex>
|
||||
#include <queue>
|
||||
#include <shared_mutex>
|
||||
#include <signal.h>
|
||||
#include <stdio.h>
|
||||
#include <string_view>
|
||||
#include <sys/stat.h>
|
||||
#include <type_traits>
|
||||
#include <unistd.h>
|
||||
#include <unordered_map>
|
||||
#include <utility>
|
||||
#include <xxhash.h>
|
||||
|
||||
@@ -77,7 +71,7 @@ namespace FEXCore::Context {
|
||||
ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
|
||||
: HostFeatures {Features}
|
||||
, CPUID {this}
|
||||
, IRCaptureCache {this} {
|
||||
, CodeCache {*this} {
|
||||
if (!Config.Is64BitMode()) {
|
||||
// When operating in 32-bit mode, the virtual memory we care about is only the lower 32-bits.
|
||||
Config.VirtualMemSize = 1ULL << 32;
|
||||
@@ -99,6 +93,8 @@ ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
|
||||
|
||||
// Track atomic TSO emulation configuration.
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
// Ensure X87 precision constraints are respected.
|
||||
UpdateX87PrecisionConfig();
|
||||
}
|
||||
|
||||
struct GetFrameBlockInfoResult {
|
||||
@@ -610,7 +606,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
} else {
|
||||
ForceTSO = IR::ForceTSOMode::ForceDisabled;
|
||||
}
|
||||
} else if (DecodedInfo->ForceTSO) {
|
||||
} else if (DecodedInfo->Flags & X86Tables::DecodeFlags::FLAG_FORCE_TSO) {
|
||||
ForceTSO = IR::ForceTSOMode::ForceEnabled;
|
||||
}
|
||||
|
||||
@@ -708,9 +704,9 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst) {
|
||||
if (SourcecodeResolver && Config.GDBSymbols()) {
|
||||
auto AOTIRCacheEntry = SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
|
||||
if (AOTIRCacheEntry.Entry) {
|
||||
AOTIRCacheEntry.Entry->SourcecodeMap = SourcecodeResolver->GenerateMap(AOTIRCacheEntry.Entry->Filename, AOTIRCacheEntry.Entry->FileId);
|
||||
auto MappedSection = SyscallHandler->LookupExecutableFileSection(*Thread, GuestRIP);
|
||||
if (MappedSection) {
|
||||
MappedSection->FileInfo.SourcecodeMap = SourcecodeResolver->GenerateMap(MappedSection->FileInfo.Filename, MappedSection->FileInfo.FileId);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -785,36 +781,44 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
if (Config.BlockJITNaming()) {
|
||||
auto FragmentBasePtr = CompiledCode.BlockBegin;
|
||||
|
||||
if (DebugData) {
|
||||
auto GuestRIPLookup = SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
|
||||
auto GuestRIPLookup = SyscallHandler->LookupExecutableFileSection(*Thread, GuestRIP);
|
||||
|
||||
if (DebugData->Subblocks.size()) {
|
||||
for (auto& Subblock : DebugData->Subblocks) {
|
||||
auto BlockBasePtr = FragmentBasePtr + Subblock.HostCodeOffset;
|
||||
if (GuestRIPLookup.Entry) {
|
||||
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, CompiledCode.Size, GuestRIPLookup.Entry->Filename,
|
||||
GuestRIP - GuestRIPLookup.VAFileStart);
|
||||
} else {
|
||||
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, GuestRIP, Subblock.HostCodeSize);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
if (GuestRIPLookup.Entry) {
|
||||
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, CompiledCode.Size, GuestRIPLookup.Entry->Filename,
|
||||
GuestRIP - GuestRIPLookup.VAFileStart);
|
||||
if (DebugData->Subblocks.size()) {
|
||||
for (auto& Subblock : DebugData->Subblocks) {
|
||||
auto BlockBasePtr = FragmentBasePtr + Subblock.HostCodeOffset;
|
||||
if (GuestRIPLookup) {
|
||||
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, CompiledCode.Size, GuestRIPLookup->FileInfo.Filename,
|
||||
GuestRIP - GuestRIPLookup->FileStartVA);
|
||||
} else {
|
||||
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, GuestRIP, CompiledCode.Size);
|
||||
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, GuestRIP, Subblock.HostCodeSize);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
if (GuestRIPLookup) {
|
||||
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, CompiledCode.Size, GuestRIPLookup->FileInfo.Filename,
|
||||
GuestRIP - GuestRIPLookup->FileStartVA);
|
||||
} else {
|
||||
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, GuestRIP, CompiledCode.Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (Config.LibraryJITNaming() || Config.GDBSymbols()) {
|
||||
auto MappedSection = SyscallHandler->LookupExecutableFileSection(*Thread, GuestRIP);
|
||||
if (MappedSection) {
|
||||
if (Config.LibraryJITNaming()) {
|
||||
Symbols.RegisterNamedRegion(Thread->SymbolBuffer.get(), CodePtr, DebugData->HostCodeSize, MappedSection->FileInfo.Filename);
|
||||
}
|
||||
|
||||
if (Config.GDBSymbols()) {
|
||||
GDBJITRegister(MappedSection->FileInfo, MappedSection->FileStartVA, GuestRIP, (uintptr_t)CodePtr, *DebugData);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Clear any relocations that might have been generated
|
||||
Thread->CPUBackend->ClearRelocations();
|
||||
|
||||
if (IRCaptureCache.PostCompileCode(Thread, CompiledCode.BlockBegin, GuestRIP, StartAddr, Length, DebugData.get())) {
|
||||
// Early exit
|
||||
return (uintptr_t)CodePtr;
|
||||
if (!CodeCache.IsGeneratingCache) {
|
||||
Thread->CPUBackend->ClearRelocations();
|
||||
}
|
||||
|
||||
if (NeedsAddGuestCodeRanges) {
|
||||
@@ -893,23 +897,6 @@ void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* T
|
||||
InvalidateGuestThreadCodeRange(Thread, Accumulator, Start, Length);
|
||||
}
|
||||
|
||||
void ContextImpl::MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) {
|
||||
if (!Thread) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (!IsMemoryShared) {
|
||||
IsMemoryShared = true;
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
|
||||
if (Config.TSOAutoMigration) {
|
||||
// Only the lookup cache is cleared here, so that old code can keep running until next compilation.
|
||||
// This will leak previously compiled blocks until the CodeBuffer is cleared for some other reason.
|
||||
Thread->LookupCache->ClearCache();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
bool ContextImpl::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) {
|
||||
LogMan::Throw::AFmt(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to "
|
||||
"be unique_locked here");
|
||||
@@ -960,10 +947,10 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu
|
||||
|
||||
if (GPRSize == IR::OpSize::i64Bit) {
|
||||
IR::Ref R = emit->_StoreRegister(emit->Constant(Entrypoint), GPRSize);
|
||||
R->Reg = IR::PhysicalRegister(IR::GPRFixedClass, X86State::REG_R11).Raw;
|
||||
R->Reg = IR::PhysicalRegister(IR::RegClass::GPRFixed, X86State::REG_R11).Raw;
|
||||
} else {
|
||||
emit->_StoreContext(GPRSize, IR::FPRClass, emit->_VCastFromGPR(IR::OpSize::i64Bit, IR::OpSize::i64Bit, emit->Constant(Entrypoint)),
|
||||
offsetof(Core::CPUState, mm[0][0]));
|
||||
emit->_StoreContextFPR(GPRSize, emit->_VCastFromGPR(IR::OpSize::i64Bit, IR::OpSize::i64Bit, emit->Constant(Entrypoint)),
|
||||
offsetof(Core::CPUState, mm[0][0]));
|
||||
}
|
||||
emit->_ExitFunction(IR::OpSize::i64Bit, emit->Constant(GuestThunkEntrypoint), IR::BranchHint::None, emit->Invalid(), emit->Invalid());
|
||||
},
|
||||
@@ -1015,9 +1002,9 @@ void ContextImpl::MonoBackpatcherWrite(FEXCore::Core::CpuStateFrame* Frame, uint
|
||||
auto lk = GuardSignalDeferringSection(CTX->CodeInvalidationMutex, Thread);
|
||||
|
||||
if (Size == 8) {
|
||||
*reinterpret_cast<uint64_t *>(Address) = Value;
|
||||
*reinterpret_cast<uint64_t*>(Address) = Value;
|
||||
} else if (Size == 4) {
|
||||
*reinterpret_cast<uint32_t *>(Address) = Value;
|
||||
*reinterpret_cast<uint32_t*>(Address) = Value;
|
||||
} else {
|
||||
ERROR_AND_DIE_FMT("Unexpected write size for backpatcher: {}", Size);
|
||||
}
|
||||
@@ -1026,13 +1013,6 @@ void ContextImpl::MonoBackpatcherWrite(FEXCore::Core::CpuStateFrame* Frame, uint
|
||||
CTX->SyscallHandler->InvalidateGuestCodeRange(Thread, Address, Size);
|
||||
}
|
||||
|
||||
IR::AOTIRCacheEntry* ContextImpl::LoadAOTIRCacheEntry(const fextl::string& filename) {
|
||||
auto rv = IRCaptureCache.LoadAOTIRCacheEntry(filename);
|
||||
return rv;
|
||||
}
|
||||
|
||||
void ContextImpl::UnloadAOTIRCacheEntry(IR::AOTIRCacheEntry* Entry) {}
|
||||
|
||||
void ContextImpl::ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) {
|
||||
Thread->FrontendDecoder->SetExternalBranches(ExternalBranches);
|
||||
Thread->FrontendDecoder->SetSectionMaxAddress(SectionMaxAddress);
|
||||
|
||||
@@ -17,6 +17,7 @@
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
|
||||
|
||||
@@ -259,13 +259,13 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
|
||||
|
||||
if (HasSIB) {
|
||||
FEXCore::X86Tables::SIBDecoded SIB;
|
||||
if (DecodeInst->DecodedSIB) {
|
||||
if (DecodeInst->Flags & DecodeFlags::FLAG_DECODED_SIB) {
|
||||
SIB.Hex = DecodeInst->SIB;
|
||||
} else {
|
||||
// Haven't yet grabbed SIB, pull it now
|
||||
DecodeInst->SIB = ReadByte();
|
||||
SIB.Hex = DecodeInst->SIB;
|
||||
DecodeInst->DecodedSIB = true;
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_SIB;
|
||||
}
|
||||
|
||||
// If the SIB base is 0b101, aka BP or R13 then we have a 32bit displacement
|
||||
@@ -401,9 +401,9 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
|
||||
// If we require ModRM and haven't decoded it yet, do it now
|
||||
// Some instructions have to read modrm upfront, others do it later
|
||||
if (HasMODRM && !DecodeInst->DecodedModRM) {
|
||||
if (HasMODRM && !(DecodeInst->Flags & DecodeFlags::FLAG_DECODED_MODRM)) {
|
||||
DecodeInst->ModRM = ReadByte();
|
||||
DecodeInst->DecodedModRM = true;
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
|
||||
}
|
||||
|
||||
// New instruction size decoding
|
||||
@@ -436,9 +436,8 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
// If the default operating mode is 32bit and we have the operand size flag then the operating size drops to 16bit
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_16BIT);
|
||||
DestSize = 2;
|
||||
} else if ((HasXMMDst || HasMMDst || BlockInfo.Is64BitMode) &&
|
||||
(HasWideningDisplacement || DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BIT ||
|
||||
DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BITDEF)) {
|
||||
} else if ((HasXMMDst || HasMMDst || BlockInfo.Is64BitMode) && (HasWideningDisplacement || DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BIT ||
|
||||
DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BITDEF)) {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_64BIT);
|
||||
DestSize = 8;
|
||||
} else {
|
||||
@@ -465,9 +464,8 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
// See table 1-2. Operand-Size Overrides for this decoding
|
||||
// If the default operating mode is 32bit and we have the operand size flag then the operating size drops to 16bit
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_16BIT);
|
||||
} else if ((HasXMMSrc || HasMMSrc || BlockInfo.Is64BitMode) &&
|
||||
(HasWideningDisplacement || SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BIT ||
|
||||
SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BITDEF)) {
|
||||
} else if ((HasXMMSrc || HasMMSrc || BlockInfo.Is64BitMode) && (HasWideningDisplacement || SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BIT ||
|
||||
SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BITDEF)) {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_64BIT);
|
||||
} else {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_32BIT);
|
||||
@@ -636,11 +634,20 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
Literal = static_cast<int32_t>(Literal);
|
||||
}
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Size = DestSize;
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.SignExtend = true;
|
||||
}
|
||||
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::Literal;
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal;
|
||||
++CurrentSrc;
|
||||
|
||||
if (Bytes == 8) [[unlikely]] {
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Size = 4;
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::Literal;
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal >> 32;
|
||||
}
|
||||
|
||||
Bytes = 0;
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::Literal;
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Bytes == 0, "Inst at 0x{:x}: 0x{:04x} '{}' Had an instruction of size {} with {} remaining", DecodeInst->PC,
|
||||
@@ -672,7 +679,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
} else if (Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_11) {
|
||||
uint8_t ModRMByte = ReadByte();
|
||||
DecodeInst->ModRM = ModRMByte;
|
||||
DecodeInst->DecodedModRM = true;
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
|
||||
|
||||
FEXCore::X86Tables::ModRMDecoded ModRM;
|
||||
ModRM.Hex = DecodeInst->ModRM;
|
||||
@@ -689,18 +696,18 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
constexpr uint16_t PF_F2 = 3;
|
||||
|
||||
uint16_t PrefixType = PF_NONE;
|
||||
if (DecodeInst->LastEscapePrefix == 0xF3) {
|
||||
if (LastEscapePrefix == 0xF3) {
|
||||
PrefixType = PF_F3;
|
||||
} else if (DecodeInst->LastEscapePrefix == 0xF2) {
|
||||
} else if (LastEscapePrefix == 0xF2) {
|
||||
PrefixType = PF_F2;
|
||||
} else if (DecodeInst->LastEscapePrefix == 0x66) {
|
||||
} else if (LastEscapePrefix == 0x66) {
|
||||
PrefixType = PF_66;
|
||||
}
|
||||
|
||||
// We have ModRM
|
||||
uint8_t ModRMByte = ReadByte();
|
||||
DecodeInst->ModRM = ModRMByte;
|
||||
DecodeInst->DecodedModRM = true;
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
|
||||
|
||||
FEXCore::X86Tables::ModRMDecoded ModRM;
|
||||
ModRM.Hex = DecodeInst->ModRM;
|
||||
@@ -727,7 +734,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
// We have ModRM
|
||||
uint8_t ModRMByte = ReadByte();
|
||||
DecodeInst->ModRM = ModRMByte;
|
||||
DecodeInst->DecodedModRM = true;
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
|
||||
|
||||
uint16_t X87Op = ((Op - 0xD8) << 8) | ModRMByte;
|
||||
return NormalOp(&(*X87Table)[X87Op], X87Op);
|
||||
@@ -789,7 +796,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
// We have ModRM
|
||||
uint8_t ModRMByte = ReadByte();
|
||||
DecodeInst->ModRM = ModRMByte;
|
||||
DecodeInst->DecodedModRM = true;
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
|
||||
|
||||
FEXCore::X86Tables::ModRMDecoded ModRM;
|
||||
ModRM.Hex = DecodeInst->ModRM;
|
||||
@@ -813,6 +820,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
|
||||
bool Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
InstructionSize = 0;
|
||||
LastEscapePrefix = 0;
|
||||
Instruction.fill(0);
|
||||
|
||||
DecodeInst = &DecodedBuffer[DecodedSize];
|
||||
@@ -834,7 +842,7 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
// Decode ModRM
|
||||
uint8_t ModRMByte = ReadByte();
|
||||
DecodeInst->ModRM = ModRMByte;
|
||||
DecodeInst->DecodedModRM = true;
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
|
||||
|
||||
FEXCore::X86Tables::ModRMDecoded ModRM;
|
||||
ModRM.Hex = DecodeInst->ModRM;
|
||||
@@ -873,7 +881,7 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
uint16_t LocalOp = (Prefix << 8) | ReadByte();
|
||||
|
||||
bool NoOverlay66 = (FEXCore::X86Tables::H0F38TableOps[LocalOp].Flags & InstFlags::FLAGS_NO_OVERLAY66) != 0;
|
||||
if (DecodeInst->LastEscapePrefix == 0x66 && NoOverlay66) { // Operand Size
|
||||
if (LastEscapePrefix == 0x66 && NoOverlay66) { // Operand Size
|
||||
// Remove prefix so it doesn't effect calculations.
|
||||
// This is only an escape prefix rather than modifier now
|
||||
DecodeInst->Flags &= ~DecodeFlags::FLAG_OPERAND_SIZE;
|
||||
@@ -889,7 +897,7 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
constexpr uint16_t PF_3A_REX = (1 << 1);
|
||||
|
||||
uint16_t Prefix = PF_3A_NONE;
|
||||
if (DecodeInst->LastEscapePrefix == 0x66) { // Operand Size
|
||||
if (LastEscapePrefix == 0x66) { // Operand Size
|
||||
Prefix = PF_3A_66;
|
||||
}
|
||||
|
||||
@@ -915,17 +923,17 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
|
||||
if (NoOverlay) { // This section of the table ignores prefix extention
|
||||
return NormalOpHeader(&FEXCore::X86Tables::SecondBaseOps[EscapeOp], EscapeOp);
|
||||
} else if (DecodeInst->LastEscapePrefix == 0xF3) { // REP
|
||||
} else if (LastEscapePrefix == 0xF3) { // REP
|
||||
// Remove prefix so it doesn't effect calculations.
|
||||
// This is only an escape prefix rather tan modifier now
|
||||
DecodeInst->Flags &= ~DecodeFlags::FLAG_REP_PREFIX;
|
||||
return NormalOpHeader(&FEXCore::X86Tables::RepModOps[EscapeOp], EscapeOp);
|
||||
} else if (DecodeInst->LastEscapePrefix == 0xF2) { // REPNE
|
||||
} else if (LastEscapePrefix == 0xF2) { // REPNE
|
||||
// Remove prefix so it doesn't effect calculations.
|
||||
// This is only an escape prefix rather tan modifier now
|
||||
DecodeInst->Flags &= ~DecodeFlags::FLAG_REPNE_PREFIX;
|
||||
return NormalOpHeader(&FEXCore::X86Tables::RepNEModOps[EscapeOp], EscapeOp);
|
||||
} else if (DecodeInst->LastEscapePrefix == 0x66 && !NoOverlay66) { // Operand Size
|
||||
} else if (LastEscapePrefix == 0x66 && !NoOverlay66) { // Operand Size
|
||||
// Remove prefix so it doesn't effect calculations.
|
||||
// This is only an escape prefix rather tan modifier now
|
||||
DecodeInst->Flags &= ~DecodeFlags::FLAG_OPERAND_SIZE;
|
||||
@@ -941,7 +949,7 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
}
|
||||
case 0x66: // Operand Size prefix
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_OPERAND_SIZE;
|
||||
DecodeInst->LastEscapePrefix = Op;
|
||||
LastEscapePrefix = Op;
|
||||
DecodeFlags::PushOpAddr(&DecodeInst->Flags, DecodeFlags::FLAG_OPERAND_SIZE_LAST);
|
||||
break;
|
||||
case 0x67: // Address Size override prefix
|
||||
@@ -972,11 +980,11 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
break;
|
||||
case 0xF2: // REPNE prefix
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REPNE_PREFIX;
|
||||
DecodeInst->LastEscapePrefix = Op;
|
||||
LastEscapePrefix = Op;
|
||||
break;
|
||||
case 0xF3: // REP prefix
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REP_PREFIX;
|
||||
DecodeInst->LastEscapePrefix = Op;
|
||||
LastEscapePrefix = Op;
|
||||
break;
|
||||
case 0x64: // FS prefix
|
||||
DecodeInst->Flags = (DecodeInst->Flags & ~FEXCore::X86Tables::DecodeFlags::FLAG_SEGMENTS) | DecodeFlags::FLAG_FS_PREFIX;
|
||||
@@ -1055,10 +1063,10 @@ Decoder::DecodedBlockStatus Decoder::DecodeInstruction(uint64_t PC) {
|
||||
|
||||
if (DecodeInst->OP == 0x8b && DecodeInst->Src[0].IsGPRIndirect() &&
|
||||
IsKnownAtomicDisplacement(DecodeInst->Src[0].Data.GPRIndirect.Displacement)) {
|
||||
DecodeInst->ForceTSO = true;
|
||||
DecodeInst->Flags |= X86Tables::DecodeFlags::FLAG_FORCE_TSO;
|
||||
}
|
||||
if (DecodeInst->OP == 0x89 && DecodeInst->Dest.IsGPRIndirect() && IsKnownAtomicDisplacement(DecodeInst->Dest.Data.GPRIndirect.Displacement)) {
|
||||
DecodeInst->ForceTSO = true;
|
||||
DecodeInst->Flags |= X86Tables::DecodeFlags::FLAG_FORCE_TSO;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1301,7 +1309,7 @@ const uint8_t* Decoder::AdjustAddrForSpecialRegion(const uint8_t* _InstStream, u
|
||||
return _InstStream - EntryPoint + RIP;
|
||||
}
|
||||
|
||||
void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState *Thread, const uint8_t* _InstStream, uint64_t PC, uint64_t MaxInst) {
|
||||
void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, const uint8_t* _InstStream, uint64_t PC, uint64_t MaxInst) {
|
||||
FEXCORE_PROFILE_SCOPED("DecodeInstructions");
|
||||
BlockInfo.TotalInstructionCount = 0;
|
||||
BlockInfo.Blocks.clear();
|
||||
|
||||
@@ -49,7 +49,7 @@ public:
|
||||
};
|
||||
|
||||
Decoder(FEXCore::Core::InternalThreadState* Thread);
|
||||
void DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState *Thread, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst);
|
||||
void DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst);
|
||||
|
||||
const DecodedBlockInformation* GetDecodedBlockInfo() const {
|
||||
return &BlockInfo;
|
||||
@@ -125,6 +125,7 @@ private:
|
||||
static constexpr size_t MAX_INST_SIZE = 15;
|
||||
uint8_t InstructionSize {};
|
||||
std::array<uint8_t, MAX_INST_SIZE> Instruction;
|
||||
uint8_t LastEscapePrefix {};
|
||||
FEXCore::X86Tables::DecodedInst* DecodeInst;
|
||||
|
||||
// This is for multiblock data tracking
|
||||
|
||||
@@ -2,11 +2,13 @@
|
||||
#pragma once
|
||||
#include "Common/SoftFloat.h"
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Interpreter/Fallbacks/FallbackOpHandler.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/SHMStats.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static softfloat_state SoftFloatStateFromFCW(uint16_t FCW, bool Force80BitPrecision = false) {
|
||||
@@ -77,6 +79,12 @@ struct OpHandlers<IR::OP_F80CVTTO> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle8(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
ScopedSoftFloatState State {FCW, Frame};
|
||||
auto Context = static_cast<Context::ContextImpl*>(Frame->Thread->CTX);
|
||||
auto ReducedPrecisionMode = Context->Config.x87ReducedPrecision;
|
||||
auto StrictReducedPrecisionMode = Context->Config.x87StrictReducedPrecision;
|
||||
if (!ReducedPrecisionMode || StrictReducedPrecisionMode) {
|
||||
return X80SoftFloat::FromF64_PreserveNaN(&State.State, src);
|
||||
}
|
||||
return X80SoftFloat(&State.State, src);
|
||||
}
|
||||
};
|
||||
@@ -115,6 +123,12 @@ struct OpHandlers<IR::OP_F80CVT> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle8(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
ScopedSoftFloatState State {FCW, Frame};
|
||||
auto Context = static_cast<Context::ContextImpl*>(Frame->Thread->CTX);
|
||||
auto ReducedPrecisionMode = Context->Config.x87ReducedPrecision;
|
||||
auto StrictReducedPrecisionMode = Context->Config.x87StrictReducedPrecision;
|
||||
if (!ReducedPrecisionMode || StrictReducedPrecisionMode) {
|
||||
return X80SoftFloat(src).ToF64_PreserveNan(&State.State);
|
||||
}
|
||||
return X80SoftFloat(src).ToF64(&State.State);
|
||||
}
|
||||
};
|
||||
|
||||
@@ -87,12 +87,10 @@ void InterpreterOps::FillFallbackIndexPointers(Core::FallbackABIInfo* Info, uint
|
||||
Info[Core::OPINDEX_F64SINCOS] = {ABIHandlers[FABI_F64x2_F64_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64SINCOS>::handle)};
|
||||
Info[Core::OPINDEX_F64TAN] = {ABIHandlers[FABI_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64TAN>::handle)};
|
||||
Info[Core::OPINDEX_F64F2XM1] = {ABIHandlers[FABI_F64_F64_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64F2XM1>::handle)};
|
||||
Info[Core::OPINDEX_F64F2XM1] = {ABIHandlers[FABI_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64F2XM1>::handle)};
|
||||
|
||||
// Double Precision Binary
|
||||
Info[Core::OPINDEX_F64ATAN] = {ABIHandlers[FABI_F64_F64_F64_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64ATAN>::handle)};
|
||||
Info[Core::OPINDEX_F64ATAN] = {ABIHandlers[FABI_F64_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64ATAN>::handle)};
|
||||
Info[Core::OPINDEX_F64FPREM] = {ABIHandlers[FABI_F64_F64_F64_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM>::handle)};
|
||||
Info[Core::OPINDEX_F64FPREM1] = {ABIHandlers[FABI_F64_F64_F64_PTR],
|
||||
@@ -220,21 +218,21 @@ bool InterpreterOps::GetFallbackHandler(const IR::IROp_Header* IROp, FallbackInf
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_UNARY_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
#define COMMON_UNARY_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = {FABI_F64_F64_PTR, Core::OPINDEX_F64##OP}; \
|
||||
return true; \
|
||||
return true; \
|
||||
}
|
||||
#define COMMON_UNARYPAIR_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
#define COMMON_UNARYPAIR_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = {FABI_F64x2_F64_PTR, Core::OPINDEX_F64##OP}; \
|
||||
return true; \
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_BINARY_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
#define COMMON_BINARY_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = {FABI_F64_F64_F64_PTR, Core::OPINDEX_F64##OP}; \
|
||||
return true; \
|
||||
return true; \
|
||||
}
|
||||
|
||||
// Unary
|
||||
|
||||
@@ -372,7 +372,7 @@ DEF_OP(CondSubNZCV) {
|
||||
DEF_OP(Neg) {
|
||||
auto Op = IROp->C<IR::IROp_Neg>();
|
||||
|
||||
if (Op->Cond == FEXCore::IR::COND_AL) {
|
||||
if (Op->Cond == IR::CondClass::AL) {
|
||||
neg(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src));
|
||||
} else {
|
||||
cneg(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src), MapCC(Op->Cond));
|
||||
@@ -1046,24 +1046,19 @@ DEF_OP(Popcount) {
|
||||
|
||||
if (CTX->HostFeatures.SupportsCSSC) {
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit:
|
||||
uxtb(ARMEmitter::Size::i32Bit, Dst, Src);
|
||||
cnt(ARMEmitter::Size::i32Bit, Dst, Dst);
|
||||
break;
|
||||
case IR::OpSize::i16Bit:
|
||||
uxth(ARMEmitter::Size::i32Bit, Dst, Src);
|
||||
cnt(ARMEmitter::Size::i32Bit, Dst, Dst);
|
||||
break;
|
||||
case IR::OpSize::i32Bit:
|
||||
cnt(ARMEmitter::Size::i32Bit, Dst, Src);
|
||||
break;
|
||||
case IR::OpSize::i64Bit:
|
||||
cnt(ARMEmitter::Size::i64Bit, Dst, Src);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unsupported Popcount size: {}", OpSize);
|
||||
case IR::OpSize::i8Bit:
|
||||
uxtb(ARMEmitter::Size::i32Bit, Dst, Src);
|
||||
cnt(ARMEmitter::Size::i32Bit, Dst, Dst);
|
||||
break;
|
||||
case IR::OpSize::i16Bit:
|
||||
uxth(ARMEmitter::Size::i32Bit, Dst, Src);
|
||||
cnt(ARMEmitter::Size::i32Bit, Dst, Dst);
|
||||
break;
|
||||
case IR::OpSize::i32Bit: cnt(ARMEmitter::Size::i32Bit, Dst, Src); break;
|
||||
case IR::OpSize::i64Bit: cnt(ARMEmitter::Size::i64Bit, Dst, Src); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unsupported Popcount size: {}", OpSize);
|
||||
}
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit:
|
||||
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), Src);
|
||||
@@ -1195,6 +1190,19 @@ DEF_OP(Rev) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Rbit) {
|
||||
auto Op = IROp->C<IR::IROp_Rbit>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = ConvertSize48(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src);
|
||||
|
||||
rbit(EmitSize, Dst, Src);
|
||||
}
|
||||
|
||||
DEF_OP(Bfi) {
|
||||
auto Op = IROp->C<IR::IROp_Bfi>();
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
@@ -1278,6 +1286,16 @@ DEF_OP(Sbfe) {
|
||||
sbfx(ConvertSize(IROp), Dst, Src, Op->lsb, Op->Width);
|
||||
}
|
||||
|
||||
DEF_OP(MaskGenerateFromBitWidth) {
|
||||
auto Op = IROp->C<IR::IROp_MaskGenerateFromBitWidth>();
|
||||
auto BitWidth = GetReg(Op->BitWidth);
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, -1);
|
||||
cmp(ARMEmitter::Size::i64Bit, BitWidth, 0);
|
||||
lslv(ARMEmitter::Size::i64Bit, TMP2, TMP1, BitWidth);
|
||||
csinv(ARMEmitter::Size::i64Bit, GetReg(Node), TMP1, TMP2, ARMEmitter::Condition::CC_EQ);
|
||||
}
|
||||
|
||||
DEF_OP(Select) {
|
||||
auto Op = IROp->C<IR::IROp_Select>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -81,35 +81,32 @@ void Arm64JITCore::InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constan
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations,
|
||||
const char* EntryRelocations) {
|
||||
size_t DataIndex {};
|
||||
for (size_t j = 0; j < NumRelocations; ++j) {
|
||||
const FEXCore::CPU::Relocation* Reloc = reinterpret_cast<const FEXCore::CPU::Relocation*>(&EntryRelocations[DataIndex]);
|
||||
LOGMAN_THROW_A_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
|
||||
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, std::span<std::byte> Code, std::span<const FEXCore::CPU::Relocation> Relocations) {
|
||||
const auto OrigBase = GetBufferBase();
|
||||
const auto OrigSize = GetBufferSize();
|
||||
const auto OrigOffset = GetCursorOffset();
|
||||
|
||||
switch (Reloc->Header.Type) {
|
||||
SetBuffer(reinterpret_cast<std::uint8_t*>(Code.data()), Code.size_bytes());
|
||||
for (auto& Reloc : Relocations) {
|
||||
switch (Reloc.Header.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Reloc->NamedSymbolLiteral.Symbol);
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Reloc.NamedSymbolLiteral.Symbol);
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor
|
||||
SetCursorOffset(CursorEntry + Reloc->NamedSymbolLiteral.Offset);
|
||||
SetCursorOffset(Reloc.NamedSymbolLiteral.Offset);
|
||||
|
||||
// Generate a literal so we can place it
|
||||
dc64(Pointer);
|
||||
|
||||
DataIndex += sizeof(Reloc->NamedSymbolLiteral);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc->NamedThunkMove.Symbol));
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc.NamedThunkMove.Symbol));
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
SetCursorOffset(CursorEntry + Reloc->NamedThunkMove.Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->NamedThunkMove.RegisterIndex), Pointer, true);
|
||||
DataIndex += sizeof(Reloc->NamedThunkMove);
|
||||
SetCursorOffset(Reloc.NamedThunkMove.Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.NamedThunkMove.RegisterIndex), Pointer, true);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
|
||||
@@ -117,18 +114,27 @@ bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uin
|
||||
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
|
||||
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
|
||||
if (Pointer == ~0ULL) {
|
||||
SetBuffer(OrigBase, OrigSize);
|
||||
SetCursorOffset(OrigOffset);
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
SetCursorOffset(CursorEntry + Reloc->GuestRIPMove.Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->GuestRIPMove.RegisterIndex), Pointer, true);
|
||||
DataIndex += sizeof(Reloc->GuestRIPMove);
|
||||
SetCursorOffset(Reloc.GuestRIPMove.Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.GuestRIPMove.RegisterIndex), Pointer, true);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
SetBuffer(OrigBase, OrigSize);
|
||||
SetCursorOffset(OrigOffset);
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
fextl::vector<FEXCore::CPU::Relocation> Arm64JITCore::TakeRelocations() {
|
||||
return std::move(Relocations);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -235,16 +235,16 @@ DEF_OP(CondJump) {
|
||||
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1), "CondJump: Expected GPR");
|
||||
LOGMAN_THROW_A_FMT(isConst, "CondJump: Expected constant source");
|
||||
|
||||
if (Op->Cond.Val == FEXCore::IR::COND_EQ) {
|
||||
if (Op->Cond == IR::CondClass::EQ) {
|
||||
LOGMAN_THROW_A_FMT(Const == 0, "CondJump: Expected 0 source");
|
||||
cbz(Size, Reg, TrueTargetLabel);
|
||||
} else if (Op->Cond.Val == FEXCore::IR::COND_NEQ) {
|
||||
} else if (Op->Cond == IR::CondClass::NEQ) {
|
||||
LOGMAN_THROW_A_FMT(Const == 0, "CondJump: Expected 0 source");
|
||||
cbnz(Size, Reg, TrueTargetLabel);
|
||||
} else if (Op->Cond.Val == FEXCore::IR::COND_TSTZ) {
|
||||
} else if (Op->Cond == IR::CondClass::TSTZ) {
|
||||
LOGMAN_THROW_A_FMT(Const < 64, "CondJump: Expected valid bit source");
|
||||
tbz(Reg, Const, TrueTargetLabel);
|
||||
} else if (Op->Cond.Val == FEXCore::IR::COND_TSTNZ) {
|
||||
} else if (Op->Cond == IR::CondClass::TSTNZ) {
|
||||
LOGMAN_THROW_A_FMT(Const < 64, "CondJump: Expected valid bit source");
|
||||
tbnz(Reg, Const, TrueTargetLabel);
|
||||
} else {
|
||||
|
||||
@@ -423,11 +423,11 @@ DEF_OP(Vector_FToI) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val: frintn(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val: frintm(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val: frintp(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val: frintz(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case FEXCore::IR::Round_Host.Val: frinti(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::Nearest: frintn(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::NegInfinity: frintm(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::PosInfinity: frintp(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::TowardsZero: frintz(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::Host: frinti(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
}
|
||||
} else {
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
@@ -449,21 +449,21 @@ DEF_OP(Vector_FToI) {
|
||||
}
|
||||
|
||||
switch (Op->Round) {
|
||||
case IR::Round_Nearest.Val: ROUNDING_FN(frintn); break;
|
||||
case IR::Round_Negative_Infinity.Val: ROUNDING_FN(frintm); break;
|
||||
case IR::Round_Positive_Infinity.Val: ROUNDING_FN(frintp); break;
|
||||
case IR::Round_Towards_Zero.Val: ROUNDING_FN(frintz); break;
|
||||
case IR::Round_Host.Val: ROUNDING_FN(frinti); break;
|
||||
case IR::RoundMode::Nearest: ROUNDING_FN(frintn); break;
|
||||
case IR::RoundMode::NegInfinity: ROUNDING_FN(frintm); break;
|
||||
case IR::RoundMode::PosInfinity: ROUNDING_FN(frintp); break;
|
||||
case IR::RoundMode::TowardsZero: ROUNDING_FN(frintz); break;
|
||||
case IR::RoundMode::Host: ROUNDING_FN(frinti); break;
|
||||
}
|
||||
|
||||
#undef ROUNDING_FN
|
||||
} else {
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val: frintn(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val: frintm(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val: frintp(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val: frintz(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Host.Val: frinti(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::Nearest: frintn(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::NegInfinity: frintm(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::PosInfinity: frintp(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::TowardsZero: frintz(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::Host: frinti(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -539,11 +539,11 @@ DEF_OP(Vector_F64ToI32) {
|
||||
// Then convert to integers using fcvtzs.
|
||||
auto CVTReg = Dst.Z();
|
||||
switch (Round) {
|
||||
case IR::Round_Nearest.Val: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::Round_Negative_Infinity.Val: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::Round_Positive_Infinity.Val: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::Round_Towards_Zero.Val: CVTReg = Vector.Z(); break;
|
||||
case IR::Round_Host.Val: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::Nearest: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::NegInfinity: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::PosInfinity: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::TowardsZero: CVTReg = Vector.Z(); break;
|
||||
case IR::RoundMode::Host: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
}
|
||||
|
||||
fcvtzs(Dst.Z(), ARMEmitter::SubRegSize::i32Bit, Mask, CVTReg, ARMEmitter::SubRegSize::i64Bit);
|
||||
@@ -567,11 +567,11 @@ DEF_OP(Vector_F64ToI32) {
|
||||
|
||||
///< Round float to integral depending on rounding mode.
|
||||
switch (Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val: frintz(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Host.Val: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::Nearest: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::NegInfinity: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::PosInfinity: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::TowardsZero: frintz(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::Host: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
}
|
||||
|
||||
// Now narrow from f64 to f32.
|
||||
|
||||
@@ -0,0 +1,35 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/AllocatorHooks.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
union Relocation;
|
||||
} // namespace FEXCore::CPU
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct DebugDataSubblock {
|
||||
uint32_t HostCodeOffset;
|
||||
uint32_t HostCodeSize;
|
||||
};
|
||||
|
||||
struct DebugDataGuestOpcode {
|
||||
uint64_t GuestEntryOffset;
|
||||
ptrdiff_t HostEntryOffset;
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Contains debug data for a block of code for later debugger analysis
|
||||
*
|
||||
* Needs to remain around for as long as the code could be executed at least
|
||||
*/
|
||||
struct DebugData : public FEXCore::Allocator::FEXAllocOperators {
|
||||
uint64_t HostCodeSize; ///< The size of the code generated in the host JIT
|
||||
fextl::vector<DebugDataSubblock> Subblocks;
|
||||
fextl::vector<DebugDataGuestOpcode> GuestOpcodes;
|
||||
fextl::vector<FEXCore::CPU::Relocation>* Relocations;
|
||||
};
|
||||
} // namespace FEXCore::Core
|
||||
@@ -12,14 +12,13 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include "Common/SoftFloat.h"
|
||||
#include "FEXCore/Utils/Telemetry.h"
|
||||
#include "FEXCore/Utils/TypeDefines.h"
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/JIT/DebugData.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include "Utils/MemberFunctionToPointer.h"
|
||||
@@ -30,15 +29,16 @@ $end_info$
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
#include <FEXCore/Utils/TypeDefines.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
|
||||
#include <stdio.h>
|
||||
#include <unistd.h>
|
||||
#include <string.h>
|
||||
#include <cstdio>
|
||||
#include <cstring>
|
||||
#include <limits>
|
||||
#include <unistd.h>
|
||||
|
||||
namespace {
|
||||
struct DivRem {
|
||||
@@ -538,7 +538,8 @@ uint64_t Arm64JITCore::ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEX
|
||||
} else {
|
||||
{
|
||||
// Guard the LookupCache lock with the code invalidation mutex, to avoid issues with forking
|
||||
auto lk_inval = GuardSignalDeferringSection<std::shared_lock>(static_cast<Context::ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
auto lk_inval =
|
||||
GuardSignalDeferringSection<std::shared_lock>(static_cast<Context::ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
}
|
||||
if (!HostCode) {
|
||||
@@ -624,10 +625,10 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
|
||||
|
||||
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRClass, GeneralRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRFixedClass, StaticRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRClass, GeneralFPRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRFixedClass, StaticFPRegisters.size());
|
||||
RAPass->AddRegisters(IR::RegClass::GPR, GeneralRegisters.size());
|
||||
RAPass->AddRegisters(IR::RegClass::GPRFixed, StaticRegisters.size());
|
||||
RAPass->AddRegisters(IR::RegClass::FPR, GeneralFPRegisters.size());
|
||||
RAPass->AddRegisters(IR::RegClass::FPRFixed, StaticFPRegisters.size());
|
||||
RAPass->PairRegs = PairRegisters;
|
||||
|
||||
{
|
||||
@@ -740,48 +741,48 @@ bool Arm64JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode,
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64JITCore::EmitInterruptChecks(bool CheckTF) {
|
||||
if (CheckTF) {
|
||||
ARMEmitter::ForwardLabel l_TFUnset;
|
||||
ARMEmitter::ForwardLabel l_TFBlocked;
|
||||
void Arm64JITCore::EmitTFCheck() {
|
||||
ARMEmitter::ForwardLabel l_TFUnset;
|
||||
ARMEmitter::ForwardLabel l_TFBlocked;
|
||||
|
||||
// Note that this needs to be before the below suspend checks, as X86 checks this flag immediately after executing an instruction.
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
// Note that this needs to be before the below suspend checks, as X86 checks this flag immediately after executing an instruction.
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
|
||||
cbz(ARMEmitter::Size::i32Bit, TMP1, &l_TFUnset);
|
||||
cbz(ARMEmitter::Size::i32Bit, TMP1, &l_TFUnset);
|
||||
|
||||
// X86 semantically checks TF after executing each instruction, so e.g. setting a context with TF set will execute a single instruction
|
||||
// and then raise an exception. However on the FEX side this is simpler to implement by checking at the start of each instruction, handle this by having bit 1 being unset in the flag state indicate that TF is blocked for a single instruction.
|
||||
tbz(TMP1, 1, &l_TFBlocked);
|
||||
// X86 semantically checks TF after executing each instruction, so e.g. setting a context with TF set will execute a single instruction
|
||||
// and then raise an exception. However on the FEX side this is simpler to implement by checking at the start of each instruction, handle this by having bit 1 being unset in the flag state indicate that TF is blocked for a single instruction.
|
||||
tbz(TMP1, 1, &l_TFBlocked);
|
||||
|
||||
// Block TF for a single instruction when the frontend jumps to a new context by unsetting bit 1.
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
and_(ARMEmitter::Size::i32Bit, TMP1, TMP1, ~(1 << 1));
|
||||
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
// Block TF for a single instruction when the frontend jumps to a new context by unsetting bit 1.
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
and_(ARMEmitter::Size::i32Bit, TMP1, TMP1, ~(1 << 1));
|
||||
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
|
||||
Core::CpuStateFrame::SynchronousFaultDataStruct State = {
|
||||
.FaultToTopAndGeneratedException = 1,
|
||||
.Signal = Core::FAULT_SIGTRAP,
|
||||
.TrapNo = X86State::X86_TRAPNO_DB,
|
||||
.si_code = 2,
|
||||
.err_code = 0,
|
||||
};
|
||||
Core::CpuStateFrame::SynchronousFaultDataStruct State = {
|
||||
.FaultToTopAndGeneratedException = 1,
|
||||
.Signal = Core::FAULT_SIGTRAP,
|
||||
.TrapNo = X86State::X86_TRAPNO_DB,
|
||||
.si_code = 2,
|
||||
.err_code = 0,
|
||||
};
|
||||
|
||||
uint64_t Constant {};
|
||||
memcpy(&Constant, &State, sizeof(State));
|
||||
uint64_t Constant {};
|
||||
memcpy(&Constant, &State, sizeof(State));
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Constant);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
|
||||
br(TMP1);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Constant);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
|
||||
br(TMP1);
|
||||
|
||||
Bind(&l_TFBlocked);
|
||||
// If TF was blocked for this instruction, unblock it for the next.
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 0b11);
|
||||
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
Bind(&l_TFUnset);
|
||||
}
|
||||
Bind(&l_TFBlocked);
|
||||
// If TF was blocked for this instruction, unblock it for the next.
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 0b11);
|
||||
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
Bind(&l_TFUnset);
|
||||
}
|
||||
|
||||
void Arm64JITCore::EmitSuspendInterruptCheck() {
|
||||
if (CTX->Config.NeedsPendingInterruptFaultCheck) {
|
||||
// Trigger a fault if there are any pending interrupts
|
||||
// Used only for suspend on WIN32 at the moment
|
||||
@@ -806,7 +807,9 @@ void Arm64JITCore::EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool C
|
||||
adr(TMP1, &HeaderLabel);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, InlineJITBlockHeader));
|
||||
|
||||
EmitInterruptChecks(CheckTF);
|
||||
if (CheckTF) {
|
||||
EmitTFCheck();
|
||||
}
|
||||
|
||||
if (SpillSlots) {
|
||||
const auto TotalSpillSlotsSize = SpillSlots * MaxSpillSlotSize;
|
||||
@@ -892,6 +895,9 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
|
||||
// if there's a pending branch, and it is not fall-through
|
||||
if (PendingTargetLabel && PendingTargetLabel != Target) {
|
||||
if (PendingTargetLabel->Backward.Location) {
|
||||
EmitSuspendInterruptCheck();
|
||||
}
|
||||
b(PendingTargetLabel);
|
||||
PendingTargetLabel = nullptr;
|
||||
}
|
||||
@@ -947,6 +953,9 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
|
||||
// Make sure last branch is generated. It certainly can't be eliminated here.
|
||||
if (PendingTargetLabel) {
|
||||
if (PendingTargetLabel->Backward.Location) {
|
||||
EmitSuspendInterruptCheck();
|
||||
}
|
||||
b(PendingTargetLabel);
|
||||
}
|
||||
PendingTargetLabel = nullptr;
|
||||
|
||||
@@ -10,13 +10,17 @@ $end_info$
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/JIT/Relocations.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IntrusiveIRList.h"
|
||||
#include "Interface/IR/RegisterAllocationData.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
@@ -25,16 +29,19 @@ $end_info$
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <functional>
|
||||
#include <optional>
|
||||
#include <utility>
|
||||
#include <variant>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::Context {
|
||||
struct ExitFunctionLinkData;
|
||||
}
|
||||
namespace FEXCore::IR {
|
||||
class RegisterAllocationPass;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
class Arm64JITCore final : public CPUBackend, public Arm64Emitter {
|
||||
@@ -90,11 +97,13 @@ private:
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Register GetReg(IR::PhysicalRegister Reg) const {
|
||||
LOGMAN_THROW_A_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
const auto RegClass = Reg.AsRegClass();
|
||||
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
LOGMAN_THROW_A_FMT(RegClass == IR::RegClass::GPRFixed || RegClass == IR::RegClass::GPR, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (RegClass == IR::RegClass::GPRFixed) {
|
||||
return StaticRegisters[Reg.Reg];
|
||||
} else if (Reg.Class == IR::GPRClass.Val) {
|
||||
} else if (RegClass == IR::RegClass::GPR) {
|
||||
return GeneralRegisters[Reg.Reg];
|
||||
}
|
||||
|
||||
@@ -113,11 +122,13 @@ private:
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::VRegister GetVReg(IR::PhysicalRegister Reg) const {
|
||||
LOGMAN_THROW_A_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
const auto RegClass = Reg.AsRegClass();
|
||||
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
LOGMAN_THROW_A_FMT(RegClass == IR::RegClass::FPRFixed || RegClass == IR::RegClass::FPR, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (RegClass == IR::RegClass::FPRFixed) {
|
||||
return StaticFPRegisters[Reg.Reg];
|
||||
} else if (Reg.Class == IR::FPRClass.Val) {
|
||||
} else if (RegClass == IR::RegClass::FPR) {
|
||||
return GeneralFPRegisters[Reg.Reg];
|
||||
}
|
||||
|
||||
@@ -135,8 +146,8 @@ private:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
FEXCore::IR::RegisterClassType GetRegClass(IR::Ref Node) const {
|
||||
return FEXCore::IR::RegisterClassType {IR::PhysicalRegister(Node).Class};
|
||||
static IR::RegClass GetRegClass(IR::Ref Node) {
|
||||
return IR::PhysicalRegister(Node).AsRegClass();
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
@@ -153,7 +164,7 @@ private:
|
||||
// Converts IR-base shift type to ARMEmitter shift type.
|
||||
// Will be a no-op, only a type conversion since the two definitions match.
|
||||
[[nodiscard]]
|
||||
ARMEmitter::ShiftType ConvertIRShiftType(IR::ShiftType Shift) const {
|
||||
static ARMEmitter::ShiftType ConvertIRShiftType(IR::ShiftType Shift) {
|
||||
return Shift == IR::ShiftType::LSL ? ARMEmitter::ShiftType::LSL :
|
||||
Shift == IR::ShiftType::LSR ? ARMEmitter::ShiftType::LSR :
|
||||
Shift == IR::ShiftType::ASR ? ARMEmitter::ShiftType::ASR :
|
||||
@@ -161,18 +172,23 @@ private:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Size ConvertSize(const IR::IROp_Header* Op) {
|
||||
static ARMEmitter::Size ConvertSize(const IR::IROp_Header* Op) {
|
||||
return Op->Size == IR::OpSize::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Size ConvertSize48(const IR::IROp_Header* Op) {
|
||||
static ARMEmitter::Size ConvertSize48(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_A_FMT(Op->Size == IR::OpSize::i32Bit || Op->Size == IR::OpSize::i64Bit, "Invalid size");
|
||||
return ConvertSize(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize16(IR::OpSize ElementSize) {
|
||||
static ARMEmitter::Size ConvertSize(IR::OpSize Size) {
|
||||
return Size == IR::OpSize::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static ARMEmitter::SubRegSize ConvertSubRegSize16(IR::OpSize ElementSize) {
|
||||
LOGMAN_THROW_A_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
|
||||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
|
||||
"Invalid size");
|
||||
@@ -184,105 +200,105 @@ private:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize16(const IR::IROp_Header* Op) {
|
||||
static ARMEmitter::SubRegSize ConvertSubRegSize16(const IR::IROp_Header* Op) {
|
||||
return ConvertSubRegSize16(Op->ElementSize);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize8(IR::OpSize ElementSize) {
|
||||
static ARMEmitter::SubRegSize ConvertSubRegSize8(IR::OpSize ElementSize) {
|
||||
LOGMAN_THROW_A_FMT(ElementSize != IR::OpSize::i128Bit, "Invalid size");
|
||||
return ConvertSubRegSize16(ElementSize);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize8(const IR::IROp_Header* Op) {
|
||||
static ARMEmitter::SubRegSize ConvertSubRegSize8(const IR::IROp_Header* Op) {
|
||||
return ConvertSubRegSize8(Op->ElementSize);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize4(const IR::IROp_Header* Op) {
|
||||
static ARMEmitter::SubRegSize ConvertSubRegSize4(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i64Bit, "Invalid size");
|
||||
return ConvertSubRegSize8(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize248(const IR::IROp_Header* Op) {
|
||||
static ARMEmitter::SubRegSize ConvertSubRegSize248(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i8Bit, "Invalid size");
|
||||
return ConvertSubRegSize8(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair16(const IR::IROp_Header* Op) {
|
||||
static ARMEmitter::VectorRegSizePair ConvertSubRegSizePair16(const IR::IROp_Header* Op) {
|
||||
return ARMEmitter::ToVectorSizePair(ConvertSubRegSize16(Op));
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair8(const IR::IROp_Header* Op) {
|
||||
static ARMEmitter::VectorRegSizePair ConvertSubRegSizePair8(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i128Bit, "Invalid size");
|
||||
return ConvertSubRegSizePair16(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair248(const IR::IROp_Header* Op) {
|
||||
static ARMEmitter::VectorRegSizePair ConvertSubRegSizePair248(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i8Bit, "Invalid size");
|
||||
return ConvertSubRegSizePair8(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Condition MapCC(IR::CondClassType Cond) {
|
||||
switch (Cond.Val) {
|
||||
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
|
||||
case FEXCore::IR::COND_NEQ: return ARMEmitter::Condition::CC_NE;
|
||||
case FEXCore::IR::COND_SGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_SLT: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_SGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_SLE: return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_UGE: return ARMEmitter::Condition::CC_CS;
|
||||
case FEXCore::IR::COND_ULT: return ARMEmitter::Condition::CC_CC;
|
||||
case FEXCore::IR::COND_UGT: return ARMEmitter::Condition::CC_HI;
|
||||
case FEXCore::IR::COND_ULE: return ARMEmitter::Condition::CC_LS;
|
||||
case FEXCore::IR::COND_FLU: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_FGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_FLEU: return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_FGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_FU:
|
||||
case FEXCore::IR::COND_VS: return ARMEmitter::Condition::CC_VS;
|
||||
case FEXCore::IR::COND_FNU:
|
||||
case FEXCore::IR::COND_VC: return ARMEmitter::Condition::CC_VC;
|
||||
case FEXCore::IR::COND_MI: return ARMEmitter::Condition::CC_MI;
|
||||
case FEXCore::IR::COND_PL: return ARMEmitter::Condition::CC_PL;
|
||||
static ARMEmitter::Condition MapCC(IR::CondClass Cond) {
|
||||
switch (Cond) {
|
||||
case IR::CondClass::EQ: return ARMEmitter::Condition::CC_EQ;
|
||||
case IR::CondClass::NEQ: return ARMEmitter::Condition::CC_NE;
|
||||
case IR::CondClass::SGE: return ARMEmitter::Condition::CC_GE;
|
||||
case IR::CondClass::SLT: return ARMEmitter::Condition::CC_LT;
|
||||
case IR::CondClass::SGT: return ARMEmitter::Condition::CC_GT;
|
||||
case IR::CondClass::SLE: return ARMEmitter::Condition::CC_LE;
|
||||
case IR::CondClass::UGE: return ARMEmitter::Condition::CC_CS;
|
||||
case IR::CondClass::ULT: return ARMEmitter::Condition::CC_CC;
|
||||
case IR::CondClass::UGT: return ARMEmitter::Condition::CC_HI;
|
||||
case IR::CondClass::ULE: return ARMEmitter::Condition::CC_LS;
|
||||
case IR::CondClass::FLU: return ARMEmitter::Condition::CC_LT;
|
||||
case IR::CondClass::FGE: return ARMEmitter::Condition::CC_GE;
|
||||
case IR::CondClass::FLEU: return ARMEmitter::Condition::CC_LE;
|
||||
case IR::CondClass::FGT: return ARMEmitter::Condition::CC_GT;
|
||||
case IR::CondClass::FU:
|
||||
case IR::CondClass::VS: return ARMEmitter::Condition::CC_VS;
|
||||
case IR::CondClass::FNU:
|
||||
case IR::CondClass::VC: return ARMEmitter::Condition::CC_VC;
|
||||
case IR::CondClass::MI: return ARMEmitter::Condition::CC_MI;
|
||||
case IR::CondClass::PL: return ARMEmitter::Condition::CC_PL;
|
||||
default: LOGMAN_MSG_A_FMT("Unsupported compare type"); return ARMEmitter::Condition::CC_NV;
|
||||
}
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsFPR(IR::RegisterClassType Class) const {
|
||||
return Class == IR::FPRClass || Class == IR::FPRFixedClass;
|
||||
static bool IsFPR(IR::RegClass Class) {
|
||||
return Class == IR::RegClass::FPR || Class == IR::RegClass::FPRFixed;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsGPR(IR::RegisterClassType Class) const {
|
||||
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
|
||||
static bool IsGPR(IR::RegClass Class) {
|
||||
return Class == IR::RegClass::GPR || Class == IR::RegClass::GPRFixed;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsGPR(IR::Ref Node) {
|
||||
static bool IsGPR(IR::Ref Node) {
|
||||
return IsGPR(GetRegClass(Node));
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsFPR(IR::Ref Node) {
|
||||
static bool IsFPR(IR::Ref Node) {
|
||||
return IsFPR(GetRegClass(Node));
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsGPR(IR::OrderedNodeWrapper Wrap) {
|
||||
return IsGPR(IR::RegisterClassType {IR::PhysicalRegister(Wrap).Class});
|
||||
static bool IsGPR(IR::OrderedNodeWrapper Wrap) {
|
||||
return IsGPR(IR::PhysicalRegister(Wrap).AsRegClass());
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsFPR(IR::OrderedNodeWrapper Wrap) {
|
||||
return IsFPR(IR::RegisterClassType {IR::PhysicalRegister(Wrap).Class});
|
||||
static bool IsFPR(IR::OrderedNodeWrapper Wrap) {
|
||||
return IsFPR(IR::PhysicalRegister(Wrap).AsRegClass());
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
@@ -381,7 +397,9 @@ private:
|
||||
fextl::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
|
||||
///< Relocation code loading
|
||||
bool ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations);
|
||||
bool ApplyRelocations(uint64_t GuestEntry, std::span<std::byte> Code, std::span<const FEXCore::CPU::Relocation>);
|
||||
|
||||
fextl::vector<FEXCore::CPU::Relocation> TakeRelocations() override;
|
||||
|
||||
/** @} */
|
||||
|
||||
@@ -404,9 +422,11 @@ private:
|
||||
void Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize, ARMEmitter::VRegister Dst, ARMEmitter::VRegister IncomingDst,
|
||||
std::optional<ARMEmitter::Register> BaseAddr, ARMEmitter::VRegister VectorIndexLow,
|
||||
std::optional<ARMEmitter::VRegister> VectorIndexHigh, ARMEmitter::VRegister MaskReg, IR::OpSize VectorIndexSize,
|
||||
size_t DataElementOffsetStart, size_t IndexElementOffsetStart, uint8_t OffsetScale);
|
||||
size_t DataElementOffsetStart, size_t IndexElementOffsetStart, uint8_t OffsetScale, IR::OpSize AddrSize);
|
||||
|
||||
void EmitInterruptChecks(bool CheckTF);
|
||||
void EmitTFCheck();
|
||||
|
||||
void EmitSuspendInterruptCheck();
|
||||
|
||||
void EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool CheckTF);
|
||||
|
||||
|
||||
@@ -21,7 +21,7 @@ DEF_OP(LoadContext) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
auto Dst = GetReg(Node);
|
||||
|
||||
switch (OpSize) {
|
||||
@@ -52,7 +52,7 @@ DEF_OP(LoadContext) {
|
||||
DEF_OP(LoadContextPair) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadContextPair>();
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Dst1 = GetReg(Op->OutValue1);
|
||||
const auto Dst2 = GetReg(Op->OutValue2);
|
||||
|
||||
@@ -78,7 +78,7 @@ DEF_OP(StoreContext) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
auto Src = GetZeroableReg(Op->Value);
|
||||
|
||||
switch (OpSize) {
|
||||
@@ -110,7 +110,7 @@ DEF_OP(StoreContextPair) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreContextPair>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
auto Src1 = GetZeroableReg(Op->Value1);
|
||||
auto Src2 = GetZeroableReg(Op->Value2);
|
||||
|
||||
@@ -135,11 +135,11 @@ DEF_OP(StoreContextPair) {
|
||||
DEF_OP(LoadRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
LOGMAN_THROW_A_FMT(Op->Reg < StaticRegisters.size(), "out of range reg");
|
||||
|
||||
mov(GetReg(Node).X(), StaticRegisters[Op->Reg].X());
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
} else if (Op->Class == IR::RegClass::FPR) {
|
||||
const auto regSize = HostSupportsAVX256 ? IR::OpSize::i256Bit : IR::OpSize::i128Bit;
|
||||
LOGMAN_THROW_A_FMT(Op->Reg < StaticFPRegisters.size(), "out of range reg");
|
||||
LOGMAN_THROW_A_FMT(IROp->Size == regSize, "expected sized");
|
||||
@@ -175,12 +175,13 @@ DEF_OP(LoadAF) {
|
||||
|
||||
DEF_OP(StoreRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
auto Reg = IR::PhysicalRegister(Node);
|
||||
const auto Reg = IR::PhysicalRegister(Node);
|
||||
const auto RegClass = Reg.AsRegClass();
|
||||
|
||||
if (Reg.Class == IR::GPRFixedClass) {
|
||||
if (RegClass == IR::RegClass::GPRFixed) {
|
||||
// Always use 64-bit, it's faster. Upper bits ignored for 32-bit mode.
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Reg), GetReg(Op->Value));
|
||||
} else if (Reg.Class == IR::FPRFixedClass) {
|
||||
} else if (RegClass == IR::RegClass::FPRFixed) {
|
||||
const auto regSize = HostSupportsAVX256 ? IR::OpSize::i256Bit : IR::OpSize::i128Bit;
|
||||
LOGMAN_THROW_A_FMT(IROp->Size == regSize, "expected sized");
|
||||
|
||||
@@ -193,7 +194,7 @@ DEF_OP(StoreRegister) {
|
||||
mov(guest.Q(), host.Q());
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", Reg.Class);
|
||||
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", RegClass);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -225,7 +226,7 @@ DEF_OP(LoadContextIndexed) {
|
||||
|
||||
const auto Index = GetReg(Op->Index);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
switch (Op->Stride) {
|
||||
case 1:
|
||||
case 2:
|
||||
@@ -288,7 +289,7 @@ DEF_OP(StoreContextIndexed) {
|
||||
|
||||
const auto Index = GetReg(Op->Index);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Value = GetReg(Op->Value);
|
||||
|
||||
switch (Op->Stride) {
|
||||
@@ -348,12 +349,31 @@ DEF_OP(StoreContextIndexed) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(FormContextAddress) {
|
||||
const auto Op = IROp->C<IR::IROp_FormContextAddress>();
|
||||
const auto Index = GetReg(Op->Index);
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
switch (Op->Stride) {
|
||||
case 1:
|
||||
case 2:
|
||||
case 4:
|
||||
case 8:
|
||||
case 16:
|
||||
case 32: {
|
||||
add(ARMEmitter::Size::i64Bit, Dst, STATE, Index, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Op->Stride));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled FormContextAddress stride: {}", Op->Stride); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(SpillRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_SpillRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
const uint32_t SlotOffset = Op->Slot * MaxSpillSlotSize;
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Src = GetReg(Op->Value);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: {
|
||||
@@ -394,7 +414,7 @@ DEF_OP(SpillRegister) {
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize); break;
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::FPRClass) {
|
||||
} else if (Op->Class == FEXCore::IR::RegClass::FPR) {
|
||||
const auto Src = GetVReg(Op->Value);
|
||||
|
||||
switch (OpSize) {
|
||||
@@ -433,7 +453,7 @@ DEF_OP(SpillRegister) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unhandled SpillRegister class: {}", Op->Class.Val);
|
||||
LOGMAN_MSG_A_FMT("Unhandled SpillRegister class: {}", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -442,7 +462,7 @@ DEF_OP(FillRegister) {
|
||||
const auto OpSize = IROp->Size;
|
||||
const uint32_t SlotOffset = Op->Slot * MaxSpillSlotSize;
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Dst = GetReg(Node);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: {
|
||||
@@ -483,7 +503,7 @@ DEF_OP(FillRegister) {
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize); break;
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::FPRClass) {
|
||||
} else if (Op->Class == FEXCore::IR::RegClass::FPR) {
|
||||
const auto Dst = GetVReg(Node);
|
||||
|
||||
switch (OpSize) {
|
||||
@@ -522,7 +542,7 @@ DEF_OP(FillRegister) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unhandled FillRegister class: {}", Op->Class.Val);
|
||||
LOGMAN_MSG_A_FMT("Unhandled FillRegister class: {}", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -559,14 +579,14 @@ ARMEmitter::ExtendedMemOperand Arm64JITCore::GenerateMemOperand(
|
||||
return ARMEmitter::ExtendedMemOperand(Base.X(), ARMEmitter::IndexType::OFFSET, Const);
|
||||
} else {
|
||||
auto RegOffset = GetReg(Offset);
|
||||
switch (OffsetType.Val) {
|
||||
case IR::MEM_OFFSET_SXTX.Val:
|
||||
switch (OffsetType) {
|
||||
case IR::MemOffsetType::SXTX:
|
||||
return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::SXTX, FEXCore::ilog2(OffsetScale));
|
||||
case IR::MEM_OFFSET_UXTW.Val:
|
||||
case IR::MemOffsetType::UXTW:
|
||||
return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::UXTW, FEXCore::ilog2(OffsetScale));
|
||||
case IR::MEM_OFFSET_SXTW.Val:
|
||||
case IR::MemOffsetType::SXTW:
|
||||
return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale));
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled GenerateMemOperand OffsetType: {}", OffsetType.Val); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled GenerateMemOperand OffsetType: {}", OffsetType); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -593,20 +613,20 @@ ARMEmitter::Register Arm64JITCore::ApplyMemOperand(IR::OpSize AccessSize, ARMEmi
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, Tmp, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(OffsetScale));
|
||||
} else {
|
||||
auto RegOffset = GetReg(Offset);
|
||||
switch (OffsetType.Val) {
|
||||
case IR::MEM_OFFSET_SXTX.Val:
|
||||
switch (OffsetType) {
|
||||
case IR::MemOffsetType::SXTX:
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, RegOffset, ARMEmitter::ExtendedType::SXTX, FEXCore::ilog2(OffsetScale));
|
||||
break;
|
||||
|
||||
case IR::MEM_OFFSET_UXTW.Val:
|
||||
case IR::MemOffsetType::UXTW:
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, RegOffset, ARMEmitter::ExtendedType::UXTW, FEXCore::ilog2(OffsetScale));
|
||||
break;
|
||||
|
||||
case IR::MEM_OFFSET_SXTW.Val:
|
||||
case IR::MemOffsetType::SXTW:
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, RegOffset, ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale));
|
||||
break;
|
||||
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled OffsetType: {}", OffsetType.Val); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled OffsetType: {}", OffsetType); break;
|
||||
}
|
||||
}
|
||||
return Tmp;
|
||||
@@ -657,7 +677,7 @@ ARMEmitter::SVEMemOperand Arm64JITCore::GenerateSVEMemOperand(IR::OpSize AccessS
|
||||
// Note that we do nothing with the offset type and offset scale,
|
||||
// since SVE loads and stores don't have the ability to perform an
|
||||
// optional extension or shift as part of their behavior.
|
||||
LOGMAN_THROW_A_FMT(OffsetType.Val == IR::MEM_OFFSET_SXTX.Val, "Currently only the default offset type (SXTX) is supported.");
|
||||
LOGMAN_THROW_A_FMT(OffsetType == IR::MemOffsetType::SXTX, "Currently only the default offset type (SXTX) is supported.");
|
||||
|
||||
const auto RegOffset = GetReg(Offset);
|
||||
return ARMEmitter::SVEMemOperand(Base.X(), RegOffset.X());
|
||||
@@ -670,7 +690,7 @@ DEF_OP(LoadMem) {
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
switch (OpSize) {
|
||||
@@ -704,7 +724,7 @@ DEF_OP(LoadMemPair) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemPair>();
|
||||
const auto Addr = GetReg(Op->Addr);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Dst1 = GetReg(Op->OutValue1);
|
||||
const auto Dst2 = GetReg(Op->OutValue2);
|
||||
|
||||
@@ -732,13 +752,13 @@ DEF_OP(LoadMemTSO) {
|
||||
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid() || CTX->HostFeatures.SupportsTSOImm9, "unexpected offset");
|
||||
LOGMAN_THROW_A_FMT(Op->OffsetScale == 1, "unexpected offset scale");
|
||||
LOGMAN_THROW_A_FMT(Op->OffsetType == IR::MEM_OFFSET_SXTX, "unexpected offset type");
|
||||
LOGMAN_THROW_A_FMT(Op->OffsetType == IR::MemOffsetType::SXTX, "unexpected offset type");
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == IR::RegClass::GPR) {
|
||||
const auto Dst = GetReg(Node);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
@@ -760,7 +780,7 @@ DEF_OP(LoadMemTSO) {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
} else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
} else if (CTX->HostFeatures.SupportsRCPC && Op->Class == IR::RegClass::GPR) {
|
||||
const auto Dst = GetReg(Node);
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
@@ -775,7 +795,7 @@ DEF_OP(LoadMemTSO) {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
} else if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Dst = GetReg(Node);
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
@@ -1020,7 +1040,7 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
|
||||
ARMEmitter::VRegister IncomingDst, std::optional<ARMEmitter::Register> BaseAddr,
|
||||
ARMEmitter::VRegister VectorIndexLow, std::optional<ARMEmitter::VRegister> VectorIndexHigh,
|
||||
ARMEmitter::VRegister MaskReg, IR::OpSize VectorIndexSize, size_t DataElementOffsetStart,
|
||||
size_t IndexElementOffsetStart, uint8_t OffsetScale) {
|
||||
size_t IndexElementOffsetStart, uint8_t OffsetScale, IR::OpSize AddrSize) {
|
||||
LOGMAN_THROW_A_FMT(ElementSize >= IR::OpSize::i8Bit && ElementSize <= IR::OpSize::i64Bit, "Invalid element size");
|
||||
|
||||
const auto PerformSMove = [this](IR::OpSize ElementSize, const ARMEmitter::Register Dst, const ARMEmitter::VRegister Vector, int index) {
|
||||
@@ -1096,17 +1116,17 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
|
||||
// Calculate memory position for this gather load
|
||||
if (BaseAddr.has_value()) {
|
||||
if (VectorIndexSize == IR::OpSize::i32Bit) {
|
||||
add(ARMEmitter::Size::i64Bit, TempMemReg, *BaseAddr, WorkingReg, ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale));
|
||||
add(ConvertSize(AddrSize), TempMemReg, *BaseAddr, WorkingReg, ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale));
|
||||
} else {
|
||||
add(ARMEmitter::Size::i64Bit, TempMemReg, *BaseAddr, WorkingReg, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(OffsetScale));
|
||||
add(ConvertSize(AddrSize), TempMemReg, *BaseAddr, WorkingReg, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(OffsetScale));
|
||||
}
|
||||
} else {
|
||||
///< In this case we have no base address, All addresses come from the vector register itself
|
||||
if (VectorIndexSize == IR::OpSize::i32Bit) {
|
||||
// Sign extend and shift in to the 64-bit register
|
||||
sbfiz(ARMEmitter::Size::i64Bit, TempMemReg, WorkingReg, FEXCore::ilog2(OffsetScale), 32);
|
||||
sbfiz(ConvertSize(AddrSize), TempMemReg, WorkingReg, FEXCore::ilog2(OffsetScale), 32);
|
||||
} else {
|
||||
lsl(ARMEmitter::Size::i64Bit, TempMemReg, WorkingReg, FEXCore::ilog2(OffsetScale));
|
||||
lsl(ConvertSize(AddrSize), TempMemReg, WorkingReg, FEXCore::ilog2(OffsetScale));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1164,7 +1184,8 @@ DEF_OP(VLoadVectorGatherMasked) {
|
||||
|
||||
///< If the host supports SVE and the offset scale matches SVE limitations then it can do an SVE style load.
|
||||
const bool SupportsSVELoad = (HostSupportsSVE128 || HostSupportsSVE256) &&
|
||||
(OffsetScale == 1 || OffsetScale == IR::OpSizeToSize(VectorIndexSize)) && VectorIndexSize == IROp->ElementSize;
|
||||
(OffsetScale == 1 || OffsetScale == IR::OpSizeToSize(VectorIndexSize)) &&
|
||||
VectorIndexSize == IROp->ElementSize && Op->AddrSize == IR::OpSize::i64Bit;
|
||||
|
||||
if (SupportsSVELoad) {
|
||||
uint8_t SVEScale = FEXCore::ilog2(OffsetScale);
|
||||
@@ -1222,7 +1243,7 @@ DEF_OP(VLoadVectorGatherMasked) {
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit, "Can't emulate this gather load in the backend! Programming error!");
|
||||
Emulate128BitGather(IROp->Size, IROp->ElementSize, Dst, IncomingDst, BaseAddr, VectorIndexLow, VectorIndexHigh, MaskReg,
|
||||
VectorIndexSize, DataElementOffsetStart, IndexElementOffsetStart, OffsetScale);
|
||||
VectorIndexSize, DataElementOffsetStart, IndexElementOffsetStart, OffsetScale, Op->AddrSize);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1247,7 +1268,9 @@ DEF_OP(VLoadVectorGatherMaskedQPS) {
|
||||
!Op->VectorIndexHigh.IsInvalid() ? std::make_optional(GetVReg(Op->VectorIndexHigh)) : std::nullopt;
|
||||
|
||||
///< If the host supports SVE and the offset scale matches SVE limitations then it can do an SVE style load.
|
||||
if (HostSupportsSVE128 && (OffsetScale == 1 || OffsetScale == 4)) {
|
||||
const bool SupportsSVELoad = HostSupportsSVE128 && (OffsetScale == 1 || OffsetScale == 4) && Op->AddrSize == IR::OpSize::i64Bit;
|
||||
|
||||
if (SupportsSVELoad) {
|
||||
ARMEmitter::SVEModType ModType = ARMEmitter::SVEModType::MOD_NONE;
|
||||
if (OffsetScale != 1) {
|
||||
ModType = ARMEmitter::SVEModType::MOD_LSL;
|
||||
@@ -1301,7 +1324,7 @@ DEF_OP(VLoadVectorGatherMaskedQPS) {
|
||||
}
|
||||
} else {
|
||||
Emulate128BitGather(IR::OpSize::i128Bit, IR::OpSize::i32Bit, Dst, IncomingDst, BaseAddr, VectorIndexLow, VectorIndexHigh, MaskReg,
|
||||
IR::OpSize::i64Bit, 0, 0, OffsetScale);
|
||||
IR::OpSize::i64Bit, 0, 0, OffsetScale, Op->AddrSize);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1602,7 +1625,7 @@ DEF_OP(StoreMem) {
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: strb(Src, MemSrc); break;
|
||||
@@ -1713,7 +1736,7 @@ DEF_OP(StoreMemPair) {
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Addr = GetReg(Op->Addr);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Src1 = GetZeroableReg(Op->Value1);
|
||||
const auto Src2 = GetZeroableReg(Op->Value2);
|
||||
switch (OpSize) {
|
||||
@@ -1740,13 +1763,13 @@ DEF_OP(StoreMemTSO) {
|
||||
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid() || CTX->HostFeatures.SupportsTSOImm9, "unexpected offset");
|
||||
LOGMAN_THROW_A_FMT(Op->OffsetScale == 1, "unexpected offset scale");
|
||||
LOGMAN_THROW_A_FMT(Op->OffsetType == IR::MEM_OFFSET_SXTX, "unexpected offset type");
|
||||
LOGMAN_THROW_A_FMT(Op->OffsetType == IR::MemOffsetType::SXTX, "unexpected offset type");
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == IR::RegClass::GPR) {
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
@@ -1767,7 +1790,7 @@ DEF_OP(StoreMemTSO) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
} else if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
@@ -2280,7 +2303,7 @@ DEF_OP(ParanoidLoadMemTSO) {
|
||||
|
||||
auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == IR::RegClass::GPR) {
|
||||
const auto Dst = GetReg(Node);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
@@ -2301,7 +2324,7 @@ DEF_OP(ParanoidLoadMemTSO) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
} else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
} else if (CTX->HostFeatures.SupportsRCPC && Op->Class == IR::RegClass::GPR) {
|
||||
const auto Dst = GetReg(Node);
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
@@ -2315,7 +2338,7 @@ DEF_OP(ParanoidLoadMemTSO) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
} else if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Dst = GetReg(Node);
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
switch (OpSize) {
|
||||
@@ -2368,7 +2391,7 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
|
||||
auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == IR::RegClass::GPR) {
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
@@ -2388,7 +2411,7 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
} else if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP1, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
switch (OpSize) {
|
||||
|
||||
@@ -10,10 +10,12 @@ $end_info$
|
||||
#endif
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/JIT/DebugData.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
@@ -46,10 +48,10 @@ DEF_OP(GuestOpcode) {
|
||||
DEF_OP(Fence) {
|
||||
auto Op = IROp->C<IR::IROp_Fence>();
|
||||
switch (Op->Fence) {
|
||||
case IR::Fence_Load.Val: dmb(ARMEmitter::BarrierScope::LD); break;
|
||||
case IR::Fence_LoadStore.Val: dmb(ARMEmitter::BarrierScope::SY); break;
|
||||
case IR::Fence_Store.Val: dmb(ARMEmitter::BarrierScope::ST); break;
|
||||
case IR::Fence_Inst.Val: isb(); break;
|
||||
case IR::FenceType::Load: dmb(ARMEmitter::BarrierScope::LD); break;
|
||||
case IR::FenceType::LoadStore: dmb(ARMEmitter::BarrierScope::SY); break;
|
||||
case IR::FenceType::Store: dmb(ARMEmitter::BarrierScope::ST); break;
|
||||
case IR::FenceType::Inst: isb(); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Fence: {}", Op->Fence); break;
|
||||
}
|
||||
}
|
||||
@@ -106,10 +108,10 @@ DEF_OP(GetRoundingMode) {
|
||||
// zero. Just swapping 01 and 10. That's a bitfield reverse. Round mode is in
|
||||
// bottom two bits. After reversing as a 32-bit operation, it'll be in [31:30]
|
||||
// and ripe for reinsertion back at 0.
|
||||
static_assert(IR::ROUND_MODE_NEAREST == 0);
|
||||
static_assert(IR::ROUND_MODE_NEGATIVE_INFINITY == 1);
|
||||
static_assert(IR::ROUND_MODE_POSITIVE_INFINITY == 2);
|
||||
static_assert(IR::ROUND_MODE_TOWARDS_ZERO == 3);
|
||||
static_assert(FEXCore::ToUnderlying(IR::RoundMode::Nearest) == 0);
|
||||
static_assert(FEXCore::ToUnderlying(IR::RoundMode::NegInfinity) == 1);
|
||||
static_assert(FEXCore::ToUnderlying(IR::RoundMode::PosInfinity) == 2);
|
||||
static_assert(FEXCore::ToUnderlying(IR::RoundMode::TowardsZero) == 3);
|
||||
|
||||
rbit(ARMEmitter::Size::i32Bit, TMP1, Dst);
|
||||
bfi(ARMEmitter::Size::i64Bit, Dst, TMP1, 30, 2);
|
||||
|
||||
@@ -18,18 +18,4 @@ DEF_OP(RMWHandle) {
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), GetReg(IROp->Args[0]));
|
||||
}
|
||||
|
||||
DEF_OP(Swap1) {
|
||||
auto Op = IROp->C<IR::IROp_Swap1>();
|
||||
auto A = GetReg(Op->A), B = GetReg(Op->B);
|
||||
LOGMAN_THROW_A_FMT(B == GetReg(Node), "Invariant");
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, TMP1, A);
|
||||
mov(ARMEmitter::Size::i64Bit, A, B);
|
||||
mov(ARMEmitter::Size::i64Bit, B, TMP1);
|
||||
}
|
||||
|
||||
DEF_OP(Swap2) {
|
||||
// Implemented above
|
||||
}
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -41,6 +41,7 @@ namespace FEXCore::CPU {
|
||||
const auto Op = IROp->C<IR::IROp_##FEXOp>(); \
|
||||
const auto OpSize = IROp->Size; \
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit; \
|
||||
const auto Is128Bit = OpSize == IR::OpSize::i128Bit; \
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__); \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
@@ -49,8 +50,10 @@ namespace FEXCore::CPU {
|
||||
\
|
||||
if (HostSupportsSVE256 && Is256Bit) { \
|
||||
ARMOp(Dst.Z(), Vector1.Z(), Vector2.Z()); \
|
||||
} else { \
|
||||
} else if (Is128Bit) { \
|
||||
ARMOp(Dst.Q(), Vector1.Q(), Vector2.Q()); \
|
||||
} else { \
|
||||
ARMOp(Dst.D(), Vector1.D(), Vector2.D()); \
|
||||
} \
|
||||
}
|
||||
|
||||
@@ -744,11 +747,11 @@ DEF_OP(VFToIScalarInsert) {
|
||||
auto Src = *std::get_if<ARMEmitter::VRegister>(&SrcVar);
|
||||
|
||||
switch (RoundMode) {
|
||||
case IR::Round_Nearest: frintn(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::Round_Negative_Infinity: frintm(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::Round_Positive_Infinity: frintp(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::Round_Towards_Zero: frintz(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::Round_Host: frinti(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::RoundMode::Nearest: frintn(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::RoundMode::NegInfinity: frintm(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::RoundMode::PosInfinity: frintp(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::RoundMode::TowardsZero: frintz(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::RoundMode::Host: frinti(SubRegSize.Scalar, Dst, Src); break;
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -139,27 +139,28 @@ public:
|
||||
FlushRegisterCache();
|
||||
return _Jump(_TargetBlock);
|
||||
}
|
||||
IRPair<IROp_CondJump> CondJump(Ref _Cmp1, Ref _Cmp2, Ref _TrueBlock, Ref _FalseBlock, CondClassType _Cond = {COND_NEQ},
|
||||
IRPair<IROp_CondJump> CondJump(Ref _Cmp1, Ref _Cmp2, Ref _TrueBlock, Ref _FalseBlock, CondClass _Cond = CondClass::NEQ,
|
||||
IR::OpSize _CompareSize = OpSize::iInvalid) {
|
||||
FlushRegisterCache();
|
||||
return _CondJump(_Cmp1, _Cmp2, _TrueBlock, _FalseBlock, _Cond, _CompareSize);
|
||||
}
|
||||
IRPair<IROp_CondJump> CondJump(Ref ssa0, CondClassType cond = {COND_NEQ}) {
|
||||
IRPair<IROp_CondJump> CondJump(Ref ssa0, CondClass cond = CondClass::NEQ) {
|
||||
FlushRegisterCache();
|
||||
return _CondJump(ssa0, cond);
|
||||
}
|
||||
IRPair<IROp_CondJump> CondJump(Ref ssa0, Ref ssa1, Ref ssa2, CondClassType cond = {COND_NEQ}) {
|
||||
IRPair<IROp_CondJump> CondJump(Ref ssa0, Ref ssa1, Ref ssa2, CondClass cond = CondClass::NEQ) {
|
||||
FlushRegisterCache();
|
||||
return _CondJump(ssa0, ssa1, ssa2, cond);
|
||||
}
|
||||
IRPair<IROp_CondJump> CondJumpNZCV(CondClassType Cond) {
|
||||
IRPair<IROp_CondJump> CondJumpNZCV(CondClass Cond) {
|
||||
FlushRegisterCache();
|
||||
return _CondJump(InvalidNode, InvalidNode, InvalidNode, InvalidNode, Cond, OpSize::iInvalid, true);
|
||||
}
|
||||
IRPair<IROp_CondJump> CondJumpBit(Ref Src, unsigned Bit, bool Set) {
|
||||
FlushRegisterCache();
|
||||
auto InlineConst = _InlineConstant(Bit);
|
||||
return _CondJump(Src, InlineConst, InvalidNode, InvalidNode, {Set ? COND_TSTNZ : COND_TSTZ}, OpSize::iInvalid, false);
|
||||
auto Cond = Set ? CondClass::TSTNZ : CondClass::TSTZ;
|
||||
return _CondJump(Src, InlineConst, InvalidNode, InvalidNode, Cond, OpSize::iInvalid, false);
|
||||
}
|
||||
IRPair<IROp_ExitFunction> ExitFunction(Ref NewRIP, BranchHint Hint = BranchHint::None) {
|
||||
FlushRegisterCache();
|
||||
@@ -209,10 +210,17 @@ public:
|
||||
}
|
||||
|
||||
static bool CanHaveSideEffects(const FEXCore::X86Tables::X86InstInfo* TableInfo, FEXCore::X86Tables::DecodedOp Op) {
|
||||
if (TableInfo && TableInfo->Flags & X86Tables::InstFlags::FLAGS_DEBUG_MEM_ACCESS) {
|
||||
// If it is marked as having memory access then always say it has a side-effect.
|
||||
// Not always true but better to be safe.
|
||||
return true;
|
||||
if (TableInfo) {
|
||||
if (TableInfo->Flags & X86Tables::InstFlags::FLAGS_DEBUG_MEM_ACCESS) {
|
||||
// If it is marked as having memory access then always say it has a side-effect.
|
||||
// Not always true but better to be safe.
|
||||
return true;
|
||||
}
|
||||
|
||||
if (TableInfo->Flags & (X86Tables::InstFlags::FLAGS_SETS_RIP | X86Tables::InstFlags::FLAGS_BLOCK_END)) {
|
||||
// Cooperative suspend interrupts can be triggered at any back-edge, the RIP must be reconstructed correctly in such cases
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
auto CanHaveSideEffects = false;
|
||||
@@ -244,7 +252,7 @@ public:
|
||||
auto ExitBlock = CreateNewCodeBlockAfter(BackwardBlock);
|
||||
|
||||
auto DF = GetRFLAG(X86State::RFLAG_DF_RAW_LOC);
|
||||
CondJump(DF, Zero, ForwardBlock, BackwardBlock, {COND_EQ});
|
||||
CondJump(DF, Zero, ForwardBlock, BackwardBlock, CondClass::EQ);
|
||||
|
||||
for (auto D = 0; D < 2; ++D) {
|
||||
SetCurrentCodeBlock(D ? BackwardBlock : ForwardBlock);
|
||||
@@ -293,7 +301,8 @@ public:
|
||||
return ShouldDump;
|
||||
}
|
||||
|
||||
void BeginFunction(uint64_t RIP, const fextl::vector<FEXCore::Frontend::Decoder::DecodedBlocks>* Blocks, uint32_t NumInstructions, bool Is64BitMode, bool MonoBackpatcherBlock);
|
||||
void BeginFunction(uint64_t RIP, const fextl::vector<FEXCore::Frontend::Decoder::DecodedBlocks>* Blocks, uint32_t NumInstructions,
|
||||
bool Is64BitMode, bool MonoBackpatcherBlock);
|
||||
void Finalize();
|
||||
|
||||
// Dispatch builder functions
|
||||
@@ -310,6 +319,7 @@ public:
|
||||
|
||||
void UnhandledOp(OpcodeArgs);
|
||||
void MOVGPROp(OpcodeArgs, uint32_t SrcIndex);
|
||||
void MOVGPRImmediate(OpcodeArgs);
|
||||
void MOVGPRNTOp(OpcodeArgs);
|
||||
void MOVVectorAlignedOp(OpcodeArgs);
|
||||
void MOVVectorUnalignedOp(OpcodeArgs);
|
||||
@@ -555,7 +565,7 @@ public:
|
||||
template<IR::OpSize DstElementSize, IR::OpSize SrcElementSize>
|
||||
void AVXInsertScalar_CVT_Float_To_Float(OpcodeArgs);
|
||||
|
||||
RoundType TranslateRoundType(uint8_t Mode);
|
||||
RoundMode TranslateRoundType(uint8_t Mode);
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void InsertScalarRound(OpcodeArgs);
|
||||
@@ -749,7 +759,6 @@ public:
|
||||
void X87FXTRACT(OpcodeArgs);
|
||||
void X87FYL2X(OpcodeArgs, bool IsFYL2XP1);
|
||||
void X87LDENV(OpcodeArgs);
|
||||
void X87LDSW(OpcodeArgs);
|
||||
void X87ModifySTP(OpcodeArgs, bool Inc);
|
||||
void X87OpHelper(OpcodeArgs, FEXCore::IR::IROps IROp, bool ZeroC2);
|
||||
|
||||
@@ -898,6 +907,10 @@ public:
|
||||
void VPCLMULQDQOp(OpcodeArgs);
|
||||
|
||||
void CRC32(OpcodeArgs);
|
||||
void Extrq_imm(OpcodeArgs);
|
||||
void Insertq_imm(OpcodeArgs);
|
||||
void Extrq(OpcodeArgs);
|
||||
void Insertq(OpcodeArgs);
|
||||
|
||||
void BreakOp(OpcodeArgs, FEXCore::IR::BreakDefinition BreakDefinition);
|
||||
void UnimplementedOp(OpcodeArgs);
|
||||
@@ -976,7 +989,6 @@ public:
|
||||
void AVX128_VPSIGN(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void AVX128_UCOMISx(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void AVX128_VectorScalarInsertALU(OpcodeArgs, FEXCore::IR::IROps IROp, IR::OpSize ElementSize);
|
||||
Ref AVX128_VFCMPImpl(IR::OpSize ElementSize, Ref Src1, Ref Src2, uint8_t CompType);
|
||||
void AVX128_VFCMP(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void AVX128_InsertScalarFCMP(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void AVX128_MOVBetweenGPR_FPR(OpcodeArgs);
|
||||
@@ -995,9 +1007,7 @@ public:
|
||||
void AVX128_VINSERT(OpcodeArgs);
|
||||
void AVX128_VINSERTPS(OpcodeArgs);
|
||||
|
||||
Ref AVX128_PHSUBImpl(Ref Src1, Ref Src2, size_t ElementSize);
|
||||
void AVX128_VPHSUB(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
void AVX128_VPHSUBSW(OpcodeArgs);
|
||||
|
||||
void AVX128_VADDSUBP(OpcodeArgs, IR::OpSize ElementSize);
|
||||
@@ -1089,8 +1099,8 @@ public:
|
||||
void AVX128_VFMAScalarImpl(OpcodeArgs, IROps IROp, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx);
|
||||
void AVX128_VFMAddSubImpl(OpcodeArgs, bool AddSub, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx);
|
||||
|
||||
RefPair AVX128_VPGatherQPSImpl(Ref Dest, Ref Mask, RefVSIB VSIB);
|
||||
RefPair AVX128_VPGatherImpl(OpSize Size, OpSize ElementLoadSize, OpSize AddrElementSize, RefPair Dest, RefPair Mask, RefVSIB VSIB);
|
||||
RefPair AVX128_VPGatherQPSImpl(OpcodeArgs, Ref Dest, Ref Mask, RefVSIB VSIB);
|
||||
RefPair AVX128_VPGatherImpl(OpcodeArgs, OpSize Size, OpSize ElementLoadSize, OpSize AddrElementSize, RefPair Dest, RefPair Mask, RefVSIB VSIB);
|
||||
|
||||
void AVX128_VPGATHER(OpcodeArgs, OpSize AddrElementSize);
|
||||
|
||||
@@ -1100,8 +1110,8 @@ public:
|
||||
// End of AVX 128-bit implementation
|
||||
|
||||
// AVX 256-bit operations
|
||||
void StoreResult_WithAVXInsert(VectorOpType Type, FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, Ref Value,
|
||||
IR::OpSize Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
|
||||
void StoreResult_WithAVXInsert(VectorOpType Type, RegClass Class, FEXCore::X86Tables::DecodedOp Op, Ref Value,
|
||||
IR::OpSize Align = IR::OpSize::iInvalid, MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
|
||||
if (Op->Dest.IsGPR() && Op->Dest.Data.GPR.GPR >= X86State::REG_XMM_0 && Op->Dest.Data.GPR.GPR <= X86State::REG_XMM_15 &&
|
||||
GetGuestVectorLength() == OpSize::i256Bit && Type == VectorOpType::SSE) {
|
||||
const auto gpr = Op->Dest.Data.GPR.GPR;
|
||||
@@ -1152,7 +1162,7 @@ public:
|
||||
}
|
||||
}
|
||||
|
||||
void StoreContextHelper(IR::OpSize Size, RegisterClassType Class, Ref Value, uint32_t Offset) {
|
||||
void StoreContextHelper(IR::OpSize Size, RegClass Class, Ref Value, uint32_t Offset) {
|
||||
// For i128Bit, we won't see a normal Constant to inline, but as a special
|
||||
// case we can replace with a 2x64-bit store which can use inline zeroes.
|
||||
if (Size == OpSize::i128Bit) {
|
||||
@@ -1164,7 +1174,7 @@ public:
|
||||
|
||||
if (Const->Constant == IR::NamedVectorConstant::NAMED_VECTOR_ZERO) {
|
||||
Ref Zero = _Constant(0);
|
||||
Ref STP = _StoreContextPair(IR::OpSize::i64Bit, GPRClass, Zero, Zero, Offset);
|
||||
Ref STP = _StoreContextPair(IR::OpSize::i64Bit, RegClass::GPR, Zero, Zero, Offset);
|
||||
|
||||
// XXX: This works around InlineConstant not having an associated
|
||||
// register class, else we'd just do InlineConstant above.
|
||||
@@ -1220,16 +1230,16 @@ public:
|
||||
|
||||
if (Index >= GPR0Index && Index <= GPR15Index) {
|
||||
Ref R = _StoreRegister(Value, GPRSize);
|
||||
R->Reg = PhysicalRegister(GPRFixedClass, Index - GPR0Index).Raw;
|
||||
R->Reg = PhysicalRegister(RegClass::GPRFixed, Index - GPR0Index).Raw;
|
||||
} else if (Index == PFIndex) {
|
||||
_StorePF(Value, GPRSize);
|
||||
} else if (Index == AFIndex) {
|
||||
_StoreAF(Value, GPRSize);
|
||||
} else if (Index >= FPR0Index && Index <= FPR15Index) {
|
||||
Ref R = _StoreRegister(Value, VectorSize);
|
||||
R->Reg = PhysicalRegister(FPRFixedClass, Index - FPR0Index).Raw;
|
||||
R->Reg = PhysicalRegister(RegClass::FPRFixed, Index - FPR0Index).Raw;
|
||||
} else if (Index == DFIndex) {
|
||||
_StoreContext(OpSize::i8Bit, GPRClass, Value, offsetof(Core::CPUState, flags[X86State::RFLAG_DF_RAW_LOC]));
|
||||
_StoreContextGPR(OpSize::i8Bit, Value, offsetof(Core::CPUState, flags[X86State::RFLAG_DF_RAW_LOC]));
|
||||
} else {
|
||||
bool Partial = RegCache.Partial & (1ull << Index);
|
||||
auto Size = Partial ? OpSize::i64Bit : CacheIndexToOpSize(Index);
|
||||
@@ -1254,7 +1264,7 @@ public:
|
||||
StoreContextHelper(Size, Class, Value, Offset);
|
||||
// If Partial and MMX register, then we need to store all 1s in bits 64-80
|
||||
if (Partial && Index >= MM0Index && Index <= MM7Index) {
|
||||
_StoreContext(OpSize::i16Bit, IR::GPRClass, Constant(0xFFFF), Offset + 8);
|
||||
_StoreContextGPR(OpSize::i16Bit, Constant(0xFFFF), Offset + 8);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1320,6 +1330,7 @@ protected:
|
||||
|
||||
private:
|
||||
FEX_CONFIG_OPT(ReducedPrecisionMode, X87REDUCEDPRECISION);
|
||||
FEX_CONFIG_OPT(StrictReducedPrecisionMode, X87STRICTREDUCEDPRECISION);
|
||||
|
||||
struct JumpTargetInfo {
|
||||
Ref BlockEntry;
|
||||
@@ -1544,23 +1555,63 @@ private:
|
||||
|
||||
AddressMode DecodeAddress(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, MemoryAccessType AccessType, bool IsLoad);
|
||||
|
||||
Ref LoadSource(RegisterClassType Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags,
|
||||
Ref LoadSource(RegClass Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags,
|
||||
const LoadSourceOptions& Options = {});
|
||||
Ref LoadSource_WithOpSize(RegisterClassType Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand,
|
||||
IR::OpSize OpSize, uint32_t Flags, const LoadSourceOptions& Options = {});
|
||||
void StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op,
|
||||
const FEXCore::X86Tables::DecodedOperand& Operand, const Ref Src, IR::OpSize OpSize, IR::OpSize Align,
|
||||
MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
|
||||
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, const FEXCore::X86Tables::DecodedOperand& Operand,
|
||||
const Ref Src, IR::OpSize Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
|
||||
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, const Ref Src, IR::OpSize Align,
|
||||
Ref LoadSourceGPR(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags,
|
||||
const LoadSourceOptions& Options = {}) {
|
||||
return LoadSource(RegClass::GPR, Op, Operand, Flags, Options);
|
||||
}
|
||||
Ref LoadSourceFPR(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags,
|
||||
const LoadSourceOptions& Options = {}) {
|
||||
return LoadSource(RegClass::FPR, Op, Operand, Flags, Options);
|
||||
}
|
||||
|
||||
Ref LoadSource_WithOpSize(RegClass Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, IR::OpSize OpSize,
|
||||
uint32_t Flags, const LoadSourceOptions& Options = {});
|
||||
Ref LoadSourceGPR_WithOpSize(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, IR::OpSize OpSize, uint32_t Flags,
|
||||
const LoadSourceOptions& Options = {}) {
|
||||
return LoadSource_WithOpSize(RegClass::GPR, Op, Operand, OpSize, Flags, Options);
|
||||
}
|
||||
Ref LoadSourceFPR_WithOpSize(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, IR::OpSize OpSize, uint32_t Flags,
|
||||
const LoadSourceOptions& Options = {}) {
|
||||
return LoadSource_WithOpSize(RegClass::FPR, Op, Operand, OpSize, Flags, Options);
|
||||
}
|
||||
|
||||
void StoreResult_WithOpSize(RegClass Class, X86Tables::DecodedOp Op, const X86Tables::DecodedOperand& Operand, Ref Src, IR::OpSize OpSize,
|
||||
IR::OpSize Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
|
||||
void StoreResultGPR_WithOpSize(X86Tables::DecodedOp Op, const X86Tables::DecodedOperand& Operand, Ref Src, IR::OpSize OpSize,
|
||||
IR::OpSize Align = IR::OpSize::iInvalid, MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
|
||||
StoreResult_WithOpSize(RegClass::GPR, Op, Operand, Src, OpSize, Align, AccessType);
|
||||
}
|
||||
void StoreResultFPR_WithOpSize(X86Tables::DecodedOp Op, const X86Tables::DecodedOperand& Operand, Ref Src, IR::OpSize OpSize,
|
||||
IR::OpSize Align = IR::OpSize::iInvalid, MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
|
||||
StoreResult_WithOpSize(RegClass::FPR, Op, Operand, Src, OpSize, Align, AccessType);
|
||||
}
|
||||
|
||||
void StoreResult(RegClass Class, X86Tables::DecodedOp Op, const X86Tables::DecodedOperand& Operand, Ref Src, OpSize Align,
|
||||
MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
|
||||
void StoreResultGPR(X86Tables::DecodedOp Op, const X86Tables::DecodedOperand& Operand, Ref Src, OpSize Align = OpSize::iInvalid,
|
||||
MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
|
||||
StoreResult(RegClass::GPR, Op, Operand, Src, Align, AccessType);
|
||||
}
|
||||
void StoreResultFPR(X86Tables::DecodedOp Op, const X86Tables::DecodedOperand& Operand, Ref Src, OpSize Align = OpSize::iInvalid,
|
||||
MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
|
||||
StoreResult(RegClass::FPR, Op, Operand, Src, Align, AccessType);
|
||||
}
|
||||
|
||||
void StoreResult(RegClass Class, X86Tables::DecodedOp Op, Ref Src, OpSize Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
|
||||
void StoreResultGPR(X86Tables::DecodedOp Op, Ref Src, OpSize Align = OpSize::iInvalid, MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
|
||||
StoreResult(RegClass::GPR, Op, Src, Align, AccessType);
|
||||
}
|
||||
void StoreResultFPR(X86Tables::DecodedOp Op, Ref Src, OpSize Align = OpSize::iInvalid, MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
|
||||
StoreResult(RegClass::FPR, Op, Src, Align, AccessType);
|
||||
}
|
||||
|
||||
// In several instances, it's desirable to get a base address with the segment offset
|
||||
// applied to it. This pulls all the common-case appending into a single set of functions.
|
||||
[[nodiscard]]
|
||||
Ref MakeSegmentAddress(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, IR::OpSize OpSize) {
|
||||
Ref Mem = LoadSource_WithOpSize(GPRClass, Op, Operand, OpSize, Op->Flags, {.LoadData = false});
|
||||
Ref Mem = LoadSourceGPR_WithOpSize(Op, Operand, OpSize, Op->Flags, {.LoadData = false});
|
||||
return AppendSegmentOffset(Mem, Op->Flags);
|
||||
}
|
||||
[[nodiscard]]
|
||||
@@ -1796,14 +1847,15 @@ private:
|
||||
// For DF, we need to transform 0/1 into 1/-1
|
||||
StoreDF(_SubShift(OpSize::i64Bit, Constant(1), Value, ShiftType::LSL, 1));
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_TF_RAW_LOC) {
|
||||
auto PackedTF = _LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
|
||||
auto PackedTF = _LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
|
||||
// An exception should still be raised after an instruction that unsets TF, leave the unblocked bit set but unset
|
||||
// the TF bit to cause such behaviour. The handling code at the start of the next block will then unset the
|
||||
// unblocked bit before raising the exception.
|
||||
auto NewPackedTF = _Select(FEXCore::IR::COND_EQ, Value, Constant(0), _And(OpSize::i32Bit, PackedTF, Constant(~1)), Constant(1));
|
||||
_StoreContext(OpSize::i8Bit, GPRClass, NewPackedTF, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
|
||||
auto NewPackedTF =
|
||||
_Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, Value, Constant(0), _And(OpSize::i32Bit, PackedTF, Constant(~1)), Constant(1));
|
||||
_StoreContextGPR(OpSize::i8Bit, NewPackedTF, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
|
||||
} else {
|
||||
_StoreContext(OpSize::i8Bit, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
|
||||
_StoreContextGPR(OpSize::i8Bit, Value, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1829,12 +1881,12 @@ private:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static CondClassType CondForNZCVBit(unsigned BitOffset, bool Invert) {
|
||||
static CondClass CondForNZCVBit(unsigned BitOffset, bool Invert) {
|
||||
switch (BitOffset) {
|
||||
case X86State::RFLAG_SF_RAW_LOC: return {Invert ? COND_PL : COND_MI};
|
||||
case X86State::RFLAG_ZF_RAW_LOC: return {Invert ? COND_NEQ : COND_EQ};
|
||||
case X86State::RFLAG_CF_RAW_LOC: return {Invert ? COND_ULT : COND_UGE};
|
||||
case X86State::RFLAG_OF_RAW_LOC: return {Invert ? COND_FNU : COND_FU};
|
||||
case X86State::RFLAG_SF_RAW_LOC: return Invert ? CondClass::PL : CondClass::MI;
|
||||
case X86State::RFLAG_ZF_RAW_LOC: return Invert ? CondClass::NEQ : CondClass::EQ;
|
||||
case X86State::RFLAG_CF_RAW_LOC: return Invert ? CondClass::ULT : CondClass::UGE;
|
||||
case X86State::RFLAG_OF_RAW_LOC: return Invert ? CondClass::FNU : CondClass::FU;
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
@@ -1845,10 +1897,10 @@ private:
|
||||
static const int PFIndex = 16;
|
||||
static const int AFIndex = 17;
|
||||
/* Gap 18..19 */
|
||||
/* Note this range is only valid if MMXState = MMXState_MMX */
|
||||
static const int MM0Index = 20;
|
||||
static const int MM7Index = 27;
|
||||
static const int AbridgedFTWIndex = 28;
|
||||
/* Gap 29..30 */
|
||||
/* Gap 28..30 */
|
||||
static const int DFIndex = 31;
|
||||
static const int FPR0Index = 32;
|
||||
static const int FPR15Index = 47;
|
||||
@@ -1860,17 +1912,16 @@ private:
|
||||
switch (Index) {
|
||||
case MM0Index ... MM7Index: return offsetof(FEXCore::Core::CPUState, mm[Index - MM0Index]);
|
||||
case AVXHigh0Index ... AVXHigh15Index: return offsetof(FEXCore::Core::CPUState, avx_high[Index - AVXHigh0Index][0]);
|
||||
case AbridgedFTWIndex: return offsetof(FEXCore::Core::CPUState, AbridgedFTW);
|
||||
default: return ~0U;
|
||||
}
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static RegisterClassType CacheIndexClass(int Index) {
|
||||
static RegClass CacheIndexClass(int Index) {
|
||||
if ((Index >= MM0Index && Index <= MM7Index) || Index >= FPR0Index) {
|
||||
return FPRClass;
|
||||
return RegClass::FPR;
|
||||
} else {
|
||||
return GPRClass;
|
||||
return RegClass::GPR;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1902,14 +1953,14 @@ private:
|
||||
RegCache.Written &= ~Bit;
|
||||
}
|
||||
|
||||
Ref LoadRegCache(uint64_t Offset, uint8_t Index, RegisterClassType RegClass, IR::OpSize Size) {
|
||||
Ref LoadRegCache(uint64_t Offset, uint8_t Index, RegClass Class, IR::OpSize Size) {
|
||||
LOGMAN_THROW_A_FMT(Index < 64, "valid index");
|
||||
uint64_t Bit = (1ull << (uint64_t)Index);
|
||||
|
||||
if (Size == OpSize::i128Bit && (RegCache.Partial & Bit)) {
|
||||
// We need to load the full register extend if we previously did a partial access.
|
||||
Ref Value = RegCache.Value[Index];
|
||||
Ref Full = _LoadContext(Size, RegClass, Offset);
|
||||
Ref Full = _LoadContext(Size, Class, Offset);
|
||||
|
||||
// If we did a partial store, we're inserting into the full register
|
||||
if (RegCache.Written & Bit) {
|
||||
@@ -1922,8 +1973,8 @@ private:
|
||||
if (!(RegCache.Cached & Bit)) {
|
||||
if (Index == DFIndex) {
|
||||
RegCache.Value[Index] = _LoadDF();
|
||||
} else if ((Index >= MM0Index && Index <= AbridgedFTWIndex) || Index >= AVXHigh0Index) {
|
||||
RegCache.Value[Index] = _LoadContext(Size, RegClass, Offset);
|
||||
} else if ((Index >= MM0Index && Index <= MM7Index) || Index >= AVXHigh0Index) {
|
||||
RegCache.Value[Index] = _LoadContext(Size, Class, Offset);
|
||||
|
||||
// We may have done a partial load, this requires special handling.
|
||||
if (Size == OpSize::i64Bit) {
|
||||
@@ -1934,7 +1985,7 @@ private:
|
||||
} else if (Index == AFIndex) {
|
||||
RegCache.Value[Index] = _LoadAF(Size);
|
||||
} else {
|
||||
RegCache.Value[Index] = _LoadRegister(Offset, RegClass, Size);
|
||||
RegCache.Value[Index] = _LoadRegister(Offset, Class, Size);
|
||||
}
|
||||
|
||||
RegCache.Cached |= Bit;
|
||||
@@ -1943,21 +1994,21 @@ private:
|
||||
return RegCache.Value[Index];
|
||||
}
|
||||
|
||||
RefPair AllocatePair(FEXCore::IR::RegisterClassType Class, IR::OpSize Size) {
|
||||
if (Class == FPRClass) {
|
||||
RefPair AllocatePair(RegClass Class, IR::OpSize Size) {
|
||||
if (Class == RegClass::FPR) {
|
||||
return {_AllocateFPR(Size, Size), _AllocateFPR(Size, Size)};
|
||||
} else {
|
||||
return {_AllocateGPR(false), _AllocateGPR(false)};
|
||||
}
|
||||
}
|
||||
|
||||
RefPair LoadContextPair_Uncached(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, unsigned Offset) {
|
||||
RefPair LoadContextPair_Uncached(RegClass Class, IR::OpSize Size, unsigned Offset) {
|
||||
RefPair Values = AllocatePair(Class, Size);
|
||||
_LoadContextPair(Size, Class, Offset, Values.Low, Values.High);
|
||||
return Values;
|
||||
}
|
||||
|
||||
RefPair LoadRegCachePair(uint64_t Offset, uint8_t Index, RegisterClassType RegClass, IR::OpSize Size) {
|
||||
RefPair LoadRegCachePair(uint64_t Offset, uint8_t Index, RegClass Class, IR::OpSize Size) {
|
||||
LOGMAN_THROW_A_FMT(Index != DFIndex, "must be pairable");
|
||||
LOGMAN_THROW_A_FMT(Size != IR::OpSize::iUnsized, "Invalid size!");
|
||||
|
||||
@@ -1965,7 +2016,7 @@ private:
|
||||
uint64_t Bits = (3ull << (uint64_t)Index);
|
||||
const auto SizeInt = IR::OpSizeToSize(Size);
|
||||
if (((RegCache.Partial | RegCache.Cached) & Bits) == 0 && ((Offset / SizeInt) < 64)) {
|
||||
auto Values = LoadContextPair_Uncached(RegClass, Size, Offset);
|
||||
auto Values = LoadContextPair_Uncached(Class, Size, Offset);
|
||||
RegCache.Value[Index] = Values.Low;
|
||||
RegCache.Value[Index + 1] = Values.High;
|
||||
RegCache.Cached |= Bits;
|
||||
@@ -1977,13 +2028,13 @@ private:
|
||||
|
||||
// Fallback on a pair of loads
|
||||
return {
|
||||
.Low = LoadRegCache(Offset, Index, RegClass, Size),
|
||||
.High = LoadRegCache(Offset + SizeInt, Index + 1, RegClass, Size),
|
||||
.Low = LoadRegCache(Offset, Index, Class, Size),
|
||||
.High = LoadRegCache(Offset + SizeInt, Index + 1, Class, Size),
|
||||
};
|
||||
}
|
||||
|
||||
Ref LoadGPR(uint8_t Reg) {
|
||||
return LoadRegCache(Reg, GPR0Index + Reg, GPRClass, GetGPROpSize());
|
||||
return LoadRegCache(Reg, GPR0Index + Reg, RegClass::GPR, GetGPROpSize());
|
||||
}
|
||||
|
||||
Ref LoadContext(IR::OpSize Size, uint8_t Index) {
|
||||
@@ -1999,7 +2050,7 @@ private:
|
||||
}
|
||||
|
||||
Ref LoadXMMRegister(uint8_t Reg) {
|
||||
return LoadRegCache(Reg, FPR0Index + Reg, FPRClass, GetGuestVectorLength());
|
||||
return LoadRegCache(Reg, FPR0Index + Reg, RegClass::FPR, GetGuestVectorLength());
|
||||
}
|
||||
|
||||
Ref LoadDF() {
|
||||
@@ -2054,7 +2105,7 @@ private:
|
||||
// Recover the sign bit, it is the logical DF value
|
||||
return _Lshr(OpSize::i64Bit, LoadDF(), Constant(63));
|
||||
} else {
|
||||
return _LoadContext(OpSize::i8Bit, GPRClass, offsetof(Core::CPUState, flags[BitOffset]));
|
||||
return _LoadContextGPR(OpSize::i8Bit, offsetof(Core::CPUState, flags[BitOffset]));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2071,18 +2122,18 @@ private:
|
||||
}
|
||||
|
||||
// Safe version of NZCVSelect that handles inverted carries automatically.
|
||||
Ref NZCVSelect(OpSize OpSize, CondClassType Cond, Ref TrueV, Ref FalseV, bool CarryIsInverted = false) {
|
||||
Ref NZCVSelect(OpSize OpSize, CondClass Cond, Ref TrueV, Ref FalseV, bool CarryIsInverted = false) {
|
||||
switch (Cond) {
|
||||
case IR::COND_UGE: /* cs */
|
||||
case IR::COND_ULT: /* cc */
|
||||
case CondClass::UGE: /* cs */
|
||||
case CondClass::ULT: /* cc */
|
||||
// Invert the condition to match our expectations.
|
||||
if (CarryIsInverted != CFInverted) {
|
||||
Cond = {Cond == COND_UGE ? COND_ULT : COND_UGE};
|
||||
Cond = (Cond == CondClass::UGE) ? CondClass::ULT : CondClass::UGE;
|
||||
}
|
||||
break;
|
||||
|
||||
case IR::COND_UGT: /* hi */
|
||||
case IR::COND_ULE: /* ls */
|
||||
case CondClass::UGT: /* hi */
|
||||
case CondClass::ULE: /* ls */
|
||||
// No clever optimization we can do here, rectify carry itself.
|
||||
RectifyCarryInvert(CarryIsInverted);
|
||||
break;
|
||||
@@ -2171,7 +2222,7 @@ private:
|
||||
|
||||
HandleNZCV_RMW();
|
||||
CalculatePF(_ShiftFlags(OpSizeFromSrc(Op), Result, Dest, Shift, Src, OldPF, CFInverted));
|
||||
StoreResult(GPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultGPR(Op, Result);
|
||||
}
|
||||
|
||||
// Helper to derive Dest by a given builder-using Expression with the opcode
|
||||
@@ -2244,8 +2295,7 @@ private:
|
||||
CachedIndexedNamedVectorConstants.clear();
|
||||
}
|
||||
|
||||
std::optional<CondClassType> DecodeNZCVCondition(uint8_t OP);
|
||||
Ref SelectBit(Ref Cmp, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue);
|
||||
std::optional<CondClass> DecodeNZCVCondition(uint8_t OP);
|
||||
Ref SelectCC0All1(uint8_t OP);
|
||||
|
||||
/**
|
||||
@@ -2259,8 +2309,8 @@ private:
|
||||
if (Size != OpSize::i32Bit) {
|
||||
return;
|
||||
}
|
||||
auto Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags);
|
||||
StoreResult(GPRClass, Op, Dest, OpSize::iInvalid);
|
||||
auto Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags);
|
||||
StoreResultGPR(Op, Dest);
|
||||
}
|
||||
|
||||
using ZeroShiftFunctionPtr = void (OpDispatchBuilder::*)(FEXCore::X86Tables::DecodedOp Op);
|
||||
@@ -2295,7 +2345,7 @@ private:
|
||||
|
||||
///< Jump to zeroshift block or end block depending on if it was provided.
|
||||
IRPair<IROp_CodeBlock> TailHandling = ZeroShiftResult ? ZeroShiftBlock : EndBlock;
|
||||
CondJump(Shift, Zero, TailHandling, SetBlock, {COND_EQ});
|
||||
CondJump(Shift, Zero, TailHandling, SetBlock, CondClass::EQ);
|
||||
|
||||
SetCurrentCodeBlock(SetBlock);
|
||||
StartNewBlock();
|
||||
@@ -2338,9 +2388,7 @@ private:
|
||||
void CalculateFlags_MUL(IR::OpSize SrcSize, Ref Res, Ref High);
|
||||
void CalculateFlags_UMUL(Ref High);
|
||||
void CalculateFlags_Logical(IR::OpSize SrcSize, Ref Res);
|
||||
void CalculateFlags_ShiftLeft(IR::OpSize SrcSize, Ref Res, Ref Src1, Ref Src2);
|
||||
void CalculateFlags_ShiftLeftImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRight(IR::OpSize SrcSize, Ref Res, Ref Src1, Ref Src2);
|
||||
void CalculateFlags_ShiftRightImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRightDoubleImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRightImmediateCommon(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
@@ -2356,8 +2404,8 @@ private:
|
||||
void ChgStateX87_MMX() override {
|
||||
LOGMAN_THROW_A_FMT(MMXState == MMXState_X87, "Expected state to be x87");
|
||||
_StackForceSlow();
|
||||
SetX87Top(Constant(0)); // top reset to zero
|
||||
StoreContext(AbridgedFTWIndex, Constant(0xFFFFUL)); // all valid
|
||||
SetX87Top(Constant(0)); // top reset to zero
|
||||
_StoreContextGPR(OpSize::i8Bit, Constant(0xFFFFUL), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
MMXState = MMXState_MMX;
|
||||
}
|
||||
|
||||
@@ -2394,44 +2442,62 @@ private:
|
||||
IROp_IRHeader* CurrentHeader {};
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsTSOEnabled(FEXCore::IR::RegisterClassType Class) const {
|
||||
bool IsTSOEnabled(RegClass Class) const {
|
||||
if (ForceTSO == ForceTSOMode::ForceEnabled) {
|
||||
return true;
|
||||
} else if (ForceTSO == ForceTSOMode::ForceDisabled) {
|
||||
return false;
|
||||
} else if (Class == FPRClass) {
|
||||
} else if (Class == RegClass::FPR) {
|
||||
return CTX->IsVectorAtomicTSOEnabled();
|
||||
} else {
|
||||
return CTX->IsAtomicTSOEnabled();
|
||||
}
|
||||
}
|
||||
|
||||
Ref _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, Ref Addr, Ref Value, IR::OpSize Align = IR::OpSize::i8Bit) {
|
||||
Ref _StoreMemAutoTSO(RegClass Class, OpSize Size, Ref Addr, Ref Value, OpSize Align = OpSize::i8Bit) {
|
||||
if (IsTSOEnabled(Class)) {
|
||||
return _StoreMemTSO(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
return _StoreMemTSO(Class, Size, Value, Addr, Invalid(), Align, MemOffsetType::SXTX, 1);
|
||||
} else {
|
||||
return _StoreMem(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
return _StoreMem(Class, Size, Value, Addr, Invalid(), Align, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
}
|
||||
|
||||
Ref _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, Ref ssa0, IR::OpSize Align = IR::OpSize::i8Bit) {
|
||||
if (IsTSOEnabled(Class)) {
|
||||
return _LoadMemTSO(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
} else {
|
||||
return _LoadMem(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
Ref _StoreMemGPRAutoTSO(OpSize Size, Ref Addr, Ref Value, OpSize Align = OpSize::i8Bit) {
|
||||
return _StoreMemAutoTSO(RegClass::GPR, Size, Addr, Value, Align);
|
||||
}
|
||||
Ref _StoreMemFPRAutoTSO(OpSize Size, Ref Addr, Ref Value, OpSize Align = OpSize::i8Bit) {
|
||||
return _StoreMemAutoTSO(RegClass::FPR, Size, Addr, Value, Align);
|
||||
}
|
||||
|
||||
Ref _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, AddressMode A, IR::OpSize Align = IR::OpSize::i8Bit) {
|
||||
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
A = SelectAddressMode(this, A, GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, AtomicTSO, Class != GPRClass, Size);
|
||||
Ref _LoadMemAutoTSO(RegClass Class, OpSize Size, Ref ssa0, OpSize Align = OpSize::i8Bit) {
|
||||
if (IsTSOEnabled(Class)) {
|
||||
return _LoadMemTSO(Class, Size, ssa0, Invalid(), Align, MemOffsetType::SXTX, 1);
|
||||
} else {
|
||||
return _LoadMem(Class, Size, ssa0, Invalid(), Align, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
}
|
||||
Ref _LoadMemGPRAutoTSO(OpSize Size, Ref ssa0, OpSize Align = OpSize::i8Bit) {
|
||||
return _LoadMemAutoTSO(RegClass::GPR, Size, ssa0, Align);
|
||||
}
|
||||
Ref _LoadMemFPRAutoTSO(OpSize Size, Ref ssa0, OpSize Align = OpSize::i8Bit) {
|
||||
return _LoadMemAutoTSO(RegClass::FPR, Size, ssa0, Align);
|
||||
}
|
||||
|
||||
Ref _LoadMemAutoTSO(RegClass Class, OpSize Size, const AddressMode& A, OpSize Align = OpSize::i8Bit) {
|
||||
const bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
const auto B = SelectAddressMode(this, A, GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, AtomicTSO, Class != RegClass::GPR, Size);
|
||||
|
||||
if (AtomicTSO) {
|
||||
return _LoadMemTSO(Class, Size, A.Base, A.Index, Align, A.IndexType, A.IndexScale);
|
||||
return _LoadMemTSO(Class, Size, B.Base, B.Index, Align, B.IndexType, B.IndexScale);
|
||||
} else {
|
||||
return _LoadMem(Class, Size, A.Base, A.Index, Align, A.IndexType, A.IndexScale);
|
||||
return _LoadMem(Class, Size, B.Base, B.Index, Align, B.IndexType, B.IndexScale);
|
||||
}
|
||||
}
|
||||
Ref _LoadMemGPRAutoTSO(OpSize Size, const AddressMode& A, OpSize Align = OpSize::i8Bit) {
|
||||
return _LoadMemAutoTSO(RegClass::GPR, Size, A, Align);
|
||||
}
|
||||
Ref _LoadMemFPRAutoTSO(OpSize Size, const AddressMode& A, OpSize Align = OpSize::i8Bit) {
|
||||
return _LoadMemAutoTSO(RegClass::FPR, Size, A, Align);
|
||||
}
|
||||
|
||||
AddressMode SelectPairAddressMode(AddressMode A, IR::OpSize Size) {
|
||||
LOGMAN_THROW_A_FMT(Size != IR::OpSize::iUnsized, "Invalid size!");
|
||||
@@ -2449,56 +2515,72 @@ private:
|
||||
}
|
||||
|
||||
|
||||
RefPair LoadMemPair(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, Ref Base, unsigned Offset) {
|
||||
RefPair LoadMemPair(RegClass Class, OpSize Size, Ref Base, uint32_t Offset) {
|
||||
RefPair Values = AllocatePair(Class, Size);
|
||||
_LoadMemPair(Class, Size, Base, Offset, Values.Low, Values.High);
|
||||
return Values;
|
||||
}
|
||||
RefPair LoadMemPairFPR(OpSize Size, Ref Base, uint32_t Offset) {
|
||||
return LoadMemPair(RegClass::FPR, Size, Base, Offset);
|
||||
}
|
||||
|
||||
RefPair _LoadMemPairAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, AddressMode A, IR::OpSize Align = IR::OpSize::i8Bit) {
|
||||
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
RefPair _LoadMemPairAutoTSO(RegClass Class, OpSize Size, const AddressMode& A, OpSize Align = OpSize::i8Bit) {
|
||||
const bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
|
||||
// Use ldp if possible, otherwise fallback on two loads.
|
||||
if (!AtomicTSO && !A.Segment && Size >= OpSize::i32Bit & Size <= OpSize::i128Bit) {
|
||||
A = SelectPairAddressMode(A, Size);
|
||||
return LoadMemPair(Class, Size, A.Base, A.Offset);
|
||||
} else {
|
||||
AddressMode HighA = A;
|
||||
HighA.Offset += 16;
|
||||
|
||||
return {
|
||||
.Low = _LoadMemAutoTSO(Class, Size, A, Align),
|
||||
.High = _LoadMemAutoTSO(Class, Size, HighA, Align),
|
||||
};
|
||||
const auto B = SelectPairAddressMode(A, Size);
|
||||
return LoadMemPair(Class, Size, B.Base, B.Offset);
|
||||
}
|
||||
|
||||
AddressMode HighA = A;
|
||||
HighA.Offset += 16;
|
||||
|
||||
return {
|
||||
.Low = _LoadMemAutoTSO(Class, Size, A, Align),
|
||||
.High = _LoadMemAutoTSO(Class, Size, HighA, Align),
|
||||
};
|
||||
}
|
||||
RefPair _LoadMemPairFPRAutoTSO(OpSize Size, const AddressMode& A, OpSize Align = OpSize::i8Bit) {
|
||||
return _LoadMemPairAutoTSO(RegClass::FPR, Size, A, Align);
|
||||
}
|
||||
|
||||
Ref _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, AddressMode A, Ref Value, IR::OpSize Align = IR::OpSize::i8Bit) {
|
||||
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
A = SelectAddressMode(this, A, GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, AtomicTSO, Class != GPRClass, Size);
|
||||
Ref _StoreMemAutoTSO(RegClass Class, OpSize Size, const AddressMode& A, Ref Value, OpSize Align = OpSize::i8Bit) {
|
||||
const bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
const auto B = SelectAddressMode(this, A, GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, AtomicTSO, Class != RegClass::GPR, Size);
|
||||
|
||||
if (AtomicTSO) {
|
||||
return _StoreMemTSO(Class, Size, Value, A.Base, A.Index, Align, A.IndexType, A.IndexScale);
|
||||
return _StoreMemTSO(Class, Size, Value, B.Base, B.Index, Align, B.IndexType, B.IndexScale);
|
||||
} else {
|
||||
return _StoreMem(Class, Size, Value, A.Base, A.Index, Align, A.IndexType, A.IndexScale);
|
||||
return _StoreMem(Class, Size, Value, B.Base, B.Index, Align, B.IndexType, B.IndexScale);
|
||||
}
|
||||
}
|
||||
Ref _StoreMemGPRAutoTSO(OpSize Size, const AddressMode& A, Ref Value, OpSize Align = OpSize::i8Bit) {
|
||||
return _StoreMemAutoTSO(RegClass::GPR, Size, A, Value, Align);
|
||||
}
|
||||
Ref _StoreMemFPRAutoTSO(OpSize Size, const AddressMode& A, Ref Value, OpSize Align = OpSize::i8Bit) {
|
||||
return _StoreMemAutoTSO(RegClass::FPR, Size, A, Value, Align);
|
||||
}
|
||||
|
||||
void _StoreMemPairAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, AddressMode A, Ref Value1, Ref Value2,
|
||||
IR::OpSize Align = IR::OpSize::i8Bit) {
|
||||
void _StoreMemPairAutoTSO(RegClass Class, OpSize Size, const AddressMode& A, Ref Value1, Ref Value2, OpSize Align = OpSize::i8Bit) {
|
||||
const auto SizeInt = IR::OpSizeToSize(Size);
|
||||
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
const bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
|
||||
// Use stp if possible, otherwise fallback on two stores.
|
||||
if (!AtomicTSO && !A.Segment && Size >= OpSize::i32Bit & Size <= OpSize::i128Bit) {
|
||||
A = SelectPairAddressMode(A, Size);
|
||||
_StoreMemPair(Class, Size, Value1, Value2, A.Base, A.Offset);
|
||||
const auto B = SelectPairAddressMode(A, Size);
|
||||
_StoreMemPair(Class, Size, Value1, Value2, B.Base, B.Offset);
|
||||
} else {
|
||||
_StoreMemAutoTSO(Class, Size, A, Value1, OpSize::i8Bit);
|
||||
A.Offset += SizeInt;
|
||||
_StoreMemAutoTSO(Class, Size, A, Value2, OpSize::i8Bit);
|
||||
auto B = A;
|
||||
|
||||
_StoreMemAutoTSO(Class, Size, B, Value1, OpSize::i8Bit);
|
||||
B.Offset += SizeInt;
|
||||
_StoreMemAutoTSO(Class, Size, B, Value2, OpSize::i8Bit);
|
||||
}
|
||||
}
|
||||
void _StoreMemPairFPRAutoTSO(OpSize Size, const AddressMode& A, Ref Value1, Ref Value2, OpSize Align = OpSize::i8Bit) {
|
||||
return _StoreMemPairAutoTSO(RegClass::FPR, Size, A, Value1, Value2, Align);
|
||||
}
|
||||
|
||||
Ref Pop(IR::OpSize Size, Ref SP_RMW) {
|
||||
Ref Value = _AllocateGPR(false);
|
||||
|
||||
@@ -35,20 +35,16 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_LoadSource_WithOpSize(
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(IsOperandMem(Operand, true), "only memory sources");
|
||||
|
||||
AddressMode A = DecodeAddress(Op, Operand, AccessType, true /* IsLoad */);
|
||||
|
||||
AddressMode HighA = A;
|
||||
HighA.Offset += 16;
|
||||
|
||||
if (Operand.IsSIB()) {
|
||||
const bool IsVSIB = (Op->Flags & X86Tables::DecodeFlags::FLAG_VSIB_BYTE) != 0;
|
||||
LOGMAN_THROW_A_FMT(!IsVSIB, "VSIB uses LoadVSIB instead");
|
||||
}
|
||||
|
||||
const AddressMode A = DecodeAddress(Op, Operand, AccessType, true /* IsLoad */);
|
||||
if (NeedsHigh) {
|
||||
return _LoadMemPairAutoTSO(FPRClass, OpSize::i128Bit, A, OpSize::i8Bit);
|
||||
return _LoadMemPairFPRAutoTSO(OpSize::i128Bit, A, OpSize::i8Bit);
|
||||
} else {
|
||||
return {.Low = _LoadMemAutoTSO(FPRClass, OpSize::i128Bit, A, OpSize::i8Bit)};
|
||||
return {.Low = _LoadMemFPRAutoTSO(OpSize::i128Bit, A, OpSize::i8Bit)};
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -95,9 +91,9 @@ void OpDispatchBuilder::AVX128_StoreResult_WithOpSize(FEXCore::X86Tables::Decode
|
||||
AddressMode A = DecodeAddress(Op, Operand, AccessType, false /* IsLoad */);
|
||||
|
||||
if (Src.High) {
|
||||
_StoreMemPairAutoTSO(FPRClass, OpSize::i128Bit, A, Src.Low, Src.High, OpSize::i8Bit);
|
||||
_StoreMemPairFPRAutoTSO(OpSize::i128Bit, A, Src.Low, Src.High, OpSize::i8Bit);
|
||||
} else {
|
||||
_StoreMemAutoTSO(FPRClass, OpSize::i128Bit, A, Src.Low, OpSize::i8Bit);
|
||||
_StoreMemFPRAutoTSO(OpSize::i128Bit, A, Src.Low, OpSize::i8Bit);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -151,13 +147,13 @@ void OpDispatchBuilder::AVX128_VMOVScalarImpl(OpcodeArgs, IR::OpSize ElementSize
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, RefPair {.Low = Result, .High = High});
|
||||
} else if (Op->Dest.IsGPR()) {
|
||||
// VMOVSS/SD xmm1, mem32/mem64
|
||||
Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], ElementSize, Op->Flags);
|
||||
Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[1], ElementSize, Op->Flags);
|
||||
auto High = LoadZeroVector(OpSize::i128Bit);
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, RefPair {.Low = Src, .High = High});
|
||||
} else {
|
||||
// VMOVSS/SD mem32/mem64, xmm1
|
||||
auto Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src.Low, ElementSize, OpSize::iInvalid);
|
||||
StoreResultFPR_WithOpSize(Op, Op->Dest, Src.Low, ElementSize);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -351,7 +347,7 @@ void OpDispatchBuilder::AVX128_MOVVectorNT(OpcodeArgs) {
|
||||
if (Op->Dest.IsGPR()) {
|
||||
///< MOVNTDQA load non-temporal comes from SSE4.1 and is extended by AVX/AVX2.
|
||||
RefPair Src {};
|
||||
Ref SrcAddr = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Ref SrcAddr = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Src.Low = _VLoadNonTemporal(OpSize::i128Bit, SrcAddr, 0);
|
||||
|
||||
if (Is128Bit) {
|
||||
@@ -362,7 +358,7 @@ void OpDispatchBuilder::AVX128_MOVVectorNT(OpcodeArgs) {
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Src);
|
||||
} else {
|
||||
auto Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, !Is128Bit, MemoryAccessType::STREAM);
|
||||
Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
|
||||
if (Is128Bit) {
|
||||
// Single store non-temporal for 128-bit operations.
|
||||
@@ -379,7 +375,7 @@ void OpDispatchBuilder::AVX128_MOVQ(OpcodeArgs) {
|
||||
if (Op->Src[0].IsGPR()) {
|
||||
Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false);
|
||||
} else {
|
||||
Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::i64Bit, Op->Flags);
|
||||
Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::i64Bit, Op->Flags);
|
||||
}
|
||||
|
||||
// This instruction is a bit special that if the destination is a register then it'll ZEXT the 64bit source to 256bit
|
||||
@@ -390,7 +386,7 @@ void OpDispatchBuilder::AVX128_MOVQ(OpcodeArgs) {
|
||||
Src.High = ZeroVector;
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Src);
|
||||
} else {
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src.Low, OpSize::i64Bit, OpSize::i64Bit);
|
||||
StoreResultFPR_WithOpSize(Op, Op->Dest, Src.Low, OpSize::i64Bit, OpSize::i64Bit);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -399,7 +395,7 @@ void OpDispatchBuilder::AVX128_VMOVLP(OpcodeArgs) {
|
||||
|
||||
if (!Op->Dest.IsGPR()) {
|
||||
///< VMOVLPS/PD mem64, xmm1
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src1.Low, OpSize::i64Bit, OpSize::i64Bit);
|
||||
StoreResultFPR_WithOpSize(Op, Op->Dest, Src1.Low, OpSize::i64Bit, OpSize::i64Bit);
|
||||
} else if (!Op->Src[1].IsGPR()) {
|
||||
///< VMOVLPS/PD xmm1, xmm2, mem64
|
||||
// Bits[63:0] come from Src2[63:0]
|
||||
@@ -463,7 +459,7 @@ void OpDispatchBuilder::AVX128_VMOVDDUP(OpcodeArgs) {
|
||||
// 128-bit operation only loads 8-bytes.
|
||||
// 256-bit operation loads a full 32-bytes.
|
||||
if (Is128Bit) {
|
||||
Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::i64Bit, Op->Flags);
|
||||
Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::i64Bit, Op->Flags);
|
||||
} else {
|
||||
Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, true);
|
||||
}
|
||||
@@ -558,18 +554,18 @@ void OpDispatchBuilder::AVX128_InsertCVTGPR_To_FPR(OpcodeArgs, IR::OpSize DstEle
|
||||
|
||||
if (Op->Src[1].IsGPR()) {
|
||||
// If the source is a GPR then convert directly from the GPR.
|
||||
auto Src2 = LoadSource_WithOpSize(GPRClass, Op, Op->Src[1], GetGPROpSize(), Op->Flags);
|
||||
auto Src2 = LoadSourceGPR_WithOpSize(Op, Op->Src[1], GetGPROpSize(), Op->Flags);
|
||||
Result.Low = _VSToFGPRInsert(OpSize::i128Bit, DstElementSize, SrcSize, Src1.Low, Src2, false);
|
||||
} else if (SrcSize != DstElementSize) {
|
||||
// If the source is from memory but the Source size and destination size aren't the same,
|
||||
// then it is more optimal to load in to a GPR and convert between GPR->FPR.
|
||||
// ARM GPR->FPR conversion supports different size source and destinations while FPR->FPR doesn't.
|
||||
auto Src2 = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags);
|
||||
auto Src2 = LoadSourceGPR(Op, Op->Src[1], Op->Flags);
|
||||
Result.Low = _VSToFGPRInsert(DstSize, DstElementSize, SrcSize, Src1.Low, Src2, false);
|
||||
} else {
|
||||
// In the case of cvtsi2s{s,d} where the source and destination are the same size,
|
||||
// then it is more optimal to load in to the FPR register directly and convert there.
|
||||
auto Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
auto Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
// Always signed
|
||||
Result.Low = _VSToFVectorInsert(DstSize, DstElementSize, DstElementSize, Src1.Low, Src2, false, false);
|
||||
}
|
||||
@@ -589,11 +585,11 @@ void OpDispatchBuilder::AVX128_CVTFPR_To_GPR(OpcodeArgs, IR::OpSize SrcElementSi
|
||||
if (Op->Src[0].IsGPR()) {
|
||||
Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false);
|
||||
} else {
|
||||
Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcElementSize, Op->Flags);
|
||||
Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcElementSize, Op->Flags);
|
||||
}
|
||||
|
||||
Ref Result = CVTFPR_To_GPRImpl(Op, Src.Low, SrcElementSize, HostRoundingMode);
|
||||
StoreResult(GPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultGPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VANDN(OpcodeArgs) {
|
||||
@@ -636,7 +632,7 @@ void OpDispatchBuilder::AVX128_UCOMISx(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
if (Op->Src[0].IsGPR()) {
|
||||
Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false);
|
||||
} else {
|
||||
Src2.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
Src2.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
}
|
||||
|
||||
Comiss(ElementSize, Src1.Low, Src2.Low);
|
||||
@@ -653,7 +649,7 @@ void OpDispatchBuilder::AVX128_VectorScalarInsertALU(OpcodeArgs, FEXCore::IR::IR
|
||||
if (Op->Src[1].IsGPR()) {
|
||||
Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false);
|
||||
} else {
|
||||
Src2.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags);
|
||||
Src2.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags);
|
||||
}
|
||||
|
||||
// If OpSize == ElementSize then it only does the lower scalar op
|
||||
@@ -690,7 +686,7 @@ void OpDispatchBuilder::AVX128_InsertScalarFCMP(OpcodeArgs, IR::OpSize ElementSi
|
||||
if (Op->Src[1].IsGPR()) {
|
||||
Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false);
|
||||
} else {
|
||||
Src2.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags);
|
||||
Src2.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags);
|
||||
}
|
||||
|
||||
const uint8_t CompType = Op->Src[2].Literal();
|
||||
@@ -708,12 +704,12 @@ void OpDispatchBuilder::AVX128_MOVBetweenGPR_FPR(OpcodeArgs) {
|
||||
RefPair Result {};
|
||||
if (Op->Src[0].IsGPR()) {
|
||||
// Loading from GPR and moving to Vector.
|
||||
Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], GetGPROpSize(), Op->Flags);
|
||||
Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[0], GetGPROpSize(), Op->Flags);
|
||||
// zext to 128bit
|
||||
Result.Low = _VCastFromGPR(OpSize::i128Bit, OpSizeFromSrc(Op), Src);
|
||||
} else {
|
||||
// Loading from Memory as a scalar. Zero extend
|
||||
Result.Low = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Result.Low = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
|
||||
Result.High = LoadZeroVector(OpSize::i128Bit);
|
||||
@@ -726,11 +722,11 @@ void OpDispatchBuilder::AVX128_MOVBetweenGPR_FPR(OpcodeArgs) {
|
||||
auto ElementSize = OpSizeFromDst(Op);
|
||||
// Extract element from GPR. Zero extending in the process.
|
||||
Src.Low = _VExtractToGPR(OpSizeFromSrc(Op), ElementSize, Src.Low, 0);
|
||||
StoreResult(GPRClass, Op, Op->Dest, Src.Low, OpSize::iInvalid);
|
||||
StoreResultGPR(Op, Op->Dest, Src.Low);
|
||||
} else {
|
||||
// Storing first element to memory.
|
||||
Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
_StoreMem(FPRClass, OpSizeFromDst(Op), Dest, Src.Low, OpSize::i8Bit);
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
_StoreMemFPR(OpSizeFromDst(Op), Dest, Src.Low, OpSize::i8Bit);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -758,7 +754,7 @@ void OpDispatchBuilder::AVX128_PExtr(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
// Extract already zero extends the result.
|
||||
Ref Result = _VExtractToGPR(OpSize::i128Bit, OverridenElementSize, Src.Low, Index);
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Result, GPRSize, OpSize::iInvalid);
|
||||
StoreResultGPR_WithOpSize(Op, Op->Dest, Result, GPRSize);
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -779,7 +775,7 @@ void OpDispatchBuilder::AVX128_ExtendVectorElements(OpcodeArgs, IR::OpSize Eleme
|
||||
const auto SrcSize = OpSizeFromSrc(Op);
|
||||
const auto LoadSize = Is256Bit ? IR::SizeToOpSize(IR::OpSizeToSize(SrcSize) * 2) : SrcSize;
|
||||
|
||||
return LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], LoadSize, Op->Flags);
|
||||
return LoadSourceFPR_WithOpSize(Op, Op->Src[0], LoadSize, Op->Flags);
|
||||
}
|
||||
};
|
||||
|
||||
@@ -868,7 +864,7 @@ void OpDispatchBuilder::AVX128_MOVMSK(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
auto GPRHigh = Mask8Byte(Src.High);
|
||||
GPR = _Orlshl(OpSize::i64Bit, GPRLow, GPRHigh, 2);
|
||||
}
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, GPR, GetGPROpSize(), OpSize::iInvalid);
|
||||
StoreResultGPR_WithOpSize(Op, Op->Dest, GPR, GetGPROpSize());
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_MOVMSKB(OpcodeArgs) {
|
||||
@@ -897,7 +893,7 @@ void OpDispatchBuilder::AVX128_MOVMSKB(OpcodeArgs) {
|
||||
Result = _Orlshl(OpSize::i64Bit, Result, ResultHigh, 16);
|
||||
}
|
||||
|
||||
StoreResult(GPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultGPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_PINSRImpl(OpcodeArgs, IR::OpSize ElementSize, const X86Tables::DecodedOperand& Src1Op,
|
||||
@@ -910,7 +906,7 @@ void OpDispatchBuilder::AVX128_PINSRImpl(OpcodeArgs, IR::OpSize ElementSize, con
|
||||
|
||||
if (Src2Op.IsGPR()) {
|
||||
// If the source is a GPR then convert directly from the GPR.
|
||||
auto Src2 = LoadSource_WithOpSize(GPRClass, Op, Src2Op, GetGPROpSize(), Op->Flags);
|
||||
auto Src2 = LoadSourceGPR_WithOpSize(Op, Src2Op, GetGPROpSize(), Op->Flags);
|
||||
Result.Low = _VInsGPR(OpSize::i128Bit, ElementSize, Index, Src1.Low, Src2);
|
||||
} else {
|
||||
// If loading from memory then we only load the element size
|
||||
@@ -1047,7 +1043,7 @@ void OpDispatchBuilder::AVX128_InsertScalar_CVT_Float_To_Float(OpcodeArgs, IR::O
|
||||
// Then zero extends the top 128-bit.
|
||||
const auto SrcSize = Op->Src[1].IsGPR() ? OpSize::i128Bit : SrcElementSize;
|
||||
auto Src1 = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false);
|
||||
Ref Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags, {.AllowUpperGarbage = true});
|
||||
Ref Src2 = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
Ref Result = _VFToFScalarInsert(OpSize::i128Bit, DstElementSize, SrcElementSize, Src1.Low, Src2, false);
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, AVX128_Zext(Result));
|
||||
@@ -1076,7 +1072,7 @@ void OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Float(OpcodeArgs, IR::OpSize
|
||||
} else {
|
||||
// Handle 64-bit memory source.
|
||||
// In the case of cvtps2pd xmm, m64.
|
||||
Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], LoadSize, Op->Flags);
|
||||
Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], LoadSize, Op->Flags);
|
||||
}
|
||||
|
||||
RefPair Result {};
|
||||
@@ -1154,7 +1150,7 @@ void OpDispatchBuilder::AVX128_Vector_CVT_Int_To_Float(OpcodeArgs, IR::OpSize Sr
|
||||
// unnecessarily zero extend the vector. Otherwise, if
|
||||
// memory, then we want to load the element size exactly.
|
||||
const auto LoadSize = IR::SizeToOpSize(8 * (IR::OpSizeToSize(Size) / 16));
|
||||
return RefPair {.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], LoadSize, Op->Flags)};
|
||||
return RefPair {.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], LoadSize, Op->Flags)};
|
||||
} else {
|
||||
return AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, !Is128Bit);
|
||||
}
|
||||
@@ -1304,7 +1300,7 @@ void OpDispatchBuilder::AVX128_InsertScalarRound(OpcodeArgs, IR::OpSize ElementS
|
||||
if (Op->Src[1].IsGPR()) {
|
||||
Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false);
|
||||
} else {
|
||||
Src2.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags);
|
||||
Src2.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags);
|
||||
}
|
||||
|
||||
// If OpSize == ElementSize then it only does the lower scalar op
|
||||
@@ -1573,20 +1569,20 @@ void OpDispatchBuilder::AVX128_VMASKMOVImpl(OpcodeArgs, IR::OpSize ElementSize,
|
||||
auto Address = MakeAddress(Op->Dest);
|
||||
|
||||
auto Data = AVX128_LoadSource_WithOpSize(Op, DataOp, Op->Flags, !Is128Bit);
|
||||
_VStoreVectorMasked(OpSize::i128Bit, ElementSize, Mask.Low, Data.Low, Address, Invalid(), MEM_OFFSET_SXTX, 1);
|
||||
_VStoreVectorMasked(OpSize::i128Bit, ElementSize, Mask.Low, Data.Low, Address, Invalid(), MemOffsetType::SXTX, 1);
|
||||
if (!Is128Bit) {
|
||||
_VStoreVectorMasked(OpSize::i128Bit, ElementSize, Mask.High, Data.High, Address, _InlineConstant(16), MEM_OFFSET_SXTX, 1);
|
||||
_VStoreVectorMasked(OpSize::i128Bit, ElementSize, Mask.High, Data.High, Address, _InlineConstant(16), MemOffsetType::SXTX, 1);
|
||||
}
|
||||
} else {
|
||||
auto Address = MakeAddress(DataOp);
|
||||
|
||||
RefPair Result {};
|
||||
Result.Low = _VLoadVectorMasked(OpSize::i128Bit, ElementSize, Mask.Low, Address, Invalid(), MEM_OFFSET_SXTX, 1);
|
||||
Result.Low = _VLoadVectorMasked(OpSize::i128Bit, ElementSize, Mask.Low, Address, Invalid(), MemOffsetType::SXTX, 1);
|
||||
|
||||
if (Is128Bit) {
|
||||
Result.High = LoadZeroVector(OpSize::i128Bit);
|
||||
} else {
|
||||
Result.High = _VLoadVectorMasked(OpSize::i128Bit, ElementSize, Mask.High, Address, _InlineConstant(16), MEM_OFFSET_SXTX, 1);
|
||||
Result.High = _VLoadVectorMasked(OpSize::i128Bit, ElementSize, Mask.High, Address, _InlineConstant(16), MemOffsetType::SXTX, 1);
|
||||
}
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
}
|
||||
@@ -1616,11 +1612,11 @@ void OpDispatchBuilder::AVX128_MASKMOV(OpcodeArgs) {
|
||||
// RDI source (DS prefix by default)
|
||||
auto MemDest = MakeSegmentAddress(X86State::REG_RDI, Op->Flags, X86Tables::DecodeFlags::FLAG_DS_PREFIX);
|
||||
|
||||
Ref XMMReg = _LoadMem(FPRClass, Size, MemDest, OpSize::i8Bit);
|
||||
Ref XMMReg = _LoadMemFPR(Size, MemDest, OpSize::i8Bit);
|
||||
|
||||
// If the Mask element high bit is set then overwrite the element with the source, else keep the memory variant
|
||||
XMMReg = _VBSL(Size, MaskSrc.Low, VectorSrc.Low, XMMReg);
|
||||
_StoreMem(FPRClass, Size, MemDest, XMMReg, OpSize::i8Bit);
|
||||
_StoreMemFPR(Size, MemDest, XMMReg, OpSize::i8Bit);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VectorVariableBlend(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
@@ -1660,7 +1656,7 @@ void OpDispatchBuilder::AVX128_SaveAVXState(Ref MemBase) {
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; i += 2) {
|
||||
RefPair Pair = LoadContextPair(OpSize::i128Bit, AVXHigh0Index + i);
|
||||
_StoreMemPair(FPRClass, OpSize::i128Bit, Pair.Low, Pair.High, MemBase, i * 16 + 576);
|
||||
_StoreMemPairFPR(OpSize::i128Bit, Pair.Low, Pair.High, MemBase, i * 16 + 576);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1668,7 +1664,7 @@ void OpDispatchBuilder::AVX128_RestoreAVXState(Ref MemBase) {
|
||||
const auto NumRegs = Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; i += 2) {
|
||||
auto YMMHRegs = LoadMemPair(FPRClass, OpSize::i128Bit, MemBase, i * 16 + 576);
|
||||
auto YMMHRegs = LoadMemPairFPR(OpSize::i128Bit, MemBase, i * 16 + 576);
|
||||
|
||||
AVX128_StoreXMMRegister(i, YMMHRegs.Low, true);
|
||||
AVX128_StoreXMMRegister(i + 1, YMMHRegs.High, true);
|
||||
@@ -1968,7 +1964,7 @@ void OpDispatchBuilder::AVX128_VFMAScalarImpl(OpcodeArgs, IROps IROp, uint8_t Sr
|
||||
if (Op->Src[1].IsGPR()) {
|
||||
Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false).Low;
|
||||
} else {
|
||||
Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags);
|
||||
Src2 = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags);
|
||||
}
|
||||
|
||||
Ref Sources[3] = {Dest, Src1, Src2};
|
||||
@@ -2016,8 +2012,8 @@ void OpDispatchBuilder::AVX128_VFMAddSubImpl(OpcodeArgs, bool AddSub, uint8_t Sr
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
}
|
||||
|
||||
OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherImpl(OpSize Size, OpSize ElementLoadSize, OpSize AddrElementSize, RefPair Dest,
|
||||
RefPair Mask, RefVSIB VSIB) {
|
||||
OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherImpl(OpcodeArgs, OpSize Size, OpSize ElementLoadSize, OpSize AddrElementSize,
|
||||
RefPair Dest, RefPair Mask, RefVSIB VSIB) {
|
||||
LOGMAN_THROW_A_FMT(AddrElementSize == OpSize::i32Bit || AddrElementSize == OpSize::i64Bit, "Unknown address element size");
|
||||
const auto Is128Bit = Size == OpSize::i128Bit;
|
||||
|
||||
@@ -2061,10 +2057,13 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherImpl(OpSize Size, O
|
||||
}
|
||||
}
|
||||
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
auto AddrSize = (Op->Flags & X86Tables::DecodeFlags::FLAG_ADDRESS_SIZE) != 0 ? (GPRSize >> 1) : GPRSize;
|
||||
|
||||
RefPair Result {};
|
||||
///< Calculate the low-half.
|
||||
Result.Low = _VLoadVectorGatherMasked(OpSize::i128Bit, ElementLoadSize, Dest.Low, Mask.Low, BaseAddr, VSIB.Low, VSIB.High,
|
||||
AddrElementSize, VSIB.Scale, 0, 0);
|
||||
AddrElementSize, VSIB.Scale, 0, 0, AddrSize);
|
||||
|
||||
if (Is128Bit) {
|
||||
Result.High = LoadZeroVector(OpSize::i128Bit);
|
||||
@@ -2101,7 +2100,7 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherImpl(OpSize Size, O
|
||||
|
||||
///< Calculate the high-half.
|
||||
auto ResultHigh = _VLoadVectorGatherMasked(OpSize::i128Bit, ElementLoadSize, DestReg, MaskReg, BaseAddr, AddrAddressing.Low,
|
||||
AddrAddressing.High, AddrElementSize, VSIB.Scale, DataElementOffset, IndexElementOffset);
|
||||
AddrAddressing.High, AddrElementSize, VSIB.Scale, DataElementOffset, IndexElementOffset, AddrSize);
|
||||
|
||||
if (AddrElementSize == OpSize::i64Bit && ElementLoadSize == OpSize::i32Bit) {
|
||||
// If we only fetched 128-bits worth of data then the upper-result is all zero.
|
||||
@@ -2114,7 +2113,7 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherImpl(OpSize Size, O
|
||||
return Result;
|
||||
}
|
||||
|
||||
OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherQPSImpl(Ref Dest, Ref Mask, RefVSIB VSIB) {
|
||||
OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherQPSImpl(OpcodeArgs, Ref Dest, Ref Mask, RefVSIB VSIB) {
|
||||
|
||||
///< BaseAddr doesn't need to exist, calculate that here.
|
||||
Ref BaseAddr = VSIB.BaseAddr;
|
||||
@@ -2142,8 +2141,11 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherQPSImpl(Ref Dest, R
|
||||
|
||||
RefPair Result {};
|
||||
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
auto AddrSize = (Op->Flags & X86Tables::DecodeFlags::FLAG_ADDRESS_SIZE) != 0 ? (GPRSize >> 1) : GPRSize;
|
||||
|
||||
///< Calculate the low-half.
|
||||
Result.Low = _VLoadVectorGatherMaskedQPS(OpSize::i128Bit, OpSize::i32Bit, Dest, Mask, BaseAddr, VSIB.Low, VSIB.High, VSIB.Scale);
|
||||
Result.Low = _VLoadVectorGatherMaskedQPS(OpSize::i128Bit, OpSize::i32Bit, Dest, Mask, BaseAddr, VSIB.Low, VSIB.High, VSIB.Scale, AddrSize);
|
||||
Result.High = LoadZeroVector(OpSize::i128Bit);
|
||||
if (VSIB.High == Invalid()) {
|
||||
// Special case for only loading two floats.
|
||||
@@ -2202,15 +2204,15 @@ void OpDispatchBuilder::AVX128_VPGATHER(OpcodeArgs, OpSize AddrElementSize) {
|
||||
}
|
||||
|
||||
///< AddressElementSize is now OpSize::i64Bit
|
||||
Result = AVX128_VPGatherQPSImpl(Dest.Low, Mask.Low, VSIBLow);
|
||||
Result = AVX128_VPGatherQPSImpl(Op, Dest.Low, Mask.Low, VSIBLow);
|
||||
if (NeedsHighAddrBytes) {
|
||||
auto Res = AVX128_VPGatherQPSImpl(Dest.High, Mask.High, VSIBHigh);
|
||||
auto Res = AVX128_VPGatherQPSImpl(Op, Dest.High, Mask.High, VSIBHigh);
|
||||
Result.High = Res.Low;
|
||||
}
|
||||
} else if (AddrElementSize == OpSize::i64Bit && ElementLoadSize == OpSize::i32Bit) {
|
||||
Result = AVX128_VPGatherQPSImpl(Dest.Low, Mask.Low, VSIB);
|
||||
Result = AVX128_VPGatherQPSImpl(Op, Dest.Low, Mask.Low, VSIB);
|
||||
} else {
|
||||
Result = AVX128_VPGatherImpl(Size, ElementLoadSize, AddrElementSize, Dest, Mask, VSIB);
|
||||
Result = AVX128_VPGatherImpl(Op, Size, ElementLoadSize, AddrElementSize, Dest, Mask, VSIB);
|
||||
}
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
|
||||
@@ -2234,7 +2236,7 @@ void OpDispatchBuilder::AVX128_VCVTPH2PS(OpcodeArgs) {
|
||||
// In the event that a memory operand is used as the source operand,
|
||||
// the access width will always be half the size of the destination vector width
|
||||
// (i.e. 128-bit vector -> 64-bit mem, 256-bit vector -> 128-bit mem)
|
||||
Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
}
|
||||
|
||||
RefPair Result {};
|
||||
@@ -2289,7 +2291,7 @@ void OpDispatchBuilder::AVX128_VCVTPS2PH(OpcodeArgs) {
|
||||
}
|
||||
|
||||
if (!Op->Dest.IsGPR()) {
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result.Low, StoreSize, OpSize::iInvalid);
|
||||
StoreResultFPR_WithOpSize(Op, Op->Dest, Result.Low, StoreSize);
|
||||
} else {
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
}
|
||||
|
||||
@@ -53,7 +53,7 @@ constexpr inline DispatchTableEntry OpDispatch_BaseOpTable[] = {
|
||||
{0xAA, 2, &OpDispatchBuilder::STOSOp},
|
||||
{0xAC, 2, &OpDispatchBuilder::LODSOp},
|
||||
{0xAE, 2, &OpDispatchBuilder::SCASOp},
|
||||
{0xB0, 16, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVGPROp, 0>},
|
||||
{0xB0, 16, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVGPRImmediate>},
|
||||
{0xC2, 2, &OpDispatchBuilder::RETOp},
|
||||
{0xC8, 1, &OpDispatchBuilder::EnterOp},
|
||||
{0xC9, 1, &OpDispatchBuilder::LEAVEOp},
|
||||
|
||||
@@ -23,8 +23,8 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
// ARMv8 SHA1 extension provides a `SHA1H` instruction which does a fixed rotate by 30.
|
||||
// This only operates on element 0 rather than element 3. We don't have the luxury of rewriting the x86 SHA algorithm to take advantage of this.
|
||||
@@ -36,7 +36,7 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
|
||||
auto Tmp = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src, RotatedNode);
|
||||
auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 3, 3, Src, Tmp);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
|
||||
@@ -44,15 +44,15 @@ void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
Ref NewVec = _VExtr(OpSize::i128Bit, OpSize::i64Bit, Dest, Src, 1);
|
||||
|
||||
// [W0, W1, W2, W3] ^ [W2, W3, W4, W5]
|
||||
Ref Result = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, NewVec);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
@@ -60,8 +60,8 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
// ARM SHA1 mostly matches x86 semantics, except the input and outputs are both flipped from elements 0,1,2,3 to 3,2,1,0.
|
||||
auto Src1 = SHADataShuffle(Dest);
|
||||
@@ -70,7 +70,7 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
// The result is swizzled differently than expected
|
||||
auto Result = SHADataShuffle(_VSha1SU1(Src1, Src2));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
@@ -79,8 +79,8 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
return;
|
||||
}
|
||||
const uint64_t Imm8 = Op->Src[1].Literal() & 0b11;
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
Ref Result {};
|
||||
Ref ConstantVector {};
|
||||
@@ -112,7 +112,7 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
case 3: Result = SHADataShuffle(_VSha1P(Src1, ZeroRegister, Src2)); break;
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
|
||||
@@ -120,12 +120,12 @@ void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto Result = _VSha256U0(Dest, Src);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
@@ -133,8 +133,8 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto Src1 = _VExtr(OpSize::i128Bit, OpSize::i32Bit, Dest, Dest, 3);
|
||||
auto DupDst = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
@@ -142,7 +142,7 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
|
||||
auto Result = _VSha256U1(Src1, Src2);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
@@ -150,8 +150,8 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
// Hardcoded to XMM0
|
||||
auto XMM0 = LoadXMMRegister(0);
|
||||
|
||||
@@ -177,7 +177,7 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
auto B = _VSha256H2(EFGH, ABCD, Key);
|
||||
auto Result = shuffle_abcd(A, B);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
|
||||
@@ -185,9 +185,9 @@ void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESImc(Src);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
|
||||
@@ -195,10 +195,10 @@ void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESEnc(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
|
||||
@@ -208,11 +208,11 @@ void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESENC.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENC unimplemented");
|
||||
|
||||
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
Ref Result = _VAESEnc(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
|
||||
@@ -220,10 +220,10 @@ void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESEncLast(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
|
||||
@@ -233,11 +233,11 @@ void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESENCLAST.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENCLAST unimplemented");
|
||||
|
||||
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
Ref Result = _VAESEncLast(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
|
||||
@@ -245,10 +245,10 @@ void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESDec(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
|
||||
@@ -258,11 +258,11 @@ void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESDEC.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDEC unimplemented");
|
||||
|
||||
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
Ref Result = _VAESDec(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
|
||||
@@ -270,10 +270,10 @@ void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESDecLast(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
|
||||
@@ -283,15 +283,15 @@ void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESDECLAST.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDECLAST unimplemented");
|
||||
|
||||
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
Ref Result = _VAESDecLast(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) {
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
const uint64_t RCON = Op->Src[1].Literal();
|
||||
|
||||
auto KeyGenSwizzle = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE);
|
||||
@@ -305,7 +305,7 @@ void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) {
|
||||
}
|
||||
|
||||
Ref Result = AESKeyGenAssistImpl(Op);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
|
||||
@@ -313,12 +313,12 @@ void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[1].Literal());
|
||||
|
||||
auto Res = _PCLMUL(OpSize::i128Bit, Dest, Src, Selector & 0b1'0001);
|
||||
StoreResult(FPRClass, Op, Res, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Res);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) {
|
||||
@@ -328,12 +328,12 @@ void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) {
|
||||
}
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
|
||||
Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[2].Literal());
|
||||
|
||||
Ref Res = _PCLMUL(DstSize, Src1, Src2, Selector & 0b1'0001);
|
||||
StoreResult(FPRClass, Op, Res, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Res);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -263,7 +263,7 @@ void OpDispatchBuilder::CalculateDeferredFlags() {
|
||||
Ref OpDispatchBuilder::IncrementByCarry(OpSize OpSize, Ref Src) {
|
||||
// If CF not inverted, we use .cc since the increment happens when the
|
||||
// condition is false. If CF inverted, invert to use .cs. A bit mindbendy.
|
||||
return _NZCVSelectIncrement(OpSize, {CFInverted ? COND_UGE : COND_ULT}, Src, Src);
|
||||
return _NZCVSelectIncrement(OpSize, CFInverted ? CondClass::UGE : CondClass::ULT, Src, Src);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::CalculateFlags_ADC(IR::OpSize SrcSize, Ref Src1, Ref Src2) {
|
||||
@@ -290,7 +290,7 @@ Ref OpDispatchBuilder::CalculateFlags_ADC(IR::OpSize SrcSize, Ref Src1, Ref Src2
|
||||
Res = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Res);
|
||||
|
||||
// TODO: We can fold that second Bfe in (cmp uxth).
|
||||
auto SelectCFInv = Select01(OpSize, CondClassType {COND_UGE}, Res, Src2PlusCF);
|
||||
auto SelectCFInv = Select01(OpSize, CondClass::UGE, Res, Src2PlusCF);
|
||||
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
SetCFInverted(SelectCFInv);
|
||||
@@ -324,7 +324,7 @@ Ref OpDispatchBuilder::CalculateFlags_SBB(IR::OpSize SrcSize, Ref Src1, Ref Src2
|
||||
Res = Sub(OpSize, Src1, Src2PlusCF);
|
||||
Res = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Res);
|
||||
|
||||
auto SelectCFInv = Select01(OpSize, CondClassType {COND_UGE}, Src1, Src2PlusCF);
|
||||
auto SelectCFInv = Select01(OpSize, CondClass::UGE, Src1, Src2PlusCF);
|
||||
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
SetCFInverted(SelectCFInv);
|
||||
@@ -406,7 +406,7 @@ void OpDispatchBuilder::CalculateFlags_MUL(IR::OpSize SrcSize, Ref Res, Ref High
|
||||
// If High = SignBit, then sets to nZCv. Else sets to nzcV. Since SF/ZF
|
||||
// undefined, this does what we need after inverting carry.
|
||||
auto Zero = _InlineConstant(0);
|
||||
_CondSubNZCV(OpSize::i64Bit, Zero, Zero, CondClassType {COND_EQ}, 0x1 /* nzcV */);
|
||||
_CondSubNZCV(OpSize::i64Bit, Zero, Zero, CondClass::EQ, 0x1 /* nzcV */);
|
||||
CFInverted = true;
|
||||
}
|
||||
|
||||
@@ -423,7 +423,7 @@ void OpDispatchBuilder::CalculateFlags_UMUL(Ref High) {
|
||||
|
||||
// If High = 0, then sets to nZCv. Else sets to nzcV. Since SF/ZF undefined,
|
||||
// this does what we need.
|
||||
_CondSubNZCV(Size, Zero, Zero, CondClassType {COND_EQ}, 0x1 /* nzcV */);
|
||||
_CondSubNZCV(Size, Zero, Zero, CondClass::EQ, 0x1 /* nzcV */);
|
||||
CFInverted = true;
|
||||
}
|
||||
|
||||
|
||||
@@ -151,6 +151,9 @@ constexpr DispatchTableEntry OpDispatch_SecondaryGroupTables[] = {
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F2, 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 3>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F2, 4), 4, &OpDispatchBuilder::NOPOp},
|
||||
|
||||
// GROUP 17
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_17, PF_66, 0), 1, &OpDispatchBuilder::Extrq_imm},
|
||||
|
||||
// GROUP P
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_NONE, 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_NONE, 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, true, false, 1>},
|
||||
|
||||
@@ -198,6 +198,8 @@ constexpr DispatchTableEntry OpDispatch_SecondaryRepNEModTables[] = {
|
||||
{0x5E, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x5F, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x70, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSHUFWOp, true>},
|
||||
{0x78, 1, &OpDispatchBuilder::Insertq_imm},
|
||||
{0x79, 1, &OpDispatchBuilder::Insertq},
|
||||
{0x7C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, OpSize::i32Bit>},
|
||||
{0x7D, 1, &OpDispatchBuilder::HSUBP<OpSize::i32Bit>},
|
||||
{0xD0, 1, &OpDispatchBuilder::ADDSUBPOp<OpSize::i32Bit>},
|
||||
@@ -256,6 +258,7 @@ constexpr DispatchTableEntry OpDispatch_SecondaryOpSizeModTables[] = {
|
||||
{0x75, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i16Bit>},
|
||||
{0x76, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i32Bit>},
|
||||
{0x78, 1, nullptr}, // GROUP 17
|
||||
{0x79, 1, &OpDispatchBuilder::Extrq},
|
||||
{0x7C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, OpSize::i64Bit>},
|
||||
{0x7D, 1, &OpDispatchBuilder::HSUBP<OpSize::i64Bit>},
|
||||
{0x7E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVBetweenGPR_FPR, OpDispatchBuilder::VectorOpType::SSE>},
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -28,10 +28,12 @@ class OrderedNode;
|
||||
Ref OpDispatchBuilder::GetX87Top() {
|
||||
// Yes, we are storing 3 bits in a single flag register.
|
||||
// Deal with it
|
||||
return _LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
return _LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SetX87FTW(Ref FTW) {
|
||||
_StackForceSlow(); // Invalidate x87 FTW register cache
|
||||
|
||||
// For the output, we want a 1-bit for each pair not equal to 11 (Empty).
|
||||
static_assert(static_cast<uint8_t>(FPState::X87Tag::Empty) == 0b11);
|
||||
|
||||
@@ -50,18 +52,18 @@ void OpDispatchBuilder::SetX87FTW(Ref FTW) {
|
||||
FTW = _Orlshr(OpSize::i32Bit, FTW, FTW, 4);
|
||||
|
||||
// ...and that's it. StoreContext implicitly does the final masking.
|
||||
StoreContext(AbridgedFTWIndex, FTW);
|
||||
_StoreContextGPR(OpSize::i8Bit, FTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SetX87Top(Ref Value) {
|
||||
_StoreContext(OpSize::i8Bit, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
_StoreContextGPR(OpSize::i8Bit, Value, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
}
|
||||
|
||||
// Float LoaD operation with memory operand
|
||||
void OpDispatchBuilder::FLD(OpcodeArgs, IR::OpSize Width) {
|
||||
const auto ReadWidth = (Width == OpSize::f80Bit) ? OpSize::i128Bit : Width;
|
||||
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], Width, Op->Flags);
|
||||
Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], Width, Op->Flags);
|
||||
Ref ConvertedData = Data;
|
||||
// Convert to 80bit float
|
||||
if (Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
@@ -77,14 +79,14 @@ void OpDispatchBuilder::FLDFromStack(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::FBLD(OpcodeArgs) {
|
||||
// Read from memory
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::f80Bit, Op->Flags);
|
||||
Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::f80Bit, Op->Flags);
|
||||
Ref ConvertedData = _F80BCDLoad(Data);
|
||||
_PushStack(ConvertedData, Data, OpSize::i128Bit, true);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FBSTP(OpcodeArgs) {
|
||||
Ref converted = _F80BCDStore(_ReadStackValue(0));
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, converted, OpSize::f80Bit, OpSize::i8Bit);
|
||||
StoreResultFPR_WithOpSize(Op, Op->Dest, converted, OpSize::f80Bit, OpSize::i8Bit);
|
||||
_PopStackDestroy();
|
||||
}
|
||||
|
||||
@@ -97,7 +99,7 @@ void OpDispatchBuilder::FLD_Const(OpcodeArgs, NamedVectorConstant K) {
|
||||
void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
const auto ReadWidth = OpSizeFromSrc(Op);
|
||||
// Read from memory
|
||||
Ref Data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], ReadWidth, Op->Flags);
|
||||
Ref Data = LoadSourceGPR_WithOpSize(Op, Op->Src[0], ReadWidth, Op->Flags);
|
||||
|
||||
// Sign extend to 64bits
|
||||
if (ReadWidth != OpSize::i64Bit) {
|
||||
@@ -110,15 +112,15 @@ void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
// Extract sign and make integer absolute
|
||||
auto zero = Constant(0);
|
||||
_SubNZCV(OpSize::i64Bit, Data, zero);
|
||||
auto sign = _NZCVSelect(OpSize::i64Bit, CondClassType {COND_SLT}, Constant(0x8000), zero);
|
||||
auto absolute = _Neg(OpSize::i64Bit, Data, CondClassType {COND_MI});
|
||||
auto sign = _NZCVSelect(OpSize::i64Bit, CondClass::SLT, Constant(0x8000), zero);
|
||||
auto absolute = _Neg(OpSize::i64Bit, Data, CondClass::MI);
|
||||
|
||||
// left justify the absolute integer
|
||||
auto shift = Sub(OpSize::i64Bit, Constant(63), _FindMSB(IR::OpSize::i64Bit, absolute));
|
||||
auto shifted = _Lshl(OpSize::i64Bit, absolute, shift);
|
||||
|
||||
auto adjusted_exponent = Sub(OpSize::i64Bit, Constant(0x3fff + 63), shift);
|
||||
auto zeroed_exponent = _Select(COND_EQ, absolute, zero, zero, adjusted_exponent);
|
||||
auto zeroed_exponent = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, absolute, zero, zero, adjusted_exponent);
|
||||
auto upper = _Or(OpSize::i64Bit, sign, zeroed_exponent);
|
||||
|
||||
Ref ConvertedData = _VLoadTwoGPRs(shifted, upper);
|
||||
@@ -164,12 +166,12 @@ void OpDispatchBuilder::FIST(OpcodeArgs, bool Truncate) {
|
||||
// Check for NaN/Infinity: exponent = 0x7fff
|
||||
SaveNZCV();
|
||||
_TestNZ(OpSize::i64Bit, Exponent, Constant(0x7fff));
|
||||
Ref IsSpecial = _NZCVSelect01({COND_EQ});
|
||||
Ref IsSpecial = _NZCVSelect01(CondClass::EQ);
|
||||
|
||||
// For overflow detection, check if exponent indicates a value >= 2^15
|
||||
// Biased exponent for 2^15 is 0x3fff + 15 = 0x400e
|
||||
SubWithFlags(OpSize::i64Bit, Exponent, 0x400e);
|
||||
Ref IsOverflow = _NZCVSelect01({COND_UGE});
|
||||
Ref IsOverflow = _NZCVSelect01(CondClass::UGE);
|
||||
|
||||
// Set Invalid Operation flag if overflow or special value
|
||||
Ref InvalidFlag = _Or(OpSize::i64Bit, IsSpecial, IsOverflow);
|
||||
@@ -178,7 +180,7 @@ void OpDispatchBuilder::FIST(OpcodeArgs, bool Truncate) {
|
||||
|
||||
Data = _F80CVTInt(Size, Data, Truncate);
|
||||
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Data, Size, OpSize::i8Bit);
|
||||
StoreResultGPR_WithOpSize(Op, Op->Dest, Data, Size, OpSize::i8Bit);
|
||||
|
||||
if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) {
|
||||
_PopStackDestroy();
|
||||
@@ -204,10 +206,10 @@ void OpDispatchBuilder::FADD(OpcodeArgs, IR::OpSize Width, bool Integer, OpDispa
|
||||
// We have one memory argument
|
||||
Ref Arg {};
|
||||
if (Integer) {
|
||||
Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTToInt(Arg, Width);
|
||||
} else {
|
||||
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTTo(Arg, Width);
|
||||
}
|
||||
|
||||
@@ -234,10 +236,10 @@ void OpDispatchBuilder::FMUL(OpcodeArgs, IR::OpSize Width, bool Integer, OpDispa
|
||||
// We have one memory argument
|
||||
Ref arg {};
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTToInt(arg, Width);
|
||||
} else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTTo(arg, Width);
|
||||
}
|
||||
|
||||
@@ -271,10 +273,10 @@ void OpDispatchBuilder::FDIV(OpcodeArgs, IR::OpSize Width, bool Integer, bool Re
|
||||
// We have one memory argument
|
||||
Ref arg {};
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTToInt(arg, Width);
|
||||
} else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTTo(arg, Width);
|
||||
}
|
||||
|
||||
@@ -312,10 +314,10 @@ void OpDispatchBuilder::FSUB(OpcodeArgs, IR::OpSize Width, bool Integer, bool Re
|
||||
// We have one memory argument
|
||||
Ref Arg {};
|
||||
if (Integer) {
|
||||
Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTToInt(Arg, Width);
|
||||
} else {
|
||||
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTTo(Arg, Width);
|
||||
}
|
||||
|
||||
@@ -338,7 +340,7 @@ Ref OpDispatchBuilder::GetX87FTW_Helper() {
|
||||
// bytes, we use the well-known bit twiddling algorithm:
|
||||
//
|
||||
// https://graphics.stanford.edu/~seander/bithacks.html#InterleaveBMN
|
||||
Ref X = LoadContext(AbridgedFTWIndex);
|
||||
Ref X = _LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
X = _Orlshl(OpSize::i32Bit, X, X, 4);
|
||||
X = _And(OpSize::i32Bit, X, Constant(0x0f0f0f0f));
|
||||
X = _Orlshl(OpSize::i32Bit, X, X, 2);
|
||||
@@ -379,41 +381,41 @@ void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) {
|
||||
_SyncStackToSlow();
|
||||
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
Ref Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
Ref Mem = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
{
|
||||
auto FCW = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreMem(GPRClass, Size, Mem, FCW, Size);
|
||||
auto FCW = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreMemGPR(Size, Mem, FCW, Size);
|
||||
}
|
||||
|
||||
{ _StoreMem(GPRClass, Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MEM_OFFSET_SXTX, 1); }
|
||||
{ _StoreMemGPR(Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1); }
|
||||
|
||||
auto ZeroConst = Constant(0);
|
||||
|
||||
{
|
||||
// FTW
|
||||
_StoreMem(GPRClass, Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMemGPR(Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction Offset
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction CS selector (+ Opcode)
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer offset
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer selector
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -439,20 +441,20 @@ void OpDispatchBuilder::X87LDENV(OpcodeArgs) {
|
||||
_StackForceSlow();
|
||||
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
Ref Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Ref Mem = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, Mem, OpSize::i16Bit);
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
auto NewFCW = _LoadMemGPR(OpSize::i16Bit, Mem, OpSize::i16Bit);
|
||||
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
Ref MemLocation = Add(OpSize::i64Bit, Mem, IR::OpSizeToSize(Size) * 1);
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, MemLocation, Size);
|
||||
auto NewFSW = _LoadMemGPR(Size, MemLocation, Size);
|
||||
ReconstructX87StateFromFSW_Helper(NewFSW);
|
||||
|
||||
{
|
||||
// FTW
|
||||
Ref MemLocation = Add(OpSize::i64Bit, Mem, IR::OpSizeToSize(Size) * 2);
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, MemLocation, Size));
|
||||
SetX87FTW(_LoadMemGPR(Size, MemLocation, Size));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -481,61 +483,61 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
Ref Mem = MakeSegmentAddress(Op, Op->Dest);
|
||||
Ref Top = GetX87Top();
|
||||
{
|
||||
auto FCW = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreMem(GPRClass, Size, Mem, FCW, Size);
|
||||
auto FCW = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreMemGPR(Size, Mem, FCW, Size);
|
||||
}
|
||||
|
||||
{ _StoreMem(GPRClass, Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MEM_OFFSET_SXTX, 1); }
|
||||
{ _StoreMemGPR(Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1); }
|
||||
|
||||
auto ZeroConst = Constant(0);
|
||||
|
||||
{
|
||||
// FTW
|
||||
_StoreMem(GPRClass, Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMemGPR(Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction Offset
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction CS selector (+ Opcode)
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer offset
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer selector
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
auto SevenConst = Constant(7);
|
||||
const auto LoadSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
for (int i = 0; i < 7; ++i) {
|
||||
Ref data = _LoadContextIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
|
||||
Ref data = _LoadContextFPRIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit));
|
||||
if (ReducedPrecisionMode) {
|
||||
data = _F80CVTTo(data, OpSize::i64Bit);
|
||||
}
|
||||
_StoreMem(FPRClass, OpSize::i128Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMemFPR(OpSize::i128Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
Top = _And(OpSize::i32Bit, Add(OpSize::i32Bit, Top, 1), SevenConst);
|
||||
}
|
||||
|
||||
// The final st(7) needs a bit of special handling here
|
||||
Ref data = _LoadContextIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
|
||||
Ref data = _LoadContextFPRIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit));
|
||||
if (ReducedPrecisionMode) {
|
||||
data = _F80CVTTo(data, OpSize::i64Bit);
|
||||
}
|
||||
// ST7 broken in to two parts
|
||||
// Lower 64bits [63:0]
|
||||
// upper 16 bits [79:64]
|
||||
_StoreMem(FPRClass, OpSize::i64Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMemFPR(OpSize::i64Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
auto topBytes = _VDupElement(OpSize::i128Bit, OpSize::i16Bit, data, 4);
|
||||
_StoreMem(FPRClass, OpSize::i16Bit, topBytes, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10) + 8), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMemFPR(OpSize::i16Bit, topBytes, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10) + 8), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
|
||||
// reset to default
|
||||
FNINIT(Op);
|
||||
@@ -546,8 +548,8 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
Ref Mem = MakeSegmentAddress(Op, Op->Src[0]);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, Mem, OpSize::i16Bit);
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
auto NewFCW = _LoadMemGPR(OpSize::i16Bit, Mem, OpSize::i16Bit);
|
||||
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
if (ReducedPrecisionMode) {
|
||||
// ignore the rounding precision, we're always 64-bit in F64.
|
||||
// extract rounding mode
|
||||
@@ -559,11 +561,11 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
_SetRoundingMode(roundingMode, false, roundingMode);
|
||||
}
|
||||
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MEM_OFFSET_SXTX, 1);
|
||||
auto NewFSW = _LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1);
|
||||
Ref Top = ReconstructX87StateFromFSW_Helper(NewFSW);
|
||||
{
|
||||
// FTW
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MEM_OFFSET_SXTX, 1));
|
||||
SetX87FTW(_LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1));
|
||||
}
|
||||
|
||||
auto SevenConst = Constant(7);
|
||||
@@ -572,14 +574,14 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
Ref Mask = _VLoadTwoGPRs(low, high);
|
||||
const auto StoreSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
for (int i = 0; i < 7; ++i) {
|
||||
Ref Reg = _LoadMem(FPRClass, OpSize::i128Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
Ref Reg = _LoadMemFPR(OpSize::i128Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
// Mask off the top bits
|
||||
Reg = _VAnd(OpSize::i128Bit, OpSize::i128Bit, Reg, Mask);
|
||||
if (ReducedPrecisionMode) {
|
||||
// Convert to double precision
|
||||
Reg = _F80CVT(OpSize::i64Bit, Reg);
|
||||
}
|
||||
_StoreContextIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
|
||||
_StoreContextFPRIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit));
|
||||
|
||||
Top = _And(OpSize::i32Bit, Add(OpSize::i32Bit, Top, 1), SevenConst);
|
||||
}
|
||||
@@ -588,20 +590,19 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
// ST7 broken in to two parts
|
||||
// Lower 64bits [63:0]
|
||||
// upper 16 bits [79:64]
|
||||
Ref Reg = _LoadMem(FPRClass, OpSize::i64Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
Ref RegHigh =
|
||||
_LoadMem(FPRClass, OpSize::i16Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7) + 8), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
Ref Reg = _LoadMemFPR(OpSize::i64Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
Ref RegHigh = _LoadMemFPR(OpSize::i16Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7) + 8), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
Reg = _VInsElement(OpSize::i128Bit, OpSize::i16Bit, 4, 0, Reg, RegHigh);
|
||||
if (ReducedPrecisionMode) {
|
||||
Reg = _F80CVT(OpSize::i64Bit, Reg); // Convert to double precision
|
||||
}
|
||||
_StoreContextIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
|
||||
_StoreContextFPRIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit));
|
||||
}
|
||||
|
||||
// Load / Store Control Word
|
||||
void OpDispatchBuilder::X87FSTCW(OpcodeArgs) {
|
||||
auto FCW = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
StoreResult(GPRClass, Op, FCW, OpSize::iInvalid);
|
||||
auto FCW = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
StoreResultGPR(Op, FCW);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FLDCW(OpcodeArgs) {
|
||||
@@ -609,8 +610,8 @@ void OpDispatchBuilder::X87FLDCW(OpcodeArgs) {
|
||||
// to switch for now to slow mode whenever these are manually changed.
|
||||
// Remove the next line and try DF_04.asm in fast path.
|
||||
_StackForceSlow();
|
||||
Ref NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
Ref NewFCW = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FXCH(OpcodeArgs) {
|
||||
@@ -646,10 +647,10 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs, IR::OpSize Width, bool Integer, OpDisp
|
||||
if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
// Memory arg
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTToInt(arg, Width);
|
||||
} else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTTo(arg, Width);
|
||||
}
|
||||
} else {
|
||||
@@ -765,7 +766,7 @@ Ref OpDispatchBuilder::ReconstructFSW_Helper(Ref T) {
|
||||
void OpDispatchBuilder::X87FNSTSW(OpcodeArgs) {
|
||||
Ref TopValue = _SyncStackToSlow();
|
||||
Ref StatusWord = ReconstructFSW_Helper(TopValue);
|
||||
StoreResult(GPRClass, Op, StatusWord, OpSize::iInvalid);
|
||||
StoreResultGPR(Op, StatusWord);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FNCLEX(OpcodeArgs) {
|
||||
@@ -774,6 +775,8 @@ void OpDispatchBuilder::FNCLEX(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FNINIT(OpcodeArgs) {
|
||||
_SyncStackToSlow(); // Invalidate x87 register caches
|
||||
|
||||
auto Zero = Constant(0);
|
||||
|
||||
if (ReducedPrecisionMode) {
|
||||
@@ -782,12 +785,12 @@ void OpDispatchBuilder::FNINIT(OpcodeArgs) {
|
||||
|
||||
// Init FCW to 0x037F
|
||||
auto NewFCW = Constant(0x037F);
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
// Set top to zero
|
||||
SetX87Top(Zero);
|
||||
// Tags all get marked as invalid
|
||||
StoreContext(AbridgedFTWIndex, Zero);
|
||||
_StoreContextGPR(OpSize::i8Bit, Zero, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
|
||||
// Reinits the simulated stack
|
||||
_InitStack();
|
||||
@@ -860,7 +863,7 @@ void OpDispatchBuilder::X87FXAM(OpcodeArgs) {
|
||||
auto TopValid = _StackValidTag(0);
|
||||
|
||||
// In the case of top being invalid then C3:C2:C0 is 0b101
|
||||
auto C3 = Select01(OpSize::i32Bit, CondClassType {COND_NEQ}, TopValid, Constant(1));
|
||||
auto C3 = Select01(OpSize::i32Bit, CondClass::NEQ, TopValid, Constant(1));
|
||||
|
||||
auto C2 = TopValid;
|
||||
auto C0 = C3; // Mirror C3 until something other than zero is supported
|
||||
|
||||
@@ -29,38 +29,38 @@ void OpDispatchBuilder::X87LDENVF64(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
Ref Mem = MakeSegmentAddress(Op, Op->Src[0]);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, Mem, OpSize::i16Bit);
|
||||
auto NewFCW = _LoadMemGPR(OpSize::i16Bit, Mem, OpSize::i16Bit);
|
||||
// ignore the rounding precision, we're always 64-bit in F64.
|
||||
// extract rounding mode
|
||||
Ref roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
|
||||
_SetRoundingMode(roundingMode, false, roundingMode);
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, Mem, Constant(IR::OpSizeToSize(Size)), Size, MEM_OFFSET_SXTX, 1);
|
||||
auto NewFSW = _LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size)), Size, MemOffsetType::SXTX, 1);
|
||||
ReconstructX87StateFromFSW_Helper(NewFSW);
|
||||
|
||||
{
|
||||
// FTW
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MEM_OFFSET_SXTX, 1));
|
||||
SetX87FTW(_LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1));
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FLDCWF64(OpcodeArgs) {
|
||||
_StackForceSlow();
|
||||
|
||||
Ref NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref NewFCW = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
// ignore the rounding precision, we're always 64-bit in F64.
|
||||
// extract rounding mode
|
||||
Ref roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
|
||||
_SetRoundingMode(roundingMode, false, roundingMode);
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
}
|
||||
|
||||
// F64 ops
|
||||
// Float load op with memory operand
|
||||
void OpDispatchBuilder::FLDF64(OpcodeArgs, IR::OpSize Width) {
|
||||
const auto ReadWidth = (Width == OpSize::f80Bit) ? OpSize::i128Bit : Width;
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], ReadWidth, Op->Flags);
|
||||
Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], Width, Op->Flags);
|
||||
// Convert to 64bit float
|
||||
Ref ConvertedData = Data;
|
||||
if (Width == OpSize::i32Bit) {
|
||||
@@ -73,7 +73,7 @@ void OpDispatchBuilder::FLDF64(OpcodeArgs, IR::OpSize Width) {
|
||||
|
||||
void OpDispatchBuilder::FBLDF64(OpcodeArgs) {
|
||||
// Read from memory
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::i128Bit, Op->Flags);
|
||||
Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::f80Bit, Op->Flags);
|
||||
Ref ConvertedData = _F80BCDLoad(Data);
|
||||
ConvertedData = _F80CVT(OpSize::i64Bit, ConvertedData);
|
||||
_PushStack(ConvertedData, Data, OpSize::i64Bit, true);
|
||||
@@ -82,7 +82,7 @@ void OpDispatchBuilder::FBLDF64(OpcodeArgs) {
|
||||
void OpDispatchBuilder::FBSTPF64(OpcodeArgs) {
|
||||
Ref converted = _F80CVTTo(_ReadStackValue(0), OpSize::i64Bit);
|
||||
converted = _F80BCDStore(converted);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, converted, OpSize::f80Bit, OpSize::i8Bit);
|
||||
StoreResultFPR_WithOpSize(Op, Op->Dest, converted, OpSize::f80Bit, OpSize::i8Bit);
|
||||
_PopStackDestroy();
|
||||
}
|
||||
|
||||
@@ -95,7 +95,7 @@ void OpDispatchBuilder::FILDF64(OpcodeArgs) {
|
||||
const auto ReadWidth = OpSizeFromSrc(Op);
|
||||
|
||||
// Read from memory
|
||||
Ref Data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], ReadWidth, Op->Flags);
|
||||
Ref Data = LoadSourceGPR_WithOpSize(Op, Op->Src[0], ReadWidth, Op->Flags);
|
||||
if (ReadWidth == OpSize::i16Bit) {
|
||||
Data = _Sbfe(OpSize::i64Bit, IR::OpSizeAsBits(ReadWidth), 0, Data);
|
||||
}
|
||||
@@ -112,7 +112,7 @@ void OpDispatchBuilder::FISTF64(OpcodeArgs, bool Truncate) {
|
||||
} else {
|
||||
data = _Float_ToGPR_S(Size == OpSize::i32Bit ? OpSize::i32Bit : OpSize::i64Bit, OpSize::i64Bit, data);
|
||||
}
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, data, Size, OpSize::i8Bit);
|
||||
StoreResultGPR_WithOpSize(Op, Op->Dest, data, Size, OpSize::i8Bit);
|
||||
|
||||
if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) {
|
||||
_PopStackDestroy();
|
||||
@@ -138,16 +138,16 @@ void OpDispatchBuilder::FADDF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpDi
|
||||
Ref arg {};
|
||||
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
if (Width == OpSize::i16Bit) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
@@ -176,16 +176,16 @@ void OpDispatchBuilder::FMULF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpDi
|
||||
Ref arg {};
|
||||
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
if (Width == OpSize::i16Bit) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
@@ -228,16 +228,16 @@ void OpDispatchBuilder::FDIVF64(OpcodeArgs, IR::OpSize Width, bool Integer, bool
|
||||
|
||||
if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
if (Integer) {
|
||||
Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
if (Width == OpSize::i16Bit) {
|
||||
Arg = _Sbfe(OpSize::i64Bit, 16, 0, Arg);
|
||||
}
|
||||
Arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, Arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, Arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
@@ -285,16 +285,16 @@ void OpDispatchBuilder::FSUBF64(OpcodeArgs, IR::OpSize Width, bool Integer, bool
|
||||
|
||||
if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
if (Width == OpSize::i16Bit) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
@@ -332,16 +332,16 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpD
|
||||
} else if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
// Memory arg
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
if (Width == OpSize::i16Bit) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
b = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
@@ -393,8 +393,8 @@ void OpDispatchBuilder::X87FXTRACTF64(OpcodeArgs) {
|
||||
SaveNZCV();
|
||||
_TestNZ(OpSize::i64Bit, Gpr, Constant(0x7fff'ffff'ffff'ffffUL));
|
||||
|
||||
Ref Sig = _NZCVSelectV(OpSize::i64Bit, {COND_EQ}, SigZV, SigNZV);
|
||||
Ref Exp = _NZCVSelectV(OpSize::i64Bit, {COND_EQ}, ExpZV, ExpNZV);
|
||||
Ref Sig = _NZCVSelectV(OpSize::i64Bit, CondClass::EQ, SigZV, SigNZV);
|
||||
Ref Exp = _NZCVSelectV(OpSize::i64Bit, CondClass::EQ, ExpZV, ExpNZV);
|
||||
|
||||
_PopStackDestroy();
|
||||
_PushStack(Exp, Exp, OpSize::i64Bit, true);
|
||||
|
||||
@@ -442,7 +442,7 @@ constexpr std::array<X86InstInfo, MAX_REPNE_MOD_TABLE_SIZE> RepNEModOps = []() c
|
||||
{0x70, 1, X86InstInfo{"PSHUFLW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1}},
|
||||
{0x71, 3, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0}},
|
||||
{0x74, 4, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{0x78, 1, X86InstInfo{"INSERTQ", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_SF_MOD_REG_ONLY | FLAGS_XMM_FLAGS,2}},
|
||||
{0x78, 1, X86InstInfo{"INSERTQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_REG_ONLY | FLAGS_XMM_FLAGS,2}},
|
||||
{0x79, 1, X86InstInfo{"INSERTQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_REG_ONLY | FLAGS_XMM_FLAGS, 0}},
|
||||
{0x7A, 2, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{0x7C, 1, X86InstInfo{"HADDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
|
||||
@@ -7,6 +7,7 @@ $end_info$
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <array>
|
||||
@@ -42,8 +43,9 @@ constexpr uint32_t FLAG_DS_PREFIX = (0b100 << 11);
|
||||
constexpr uint32_t FLAG_FS_PREFIX = (0b101 << 11);
|
||||
constexpr uint32_t FLAG_GS_PREFIX = (0b110 << 11);
|
||||
constexpr uint32_t FLAG_SEGMENTS = (0b111 << 11);
|
||||
// Bits 14, 15, 16 - Unused
|
||||
|
||||
constexpr uint32_t FLAG_FORCE_TSO = (1 << 14);
|
||||
constexpr uint32_t FLAG_DECODED_MODRM = (1 << 15);
|
||||
constexpr uint32_t FLAG_DECODED_SIB = (1 << 16);
|
||||
constexpr uint32_t FLAG_REP_PREFIX = (1 << 17);
|
||||
constexpr uint32_t FLAG_REPNE_PREFIX = (1 << 18);
|
||||
// Size flags
|
||||
@@ -143,6 +145,9 @@ struct DecodedOperand {
|
||||
}
|
||||
uint64_t Literal() const {
|
||||
LOGMAN_THROW_A_FMT(IsLiteral(), "Precondition: must be a literal");
|
||||
if (Data.Literal.SignExtend) {
|
||||
return static_cast<int64_t>(static_cast<int32_t>(Data.Literal.Value));
|
||||
}
|
||||
return Data.Literal.Value;
|
||||
}
|
||||
|
||||
@@ -166,8 +171,9 @@ struct DecodedOperand {
|
||||
} RIPLiteral;
|
||||
|
||||
struct LiteralType {
|
||||
uint64_t Value;
|
||||
uint8_t Size;
|
||||
uint32_t Value;
|
||||
uint8_t Size : 7 ;
|
||||
bool SignExtend : 1;
|
||||
auto operator<=>(const LiteralType&) const = default;
|
||||
} Literal;
|
||||
|
||||
@@ -193,16 +199,12 @@ struct DecodedInst {
|
||||
X86InstInfo const* TableInfo;
|
||||
|
||||
uint32_t Flags;
|
||||
uint8_t OPRaw;
|
||||
uint16_t OP;
|
||||
uint8_t OPRaw;
|
||||
|
||||
uint8_t ModRM;
|
||||
uint8_t SIB;
|
||||
uint8_t InstSize;
|
||||
uint8_t LastEscapePrefix;
|
||||
bool DecodedModRM;
|
||||
bool DecodedSIB;
|
||||
bool ForceTSO;
|
||||
};
|
||||
|
||||
union ModRMDecoded {
|
||||
@@ -558,21 +560,6 @@ constexpr static inline void GenerateTableWithCopy(X86InstInfo *FinalTable, X86T
|
||||
}
|
||||
};
|
||||
|
||||
template<typename OpcodeType>
|
||||
static inline void LateInitCopyTable(X86InstInfo *FinalTable, X86TablesInfoStruct<OpcodeType> const *OtherLocal, size_t OtherTableSize) {
|
||||
for (size_t j = 0; j < OtherTableSize; ++j) {
|
||||
X86TablesInfoStruct<OpcodeType> const &OtherOp = OtherLocal[j];
|
||||
auto OtherOpNum = OtherOp.first;
|
||||
X86InstInfo const &OtherInfo = OtherOp.Info;
|
||||
for (uint32_t i = 0; i < OtherOp.second; ++i) {
|
||||
X86InstInfo &FinalOp = FinalTable[OtherOpNum + i];
|
||||
if (FinalOp.Type == TYPE_COPY_OTHER) {
|
||||
FinalOp = OtherInfo;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template<typename OpcodeType>
|
||||
constexpr static inline void GenerateX87Table(X86InstInfo *FinalTable, X86TablesInfoStruct<OpcodeType> const *LocalTable, size_t TableSize) {
|
||||
for (size_t j = 0; j < TableSize; ++j) {
|
||||
@@ -603,12 +590,6 @@ constexpr static inline void GenerateX87Table(X86InstInfo *FinalTable, X86Tables
|
||||
}
|
||||
};
|
||||
|
||||
}
|
||||
FEX_DEFINE_ENUM_FMT_PASSTHROUGH(FEXCore::X86Tables::DecodedOperand::OpType);
|
||||
|
||||
template <>
|
||||
struct fmt::formatter<FEXCore::X86Tables::DecodedOperand::OpType> : formatter<uint32_t> {
|
||||
template <typename FormatContext>
|
||||
auto format(FEXCore::X86Tables::DecodedOperand::OpType type, FormatContext& ctx) const {
|
||||
return fmt::formatter<uint32_t>::format(static_cast<uint32_t>(type), ctx);
|
||||
}
|
||||
};
|
||||
} // namespace FEXCore::X86Tables
|
||||
@@ -42,19 +42,19 @@ void __attribute__((noinline)) __jit_debug_register_code() {
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
void GDBJITRegister(FEXCore::IR::AOTIRCacheEntry* Entry, uintptr_t VAFileStart, uint64_t GuestRIP, uintptr_t HostEntry,
|
||||
FEXCore::Core::DebugData* DebugData) {
|
||||
auto map = Entry->SourcecodeMap.get();
|
||||
void GDBJITRegister(FEXCore::ExecutableFileInfo& Entry, uintptr_t VAFileStart, uint64_t GuestRIP, uintptr_t HostEntry,
|
||||
FEXCore::Core::DebugData& DebugData) {
|
||||
auto map = Entry.SourcecodeMap.get();
|
||||
|
||||
if (map) {
|
||||
auto FileOffset = GuestRIP - VAFileStart;
|
||||
|
||||
auto Sym = map->FindSymbolMapping(FileOffset);
|
||||
|
||||
auto SymName = HLE::SourcecodeSymbolMapping::SymName(Sym, Entry->Filename, HostEntry, FileOffset);
|
||||
auto SymName = HLE::SourcecodeSymbolMapping::SymName(Sym, Entry.Filename, HostEntry, FileOffset);
|
||||
|
||||
fextl::vector<gdb_line_mapping> Lines;
|
||||
for (const auto& GuestOpcode : DebugData->GuestOpcodes) {
|
||||
for (const auto& GuestOpcode : DebugData.GuestOpcodes) {
|
||||
auto Line = map->FindLineMapping(GuestRIP + GuestOpcode.GuestEntryOffset - VAFileStart);
|
||||
if (Line) {
|
||||
Lines.push_back({Line->LineNumber, HostEntry + GuestOpcode.HostEntryOffset});
|
||||
@@ -80,7 +80,7 @@ void GDBJITRegister(FEXCore::IR::AOTIRCacheEntry* Entry, uintptr_t VAFileStart,
|
||||
for (int i = 0; i < info->nblocks; i++) {
|
||||
strncpy(blocks[i].name, SymName.c_str(), 511);
|
||||
blocks[i].start = HostEntry;
|
||||
blocks[i].end = HostEntry + DebugData->HostCodeSize;
|
||||
blocks[i].end = HostEntry + DebugData.HostCodeSize;
|
||||
}
|
||||
|
||||
info->nlines = Lines.size();
|
||||
@@ -113,7 +113,7 @@ void GDBJITRegister(FEXCore::IR::AOTIRCacheEntry* Entry, uintptr_t VAFileStart,
|
||||
} // namespace FEXCore
|
||||
#else
|
||||
namespace FEXCore {
|
||||
void GDBJITRegister(FEXCore::IR::AOTIRCacheEntry*, uintptr_t, uint64_t, uintptr_t, FEXCore::Core::DebugData*) {
|
||||
void GDBJITRegister(FEXCore::ExecutableFileInfo&, uintptr_t, uint64_t, uintptr_t, FEXCore::Core::DebugData&) {
|
||||
ERROR_AND_DIE_FMT("GDBSymbols support not compiled in");
|
||||
}
|
||||
} // namespace FEXCore
|
||||
|
||||
@@ -1,9 +1,8 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
|
||||
#include <Interface/IR/AOTIR.h>
|
||||
#include <FEXCore/Core/CodeCache.h>
|
||||
#include <Interface/Core/JIT/DebugData.h>
|
||||
|
||||
namespace FEXCore {
|
||||
void GDBJITRegister(FEXCore::IR::AOTIRCacheEntry* Entry, uintptr_t VAFileStart, uint64_t GuestRIP, uintptr_t HostEntry,
|
||||
FEXCore::Core::DebugData* DebugData);
|
||||
}
|
||||
void GDBJITRegister(FEXCore::ExecutableFileInfo&, uintptr_t VAFileStart, uint64_t GuestRIP, uintptr_t HostEntry, FEXCore::Core::DebugData&);
|
||||
}
|
||||
@@ -1,60 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "FEXHeaderUtils/Filesystem.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/IR/AOTIR.h"
|
||||
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
|
||||
#include <Interface/Core/LookupCache.h>
|
||||
#include <Interface/GDBJIT/GDBJIT.h>
|
||||
|
||||
#include <xxhash.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
bool AOTIRCaptureCache::PostCompileCode(FEXCore::Core::InternalThreadState* Thread, void* CodePtr, uint64_t GuestRIP, uint64_t StartAddr,
|
||||
uint64_t Length, FEXCore::Core::DebugData* DebugData) {
|
||||
|
||||
// Both generated ir and LibraryJITName need a named region lookup
|
||||
if (CTX->Config.LibraryJITNaming() || CTX->Config.GDBSymbols()) {
|
||||
|
||||
auto AOTIRCacheEntry = CTX->SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
|
||||
|
||||
if (AOTIRCacheEntry.Entry) {
|
||||
if (DebugData && CTX->Config.LibraryJITNaming()) {
|
||||
CTX->Symbols.RegisterNamedRegion(Thread->SymbolBuffer.get(), CodePtr, DebugData->HostCodeSize, AOTIRCacheEntry.Entry->Filename);
|
||||
}
|
||||
|
||||
if (CTX->Config.GDBSymbols()) {
|
||||
GDBJITRegister(AOTIRCacheEntry.Entry, AOTIRCacheEntry.VAFileStart, GuestRIP, (uintptr_t)CodePtr, DebugData);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
AOTIRCacheEntry* AOTIRCaptureCache::LoadAOTIRCacheEntry(const fextl::string& filename) {
|
||||
fextl::string base_filename = FHU::Filesystem::GetFilename(filename);
|
||||
|
||||
if (!base_filename.empty()) {
|
||||
auto filename_hash = XXH3_64bits(filename.c_str(), filename.size());
|
||||
|
||||
auto fileid = fextl::fmt::format("{}-{}-{}{}{}", base_filename, filename_hash,
|
||||
(CTX->Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL) ? 'S' : 's',
|
||||
CTX->Config.TSOEnabled ? 'T' : 't', CTX->Config.ABILocalFlags ? 'L' : 'l');
|
||||
|
||||
std::unique_lock lk(AOTIRCacheLock);
|
||||
|
||||
auto Inserted = AOTIRCache.insert({fileid, AOTIRCacheEntry {.FileId = fileid, .Filename = filename}});
|
||||
auto Entry = &(Inserted.first->second);
|
||||
|
||||
return Entry;
|
||||
}
|
||||
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -1,72 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/AllocatorHooks.h>
|
||||
#include <FEXCore/HLE/SourcecodeResolver.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <shared_mutex>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
union Relocation;
|
||||
} // namespace FEXCore::CPU
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
struct DebugDataSubblock {
|
||||
uint32_t HostCodeOffset;
|
||||
uint32_t HostCodeSize;
|
||||
};
|
||||
|
||||
struct DebugDataGuestOpcode {
|
||||
uint64_t GuestEntryOffset;
|
||||
ptrdiff_t HostEntryOffset;
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Contains debug data for a block of code for later debugger analysis
|
||||
*
|
||||
* Needs to remain around for as long as the code could be executed at least
|
||||
*/
|
||||
struct DebugData : public FEXCore::Allocator::FEXAllocOperators {
|
||||
uint64_t HostCodeSize; ///< The size of the code generated in the host JIT
|
||||
fextl::vector<DebugDataSubblock> Subblocks;
|
||||
fextl::vector<DebugDataGuestOpcode> GuestOpcodes;
|
||||
fextl::vector<FEXCore::CPU::Relocation>* Relocations;
|
||||
};
|
||||
} // namespace FEXCore::Core
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
struct AOTIRCacheEntry {
|
||||
fextl::unique_ptr<FEXCore::HLE::SourcecodeMap> SourcecodeMap;
|
||||
fextl::string FileId;
|
||||
fextl::string Filename;
|
||||
};
|
||||
|
||||
class AOTIRCaptureCache final {
|
||||
public:
|
||||
AOTIRCaptureCache(FEXCore::Context::ContextImpl* ctx)
|
||||
: CTX {ctx} {}
|
||||
|
||||
bool PostCompileCode(FEXCore::Core::InternalThreadState* Thread, void* CodePtr, uint64_t GuestRIP, uint64_t StartAddr, uint64_t Length,
|
||||
FEXCore::Core::DebugData* DebugData);
|
||||
|
||||
AOTIRCacheEntry* LoadAOTIRCacheEntry(const fextl::string& filename);
|
||||
|
||||
private:
|
||||
FEXCore::Context::ContextImpl* CTX;
|
||||
|
||||
std::shared_mutex AOTIRCacheLock;
|
||||
|
||||
fextl::unordered_map<fextl::string, FEXCore::IR::AOTIRCacheEntry> AOTIRCache;
|
||||
};
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -1,7 +1,9 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/ThreadPoolAllocator.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
@@ -9,11 +11,15 @@
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <functional>
|
||||
#include <iterator>
|
||||
#include <type_traits>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
class OrderedNode;
|
||||
class RegisterAllocationPass;
|
||||
|
||||
/**
|
||||
* @brief The IROp_Header is an dynamically sized array
|
||||
@@ -235,8 +241,6 @@ static_assert(sizeof(OrderedNodeHeader) == sizeof(uint32_t) * 3);
|
||||
* The second region is contiguous but they don't have any relationship with one another directly
|
||||
*/
|
||||
class OrderedNode final {
|
||||
friend class NodeWrapperIterator;
|
||||
friend class OrderedList;
|
||||
public:
|
||||
// These three values are laid out very specifically to make it fast to access the NodeWrappers specifically
|
||||
OrderedNodeHeader Header;
|
||||
@@ -426,94 +430,6 @@ static_assert(sizeof(OrderedNode) == (sizeof(OrderedNodeHeader) + 2 * sizeof(uin
|
||||
// };
|
||||
using Ref = OrderedNode*;
|
||||
|
||||
struct FEX_PACKED RegisterClassType final {
|
||||
using value_type = uint32_t;
|
||||
|
||||
value_type Val;
|
||||
[[nodiscard]] constexpr operator value_type() const {
|
||||
return Val;
|
||||
}
|
||||
[[nodiscard]]
|
||||
friend constexpr bool operator==(const RegisterClassType&, const RegisterClassType&) = default;
|
||||
};
|
||||
|
||||
struct FEX_PACKED CondClassType final {
|
||||
uint8_t Val;
|
||||
[[nodiscard]] constexpr operator uint8_t() const {
|
||||
return Val;
|
||||
}
|
||||
[[nodiscard]]
|
||||
friend constexpr bool operator==(const CondClassType&, const CondClassType&) = default;
|
||||
};
|
||||
|
||||
struct FEX_PACKED MemOffsetType final {
|
||||
uint8_t Val;
|
||||
[[nodiscard]] constexpr operator uint8_t() const {
|
||||
return Val;
|
||||
}
|
||||
[[nodiscard]]
|
||||
friend constexpr bool operator==(const MemOffsetType&, const MemOffsetType&) = default;
|
||||
};
|
||||
|
||||
struct FEX_PACKED TypeDefinition final {
|
||||
uint16_t Val;
|
||||
|
||||
[[nodiscard]] constexpr operator uint16_t() const {
|
||||
return Val;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static constexpr TypeDefinition Create(uint8_t Bytes) {
|
||||
TypeDefinition Type {};
|
||||
Type.Val = Bytes << 8;
|
||||
return Type;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static constexpr TypeDefinition Create(uint8_t Bytes, uint8_t Elements) {
|
||||
TypeDefinition Type {};
|
||||
Type.Val = (Bytes << 8) | (Elements & 255);
|
||||
return Type;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
constexpr uint8_t Bytes() const {
|
||||
return Val >> 8;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
constexpr uint8_t Elements() const {
|
||||
return Val & 255;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
friend constexpr bool operator==(const TypeDefinition&, const TypeDefinition&) = default;
|
||||
};
|
||||
|
||||
static_assert(std::is_trivially_copyable_v<TypeDefinition>);
|
||||
|
||||
struct FEX_PACKED FenceType final {
|
||||
using value_type = uint8_t;
|
||||
|
||||
value_type Val;
|
||||
[[nodiscard]] constexpr operator value_type() const {
|
||||
return Val;
|
||||
}
|
||||
[[nodiscard]]
|
||||
friend constexpr bool operator==(const FenceType&, const FenceType&) = default;
|
||||
};
|
||||
|
||||
struct FEX_PACKED RoundType final {
|
||||
uint8_t Val;
|
||||
[[nodiscard]] constexpr operator uint8_t() const {
|
||||
return Val;
|
||||
}
|
||||
[[nodiscard]]
|
||||
friend constexpr bool operator==(const RoundType&, const RoundType&) = default;
|
||||
};
|
||||
|
||||
class NodeIterator;
|
||||
|
||||
/* This iterator can be used to step though nodes.
|
||||
* Due to how our IR is laid out, this can be used to either step
|
||||
* though the CodeBlocks or though the code within a single block.
|
||||
@@ -521,8 +437,8 @@ class NodeIterator;
|
||||
class NodeIterator {
|
||||
public:
|
||||
struct value_type final {
|
||||
OrderedNode *Node;
|
||||
IROp_Header *Header;
|
||||
OrderedNode* Node;
|
||||
IROp_Header* Header;
|
||||
};
|
||||
using size_type = std::size_t;
|
||||
using difference_type = std::ptrdiff_t;
|
||||
@@ -777,6 +693,15 @@ inline NodeID NodeWrapperBase<Type>::ID() const {
|
||||
bool IsBlockExit(FEXCore::IR::IROps Op);
|
||||
|
||||
void Dump(fextl::stringstream* out, const IRListView* IR);
|
||||
|
||||
constexpr auto format_as(FEXCore::IR::NodeID ID) {
|
||||
return ID.Value;
|
||||
}
|
||||
|
||||
FEX_DEFINE_ENUM_FMT_PASSTHROUGH(FEXCore::IR::FenceType)
|
||||
FEX_DEFINE_ENUM_FMT_PASSTHROUGH(FEXCore::IR::MemOffsetType)
|
||||
FEX_DEFINE_ENUM_FMT_PASSTHROUGH(FEXCore::IR::OpSize)
|
||||
FEX_DEFINE_ENUM_FMT_PASSTHROUGH(FEXCore::IR::RegClass)
|
||||
} // namespace FEXCore::IR
|
||||
|
||||
template<>
|
||||
@@ -785,45 +710,3 @@ struct std::hash<FEXCore::IR::NodeID> {
|
||||
return std::hash<FEXCore::IR::NodeID::value_type> {}(ID.Value);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct fmt::formatter<FEXCore::IR::NodeID> : fmt::formatter<FEXCore::IR::NodeID::value_type> {
|
||||
using Base = fmt::formatter<FEXCore::IR::NodeID::value_type>;
|
||||
|
||||
// Pass-through the underlying value, so IDs can
|
||||
// be formatted like any integral value.
|
||||
template<typename FormatContext>
|
||||
auto format(const FEXCore::IR::NodeID& ID, FormatContext& ctx) const {
|
||||
return Base::format(ID.Value, ctx);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct fmt::formatter<FEXCore::IR::RegisterClassType> : fmt::formatter<FEXCore::IR::RegisterClassType::value_type> {
|
||||
using Base = fmt::formatter<FEXCore::IR::RegisterClassType::value_type>;
|
||||
|
||||
template<typename FormatContext>
|
||||
auto format(const FEXCore::IR::RegisterClassType& Class, FormatContext& ctx) const {
|
||||
return Base::format(Class.Val, ctx);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct fmt::formatter<FEXCore::IR::FenceType> : fmt::formatter<FEXCore::IR::FenceType::value_type> {
|
||||
using Base = fmt::formatter<FEXCore::IR::FenceType::value_type>;
|
||||
|
||||
template<typename FormatContext>
|
||||
auto format(const FEXCore::IR::FenceType& Fence, FormatContext& ctx) const {
|
||||
return Base::format(Fence.Val, ctx);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct fmt::formatter<FEXCore::IR::OpSize> : fmt::formatter<std::underlying_type_t<FEXCore::IR::OpSize>> {
|
||||
using Base = fmt::formatter<std::underlying_type_t<FEXCore::IR::OpSize>>;
|
||||
|
||||
template<typename FormatContext>
|
||||
auto format(const FEXCore::IR::OpSize& OpSize, FormatContext& ctx) const {
|
||||
return Base::format(FEXCore::ToUnderlying(OpSize), ctx);
|
||||
}
|
||||
};
|
||||
+123
-109
@@ -52,80 +52,68 @@
|
||||
" * These are validations that can't be automatically inferred and need to be hand-written",
|
||||
""
|
||||
],
|
||||
"Enums": {
|
||||
"class CondClass : uint8_t": [
|
||||
"EQ = 0,",
|
||||
"NEQ = 1,",
|
||||
"UGE = 2,",
|
||||
"ULT = 3,",
|
||||
"MI = 4,",
|
||||
"PL = 5,",
|
||||
"VS = 6,",
|
||||
"VC = 7,",
|
||||
"UGT = 8,",
|
||||
"ULE = 9,",
|
||||
"SGE = 10,",
|
||||
"SLT = 11,",
|
||||
"SGT = 12,",
|
||||
"SLE = 13,",
|
||||
"TSTZ = 14, /* bit test zero */",
|
||||
"TSTNZ = 15, /* bit test nonzero */",
|
||||
"",
|
||||
"FLU = 16, /* float less or unordered */",
|
||||
"FGE = 17, /* float greater or equal */",
|
||||
"FLEU = 18, /* float less or equal or unordered */",
|
||||
"FGT = 19, /* float greater */",
|
||||
"FU = 20, /* float unordered */",
|
||||
"FNU = 21, /* float not unordered */",
|
||||
"",
|
||||
"AL = 32, /* always */"
|
||||
],
|
||||
"class FenceType : uint8_t": [
|
||||
"Load = 0,",
|
||||
"Store = 1,",
|
||||
"LoadStore = 2,",
|
||||
"Inst = 3,"
|
||||
],
|
||||
"class MemOffsetType : uint8_t": [
|
||||
"SXTX = 0,",
|
||||
"UXTW = 1,",
|
||||
"SXTW = 2,"
|
||||
],
|
||||
"class RegClass : uint32_t": [
|
||||
"Invalid = 0,",
|
||||
"GPR = 1,",
|
||||
"GPRFixed = 2,",
|
||||
"FPR = 3,",
|
||||
"FPRFixed = 4,",
|
||||
"Complex = 5,"
|
||||
],
|
||||
"class RoundMode : uint8_t": [
|
||||
"Nearest = 0,",
|
||||
"NegInfinity = 1,",
|
||||
"PosInfinity = 2,",
|
||||
"TowardsZero = 3, /* Truncate */",
|
||||
"Host = 4,"
|
||||
]
|
||||
},
|
||||
"Defines": [
|
||||
"constexpr uint8_t COND_EQ = 0",
|
||||
"constexpr uint8_t COND_NEQ = 1",
|
||||
"constexpr uint8_t COND_UGE = 2",
|
||||
"constexpr uint8_t COND_ULT = 3",
|
||||
"constexpr uint8_t COND_MI = 4",
|
||||
"constexpr uint8_t COND_PL = 5",
|
||||
"constexpr uint8_t COND_VS = 6",
|
||||
"constexpr uint8_t COND_VC = 7",
|
||||
"constexpr uint8_t COND_UGT = 8",
|
||||
"constexpr uint8_t COND_ULE = 9",
|
||||
"constexpr uint8_t COND_SGE = 10",
|
||||
"constexpr uint8_t COND_SLT = 11",
|
||||
"constexpr uint8_t COND_SGT = 12",
|
||||
"constexpr uint8_t COND_SLE = 13",
|
||||
"constexpr uint8_t COND_TSTZ = 14 /* bit test zero */",
|
||||
"constexpr uint8_t COND_TSTNZ = 15 /* bit test nonzero */",
|
||||
|
||||
"constexpr uint8_t COND_FLU = 16 /* float less or unordred */",
|
||||
"constexpr uint8_t COND_FGE = 17 /* float greater or equal */",
|
||||
"constexpr uint8_t COND_FLEU = 18 /* float less or equal or unordred */",
|
||||
"constexpr uint8_t COND_FGT = 19 /* float greater */",
|
||||
"constexpr uint8_t COND_FU = 20 /* float unordred */",
|
||||
"constexpr uint8_t COND_FNU = 21 /* float not unordred */",
|
||||
|
||||
"constexpr uint8_t COND_AL = 32 /* always */",
|
||||
|
||||
"constexpr FEXCore::IR::RegisterClassType InvalidClass {0}",
|
||||
"constexpr FEXCore::IR::RegisterClassType GPRClass {1}",
|
||||
"constexpr FEXCore::IR::RegisterClassType GPRFixedClass {2}",
|
||||
"constexpr FEXCore::IR::RegisterClassType FPRClass {3}",
|
||||
"constexpr FEXCore::IR::RegisterClassType FPRFixedClass {4}",
|
||||
"constexpr FEXCore::IR::RegisterClassType ComplexClass {5}",
|
||||
"constexpr uint8_t NumClasses {6}",
|
||||
"",
|
||||
"constexpr FEXCore::IR::TypeDefinition i8 {TypeDefinition::Create(1, 0)}",
|
||||
"constexpr FEXCore::IR::TypeDefinition i16 {TypeDefinition::Create(2, 0)}",
|
||||
"constexpr FEXCore::IR::TypeDefinition i32 {TypeDefinition::Create(4, 0)}",
|
||||
"constexpr FEXCore::IR::TypeDefinition i64 {TypeDefinition::Create(8, 0)}",
|
||||
"constexpr FEXCore::IR::TypeDefinition i128 {TypeDefinition::Create(16, 0)}",
|
||||
"",
|
||||
"constexpr FEXCore::IR::TypeDefinition i8v8 {TypeDefinition::Create(1, 8)}",
|
||||
"constexpr FEXCore::IR::TypeDefinition i8v16 {TypeDefinition::Create(1, 16)}",
|
||||
"constexpr FEXCore::IR::TypeDefinition i16v4 {TypeDefinition::Create(2, 4)}",
|
||||
"constexpr FEXCore::IR::TypeDefinition i16v8 {TypeDefinition::Create(2, 8)}",
|
||||
"constexpr FEXCore::IR::TypeDefinition i32v2 {TypeDefinition::Create(4, 2)}",
|
||||
"constexpr FEXCore::IR::TypeDefinition i32v4 {TypeDefinition::Create(4, 4)}",
|
||||
"constexpr FEXCore::IR::TypeDefinition i64v2 {TypeDefinition::Create(8, 2)}",
|
||||
"",
|
||||
|
||||
"constexpr uint8_t FCMP_FLAG_EQ = 0",
|
||||
"constexpr uint8_t FCMP_FLAG_LT = 1",
|
||||
"constexpr uint8_t FCMP_FLAG_UNORDERED = 2",
|
||||
|
||||
"constexpr FEXCore::IR::FenceType Fence_Load {0}",
|
||||
"constexpr FEXCore::IR::FenceType Fence_Store {1}",
|
||||
"constexpr FEXCore::IR::FenceType Fence_LoadStore {2}",
|
||||
"constexpr FEXCore::IR::FenceType Fence_Inst {3}",
|
||||
|
||||
"constexpr uint8_t ROUND_MODE_NEAREST = 0",
|
||||
"constexpr uint8_t ROUND_MODE_NEGATIVE_INFINITY = 1",
|
||||
"constexpr uint8_t ROUND_MODE_POSITIVE_INFINITY = 2",
|
||||
"constexpr uint8_t ROUND_MODE_TOWARDS_ZERO = 3",
|
||||
"constexpr uint8_t ROUND_MODE_FLUSH_TO_ZERO = 1 << 2",
|
||||
|
||||
"constexpr FEXCore::IR::RoundType Round_Nearest {ROUND_MODE_NEAREST}",
|
||||
"constexpr FEXCore::IR::RoundType Round_Negative_Infinity {ROUND_MODE_NEGATIVE_INFINITY}",
|
||||
"constexpr FEXCore::IR::RoundType Round_Positive_Infinity {ROUND_MODE_POSITIVE_INFINITY}",
|
||||
"constexpr FEXCore::IR::RoundType Round_Towards_Zero {ROUND_MODE_TOWARDS_ZERO} /* Truncate */",
|
||||
"constexpr FEXCore::IR::RoundType Round_Host {ROUND_MODE_TOWARDS_ZERO + 1}",
|
||||
|
||||
"constexpr FEXCore::IR::MemOffsetType MEM_OFFSET_SXTX {0}",
|
||||
"constexpr FEXCore::IR::MemOffsetType MEM_OFFSET_UXTW {1}",
|
||||
"constexpr FEXCore::IR::MemOffsetType MEM_OFFSET_SXTW {2}",
|
||||
|
||||
"struct BreakDefinition {",
|
||||
" uint16_t ErrorRegister;",
|
||||
" uint8_t Signal;",
|
||||
@@ -148,13 +136,13 @@
|
||||
"GPR": "OrderedNode*",
|
||||
"FPR": "OrderedNode*",
|
||||
"FenceType": "FenceType",
|
||||
"RegisterClass": "RegisterClassType",
|
||||
"CondClass": "CondClassType",
|
||||
"RegisterClass": "RegClass",
|
||||
"CondClass": "CondClass",
|
||||
"SyscallFlags": "FEXCore::IR::SyscallFlags",
|
||||
"SHA256Sum": "SHA256Sum",
|
||||
"MemOffsetType": "MemOffsetType",
|
||||
"BreakDefinition": "BreakDefinition",
|
||||
"RoundType": "RoundType",
|
||||
"RoundType": "RoundMode",
|
||||
"FloatCompareOp": "FloatCompareOp",
|
||||
"NamedVectorConstant": "FEXCore::IR::NamedVectorConstant",
|
||||
"IndexNamedVectorConstant": "FEXCore::IR::IndexNamedVectorConstant",
|
||||
@@ -307,7 +295,7 @@
|
||||
"HasSideEffects": true,
|
||||
"RAOverride": "0"
|
||||
},
|
||||
"CondJump SSA:$Cmp1, SSA:$Cmp2, SSA:$TrueBlock, SSA:$FalseBlock, CondClass:$Cond{{COND_NEQ}}, OpSize:$CompareSize{OpSize::iInvalid}, i1:$FromNZCV{false}": {
|
||||
"CondJump SSA:$Cmp1, SSA:$Cmp2, SSA:$TrueBlock, SSA:$FalseBlock, CondClass:$Cond{CondClass::NEQ}, OpSize:$CompareSize{OpSize::iInvalid}, i1:$FromNZCV{false}": {
|
||||
"Inline": ["", "AddSub"],
|
||||
"HasSideEffects": true,
|
||||
"RAOverride": "2"
|
||||
@@ -364,20 +352,6 @@
|
||||
"GPR = Copy GPR:$Source": {
|
||||
"Desc": ["GPR copy, generated by RA to split live ranges"],
|
||||
"DestSize": "OpSize::i64Bit"
|
||||
},
|
||||
|
||||
"GPR = Swap1 GPR:$A, GPR:$B": {
|
||||
"Desc": ["GPR swap part 1, generated by RA. Returns value of first source.",
|
||||
"Destination must be second GPR."],
|
||||
"DestSize": "OpSize::i64Bit"
|
||||
},
|
||||
|
||||
"GPR = Swap2": {
|
||||
"Desc": ["GPR swap part 2, generated by RA. Returns source source.",
|
||||
"Must immediately succeed Swap1 with no intervening instructions",
|
||||
"Kludge to workaround single destination restriction on IR",
|
||||
"Hopefully temporary"],
|
||||
"DestSize": "OpSize::i64Bit"
|
||||
}
|
||||
},
|
||||
"StaticRA": {
|
||||
@@ -423,8 +397,8 @@
|
||||
],
|
||||
"DestSize": "ByteSize",
|
||||
"EmitValidation": [
|
||||
"($Class == GPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == GPRClass",
|
||||
"($Class == RegClass::GPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == RegClass::FPR",
|
||||
"($Class == RegClass::FPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == RegClass::GPR",
|
||||
"!($Offset >= offsetof(Core::CPUState, gregs[0]) && $Offset < offsetof(Core::CPUState, gregs[16])) && \"Can't LoadContext to GPR\"",
|
||||
"!($Offset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $Offset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't LoadContext to XMM\""
|
||||
]
|
||||
@@ -437,8 +411,8 @@
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "ByteSize",
|
||||
"EmitValidation": [
|
||||
"($Class == GPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == GPRClass",
|
||||
"($Class == RegClass::GPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == RegClass::FPR",
|
||||
"($Class == RegClass::FPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == RegClass::GPR",
|
||||
"!($Offset >= offsetof(Core::CPUState, gregs[0]) && $Offset < offsetof(Core::CPUState, gregs[16])) && \"Can't LoadContext to GPR\"",
|
||||
"!($Offset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $Offset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't LoadContext to XMM\""
|
||||
]
|
||||
@@ -454,8 +428,8 @@
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "ByteSize",
|
||||
"EmitValidation": [
|
||||
"($Class == GPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == GPRClass",
|
||||
"($Class == RegClass::GPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == RegClass::FPR",
|
||||
"($Class == RegClass::FPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == RegClass::GPR",
|
||||
"!($Offset >= offsetof(Core::CPUState, gregs[0]) && $Offset < offsetof(Core::CPUState, gregs[16])) && \"Can't StoreContext to GPR\"",
|
||||
"!($Offset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $Offset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't StoreContext to XMM\""
|
||||
]
|
||||
@@ -472,8 +446,8 @@
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value1) == $Class",
|
||||
"WalkFindRegClass($Value2) == $Class",
|
||||
"($Class == GPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == GPRClass",
|
||||
"($Class == RegClass::GPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == RegClass::FPR",
|
||||
"($Class == RegClass::FPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == RegClass::GPR",
|
||||
"!($Offset >= offsetof(Core::CPUState, gregs[0]) && $Offset < offsetof(Core::CPUState, gregs[16])) && \"Can't StoreContext to GPR\"",
|
||||
"!($Offset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $Offset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't StoreContext to XMM\""
|
||||
]
|
||||
@@ -485,8 +459,8 @@
|
||||
],
|
||||
"DestSize": "ByteSize",
|
||||
"EmitValidation": [
|
||||
"($Class == GPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == GPRClass",
|
||||
"($Class == RegClass::GPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == RegClass::FPR",
|
||||
"($Class == RegClass::FPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == RegClass::GPR",
|
||||
"!($BaseOffset >= offsetof(Core::CPUState, gregs[0]) && $BaseOffset < offsetof(Core::CPUState, gregs[16])) && \"Can't LoadContextIndexed to GPR\"",
|
||||
"!($BaseOffset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $BaseOffset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't LoadContextIndexed to XMM\""
|
||||
]
|
||||
@@ -499,12 +473,23 @@
|
||||
"DestSize": "ByteSize",
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value) == $Class",
|
||||
"($Class == GPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == GPRClass",
|
||||
"($Class == RegClass::GPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == RegClass::FPR",
|
||||
"($Class == RegClass::FPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == RegClass::GPR",
|
||||
"!($BaseOffset >= offsetof(Core::CPUState, gregs[0]) && $BaseOffset < offsetof(Core::CPUState, gregs[16])) && \"Can't StoreContextIndexed to GPR\"",
|
||||
"!($BaseOffset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $BaseOffset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't StoreContextIndexed to XMM\""
|
||||
]
|
||||
},
|
||||
"GPR = FormContextAddress OpSize:#Size, GPR:$Index, u32:$Stride": {
|
||||
"Desc": ["Forms an address into the context structure indexed by SSA value",
|
||||
"Dest = Ctx + Index * Stride",
|
||||
"This allows backends to compute the address once and reuse it for multiple memory operations",
|
||||
"Stride must be a power of 2"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"#Size == IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
|
||||
"SpillRegister SSA:$Value, u32:$Slot, RegisterClass:$Class": {
|
||||
"HasSideEffects": true,
|
||||
@@ -621,7 +606,7 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VLoadVectorGatherMasked OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Incoming, FPR:$Mask, GPR:$AddrBase, FPR:$VectorIndexLow, FPR:$VectorIndexHigh, OpSize:$VectorIndexElementSize, u8:$OffsetScale, u8:$DataElementOffsetStart, u8:$IndexElementOffsetStart": {
|
||||
"FPR = VLoadVectorGatherMasked OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Incoming, FPR:$Mask, GPR:$AddrBase, FPR:$VectorIndexLow, FPR:$VectorIndexHigh, OpSize:$VectorIndexElementSize, u8:$OffsetScale, u8:$DataElementOffsetStart, u8:$IndexElementOffsetStart, OpSize:$AddrSize": {
|
||||
"Desc": [
|
||||
"Does a masked load similar to VPGATHERD* where the upper bit of each element",
|
||||
"determines whether or not that element will be loaded from memory.",
|
||||
@@ -635,7 +620,7 @@
|
||||
"$VectorIndexElementSize == OpSize::i32Bit || $VectorIndexElementSize == OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"FPR = VLoadVectorGatherMaskedQPS OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Incoming, FPR:$MaskReg, GPR:$AddrBase, FPR:$VectorIndexLow, FPR:$VectorIndexHigh, u8:$OffsetScale": {
|
||||
"FPR = VLoadVectorGatherMaskedQPS OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Incoming, FPR:$MaskReg, GPR:$AddrBase, FPR:$VectorIndexLow, FPR:$VectorIndexHigh, u8:$OffsetScale, OpSize:$AddrSize": {
|
||||
"Desc": [
|
||||
"Does a masked load similar to VPGATHERQPS where the upper bit of each element",
|
||||
"determines whether or not that element will be loaded from memory.",
|
||||
@@ -750,9 +735,10 @@
|
||||
},
|
||||
"Fence FenceType:$Fence": {
|
||||
"Desc": ["Does a memory fence operation of the desired type",
|
||||
"Fence_Load: Ensures load memory operations are serialized",
|
||||
"Fence_Store: Ensures store memory operations are serialized",
|
||||
"Fence_LoadStore: Ensures loads and store memory operations are serialized",
|
||||
"FenceType::Load: Ensures load memory operations are serialized",
|
||||
"FenceType::Store: Ensures store memory operations are serialized",
|
||||
"FenceType::LoadStore: Ensures loads and store memory operations are serialized",
|
||||
"FenceType::Inst: Instruction barrier. Ensures all instructions after this point will be explicitly fetched",
|
||||
"Ensures the memory operations are globally visible"
|
||||
],
|
||||
"HasSideEffects": true
|
||||
@@ -997,7 +983,7 @@
|
||||
"DestSize": "OpSize::i64Bit"
|
||||
},
|
||||
|
||||
"GPR = Neg OpSize:#Size, GPR:$Src, CondClass:$Cond{{COND_AL}}": {
|
||||
"GPR = Neg OpSize:#Size, GPR:$Src, CondClass:$Cond{CondClass::AL}": {
|
||||
"Desc": ["Integer negation, with optional predication",
|
||||
"Dest = Cond ? -Src : Src",
|
||||
"Will truncate to 64 or 32bits"
|
||||
@@ -1075,6 +1061,13 @@
|
||||
"Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = Rbit OpSize:#Size, GPR:$Src": {
|
||||
"Desc": ["Reverses the bit order of the register"],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = Add OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": [ "Integer Add",
|
||||
"Will truncate to 64 or 32bits"
|
||||
@@ -1542,6 +1535,15 @@
|
||||
"ResultSize == FEXCore::IR::OpSize::i32Bit || ResultSize == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = MaskGenerateFromBitWidth GPR:$BitWidth": {
|
||||
"Desc": ["Generates a bit mask from with a value from [0, 63]",
|
||||
"0 is special cased to full-mask",
|
||||
"Special operation for SSE4a bitmask generation."
|
||||
],
|
||||
"DestSize": "FEXCore::IR::OpSize::i64Bit",
|
||||
"ImplicitFlagClobber": true
|
||||
},
|
||||
|
||||
"GPR = Extr OpSize:#Size, GPR:$Upper, GPR:$Lower, u8:$LSB": {
|
||||
"Desc": ["Concats the two GPRs to create a value that is the size of the full two GPRs",
|
||||
"It then extracts a bitfield width that size of a GPR from the LSB",
|
||||
@@ -2150,22 +2152,34 @@
|
||||
|
||||
"FPR = VAnd OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
"ElementSize": "ElementSize",
|
||||
"EmitValidation": [
|
||||
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
|
||||
"FPR = VAndn OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
"ElementSize": "ElementSize",
|
||||
"EmitValidation": [
|
||||
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
|
||||
"FPR = VOr OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
"ElementSize": "ElementSize",
|
||||
"EmitValidation": [
|
||||
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
|
||||
"FPR = VXor OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
"ElementSize": "ElementSize",
|
||||
"EmitValidation": [
|
||||
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
|
||||
"FPR = VUQAdd OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
@@ -2829,7 +2843,7 @@
|
||||
"Int: 64-bit, 32-bit, 16-bit"
|
||||
],
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($OriginalValue) == FPRClass || WalkFindRegClass($OriginalValue) == GPRClass"
|
||||
"WalkFindRegClass($OriginalValue) == RegClass::FPR || WalkFindRegClass($OriginalValue) == RegClass::GPR"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"X87": true
|
||||
|
||||
@@ -38,8 +38,8 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, uint64_t Arg)
|
||||
*out << fextl::fmt::format("#{:#x}", Arg);
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, CondClassType Arg) {
|
||||
if (Arg == COND_AL) {
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, CondClass Arg) {
|
||||
if (Arg == CondClass::AL) {
|
||||
*out << "ALWAYS";
|
||||
return;
|
||||
}
|
||||
@@ -48,7 +48,7 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, CondClassType
|
||||
"UGT", "ULE", "SGE", "SLT", "SGT", "SLE", "TSTZ", "TSTNZ",
|
||||
"FLU", "FGE", "FLEU", "FGT", "FU", "FNU"};
|
||||
|
||||
*out << CondNames[Arg];
|
||||
*out << CondNames[FEXCore::ToUnderlying(Arg)];
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, MemOffsetType Arg) {
|
||||
@@ -58,39 +58,39 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, MemOffsetType
|
||||
"SXTW",
|
||||
};
|
||||
|
||||
*out << Names[Arg];
|
||||
*out << Names[FEXCore::ToUnderlying(Arg)];
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, RegisterClassType Arg) {
|
||||
if (Arg == GPRClass.Val) {
|
||||
*out << "GPR";
|
||||
} else if (Arg == GPRFixedClass.Val) {
|
||||
*out << "GPRFixed";
|
||||
} else if (Arg == FPRClass.Val) {
|
||||
*out << "FPR";
|
||||
} else if (Arg == FPRFixedClass.Val) {
|
||||
*out << "FPRFixed";
|
||||
} else {
|
||||
*out << "Unknown Registerclass " << Arg;
|
||||
}
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, RegClass Arg) {
|
||||
*out << [Arg] {
|
||||
switch (Arg) {
|
||||
case RegClass::Invalid: return "Invalid";
|
||||
case RegClass::GPR: return "GPR";
|
||||
case RegClass::GPRFixed: return "GPRFixed";
|
||||
case RegClass::FPR: return "FPR";
|
||||
case RegClass::FPRFixed: return "FPRFixed";
|
||||
case RegClass::Complex: return "Complex";
|
||||
}
|
||||
return "<Unknown RegClass Type>";
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNodeWrapper Arg) {
|
||||
if (Arg.IsImmediate()) {
|
||||
auto PhyReg = PhysicalRegister(Arg);
|
||||
|
||||
switch (PhyReg.Class) {
|
||||
case FEXCore::IR::GPRClass.Val: *out << "r"; break;
|
||||
case FEXCore::IR::GPRFixedClass.Val: *out << "R"; break;
|
||||
case FEXCore::IR::FPRClass.Val: *out << "v"; break;
|
||||
case FEXCore::IR::FPRFixedClass.Val: *out << "V"; break;
|
||||
case FEXCore::IR::ComplexClass.Val: *out << "c"; break;
|
||||
case FEXCore::IR::InvalidClass.Val: *out << "invalid"; break;
|
||||
switch (PhyReg.AsRegClass()) {
|
||||
case RegClass::GPR: *out << "r"; break;
|
||||
case RegClass::GPRFixed: *out << "R"; break;
|
||||
case RegClass::FPR: *out << "v"; break;
|
||||
case RegClass::FPRFixed: *out << "V"; break;
|
||||
case RegClass::Complex: *out << "c"; break;
|
||||
case RegClass::Invalid: *out << "invalid"; break;
|
||||
default: *out << "unknown"; break;
|
||||
}
|
||||
|
||||
if (PhyReg.Class != FEXCore::IR::InvalidClass.Val) {
|
||||
*out << std::dec << (uint32_t)PhyReg.Reg;
|
||||
if (PhyReg.AsRegClass() != RegClass::Invalid) {
|
||||
*out << std::dec << uint32_t(PhyReg.Reg);
|
||||
}
|
||||
|
||||
return;
|
||||
@@ -124,41 +124,46 @@ static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNode
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::FenceType Arg) {
|
||||
if (Arg == IR::Fence_Load) {
|
||||
*out << "Loads";
|
||||
} else if (Arg == IR::Fence_Store) {
|
||||
*out << "Stores";
|
||||
} else if (Arg == IR::Fence_LoadStore) {
|
||||
*out << "LoadStores";
|
||||
} else {
|
||||
*out << "<Unknown Fence Type>";
|
||||
}
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, FenceType Arg) {
|
||||
*out << [Arg] {
|
||||
switch (Arg) {
|
||||
case FenceType::Load: return "Loads";
|
||||
case FenceType::Store: return "Stores";
|
||||
case FenceType::LoadStore: return "LoadStores";
|
||||
case FenceType::Inst: return "Instruction";
|
||||
}
|
||||
return "<Unknown Fence Type>";
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::RoundType Arg) {
|
||||
switch (Arg) {
|
||||
case FEXCore::IR::Round_Nearest: *out << "Nearest"; break;
|
||||
case FEXCore::IR::Round_Negative_Infinity: *out << "-Inf"; break;
|
||||
case FEXCore::IR::Round_Positive_Infinity: *out << "+Inf"; break;
|
||||
case FEXCore::IR::Round_Towards_Zero: *out << "Towards Zero"; break;
|
||||
case FEXCore::IR::Round_Host: *out << "Host"; break;
|
||||
default: *out << "<Unknown Round Type>"; break;
|
||||
}
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, RoundMode Arg) {
|
||||
*out << [Arg] {
|
||||
switch (Arg) {
|
||||
case RoundMode::Nearest: return "Nearest";
|
||||
case RoundMode::NegInfinity: return "-Inf";
|
||||
case RoundMode::PosInfinity: return "+Inf";
|
||||
case RoundMode::TowardsZero: return "Towards Zero";
|
||||
case RoundMode::Host: return "Host";
|
||||
}
|
||||
return "<Unknown Round Type>";
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::SyscallFlags Arg) {
|
||||
switch (Arg) {
|
||||
case FEXCore::IR::SyscallFlags::DEFAULT: *out << "Default"; break;
|
||||
case FEXCore::IR::SyscallFlags::OPTIMIZETHROUGH: *out << "Optimize Through"; break;
|
||||
case FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY: *out << "No Sync State on Entry"; break;
|
||||
case FEXCore::IR::SyscallFlags::NORETURN: *out << "No Return"; break;
|
||||
case FEXCore::IR::SyscallFlags::NOSIDEEFFECTS: *out << "No Side Effects"; break;
|
||||
default: *out << "<Unknown Round Type>"; break;
|
||||
}
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, SyscallFlags Arg) {
|
||||
*out << [Arg] {
|
||||
switch (Arg) {
|
||||
case SyscallFlags::DEFAULT: return "Default";
|
||||
case SyscallFlags::OPTIMIZETHROUGH: return "Optimize Through";
|
||||
case SyscallFlags::NOSYNCSTATEONENTRY: return "No Sync State on Entry";
|
||||
case SyscallFlags::NORETURN: return "No Return";
|
||||
case SyscallFlags::NOSIDEEFFECTS: return "No Side Effects";
|
||||
case SyscallFlags::NORETURNEDRESULT: return "No Returned Result";
|
||||
}
|
||||
return "<Unknown Syscall Flags>";
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::NamedVectorConstant Arg) {
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, NamedVectorConstant Arg) {
|
||||
*out << [Arg] {
|
||||
// clang-format off
|
||||
switch (Arg) {
|
||||
@@ -186,6 +191,22 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::N
|
||||
return "movmskps_shift";
|
||||
case NamedVectorConstant::NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE:
|
||||
return "aeskeygenassist_swizzle";
|
||||
case NamedVectorConstant::NAMED_VECTOR_BLENDPS_0110B:
|
||||
return "blendps_0110b";
|
||||
case NamedVectorConstant::NAMED_VECTOR_BLENDPS_0111B:
|
||||
return "blendps_0111b";
|
||||
case NamedVectorConstant::NAMED_VECTOR_BLENDPS_1001B:
|
||||
return "blendps_1001b";
|
||||
case NamedVectorConstant::NAMED_VECTOR_BLENDPS_1011B:
|
||||
return "blendps_1011b";
|
||||
case NamedVectorConstant::NAMED_VECTOR_BLENDPS_1101B:
|
||||
return "blendps_1101b";
|
||||
case NamedVectorConstant::NAMED_VECTOR_BLENDPS_1110B:
|
||||
return "blendps_1110b";
|
||||
case NamedVectorConstant::NAMED_VECTOR_MOVMASKB:
|
||||
return "movmaskb";
|
||||
case NamedVectorConstant::NAMED_VECTOR_MOVMASKB_UPPER:
|
||||
return "movmaskb_upper";
|
||||
case NamedVectorConstant::NAMED_VECTOR_ZERO:
|
||||
return "vectorzero";
|
||||
case NamedVectorConstant::NAMED_VECTOR_X87_ONE:
|
||||
@@ -216,9 +237,20 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::N
|
||||
return "cvtmax_i32";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_I64:
|
||||
return "cvtmax_i64";
|
||||
default:
|
||||
return "<Unknown Named Vector Constant>";
|
||||
case NamedVectorConstant::NAMED_VECTOR_F80_SIGN_MASK:
|
||||
return "f80_sign_mask";
|
||||
case NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K0:
|
||||
return "sha1rnds_k0";
|
||||
case NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K1:
|
||||
return "sha1rnds_k1";
|
||||
case NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K2:
|
||||
return "sha1rnds_k2";
|
||||
case NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K3:
|
||||
return "sha1rnds_k3";
|
||||
case NamedVectorConstant::NAMED_VECTOR_MAX:
|
||||
return "<Programming Error: Printing MAX value>";
|
||||
}
|
||||
return "<Unknown Named Vector Constant>";
|
||||
// clang-format on
|
||||
}();
|
||||
}
|
||||
@@ -241,36 +273,43 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, IndexNamedVect
|
||||
return "dppd_mask";
|
||||
case IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PBLENDW:
|
||||
return "pblendw";
|
||||
default:
|
||||
return "<Unknown Indexed Named Vector Constant>";
|
||||
case INDEXED_NAMED_VECTOR_MAX:
|
||||
return "<Programming Error: Printing MAX value>";
|
||||
}
|
||||
return "<Unknown Indexed Named Vector Constant>";
|
||||
// clang-format on
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::OpSize Arg) {
|
||||
switch (Arg) {
|
||||
case OpSize::i8Bit: *out << "i8"; break;
|
||||
case OpSize::i16Bit: *out << "i16"; break;
|
||||
case OpSize::i32Bit: *out << "i32"; break;
|
||||
case OpSize::i64Bit: *out << "i64"; break;
|
||||
case OpSize::i128Bit: *out << "i128"; break;
|
||||
case OpSize::i256Bit: *out << "i256"; break;
|
||||
case OpSize::f80Bit: *out << "f80"; break;
|
||||
default: *out << "<Unknown OpSize Type>"; break;
|
||||
}
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, OpSize Arg) {
|
||||
*out << [Arg] {
|
||||
switch (Arg) {
|
||||
case OpSize::iUnsized: return "Unsized";
|
||||
case OpSize::i8Bit: return "i8";
|
||||
case OpSize::i16Bit: return "i16";
|
||||
case OpSize::i32Bit: return "i32";
|
||||
case OpSize::i64Bit: return "i64";
|
||||
case OpSize::f80Bit: return "f80";
|
||||
case OpSize::i128Bit: return "i128";
|
||||
case OpSize::i256Bit: return "i256";
|
||||
case OpSize::iInvalid: return "Invalid";
|
||||
}
|
||||
return "<Unknown OpSize Type>";
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::FloatCompareOp Arg) {
|
||||
switch (Arg) {
|
||||
case FloatCompareOp::EQ: *out << "FEQ"; break;
|
||||
case FloatCompareOp::LT: *out << "FLT"; break;
|
||||
case FloatCompareOp::LE: *out << "FLE"; break;
|
||||
case FloatCompareOp::UNO: *out << "UNO"; break;
|
||||
case FloatCompareOp::NEQ: *out << "NEQ"; break;
|
||||
case FloatCompareOp::ORD: *out << "ORD"; break;
|
||||
default: *out << "<Unknown OpSize Type>"; break;
|
||||
}
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, FloatCompareOp Arg) {
|
||||
*out << [Arg] {
|
||||
switch (Arg) {
|
||||
case FloatCompareOp::EQ: return "FEQ";
|
||||
case FloatCompareOp::LT: return "FLT";
|
||||
case FloatCompareOp::LE: return "FLE";
|
||||
case FloatCompareOp::UNO: return "UNO";
|
||||
case FloatCompareOp::NEQ: return "NEQ";
|
||||
case FloatCompareOp::ORD: return "ORD";
|
||||
}
|
||||
return "<Unknown FloatCompareOp Type>";
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::BreakDefinition Arg) {
|
||||
@@ -280,23 +319,28 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::B
|
||||
*out << static_cast<uint32_t>(Arg.si_code) << "}";
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::ShiftType Arg) {
|
||||
switch (Arg) {
|
||||
case ShiftType::LSL: *out << "LSL"; break;
|
||||
case ShiftType::LSR: *out << "LSR"; break;
|
||||
case ShiftType::ASR: *out << "ASR"; break;
|
||||
case ShiftType::ROR: *out << "ROR"; break;
|
||||
default: *out << "<Unknown Shift Type>"; break;
|
||||
}
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, ShiftType Arg) {
|
||||
*out << [Arg] {
|
||||
switch (Arg) {
|
||||
case ShiftType::LSL: return "LSL";
|
||||
case ShiftType::LSR: return "LSR";
|
||||
case ShiftType::ASR: return "ASR";
|
||||
case ShiftType::ROR: return "ROR";
|
||||
}
|
||||
return "<Unknown Shift Type>";
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::BranchHint Arg) {
|
||||
switch (Arg) {
|
||||
case BranchHint::None: *out << "None"; break;
|
||||
case BranchHint::Call: *out << "Call"; break;
|
||||
case BranchHint::Return: *out << "Return"; break;
|
||||
default: *out << "<Unknown Branch Hint>"; break;
|
||||
}
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, BranchHint Arg) {
|
||||
*out << [Arg] {
|
||||
switch (Arg) {
|
||||
case BranchHint::None: return "None";
|
||||
case BranchHint::Call: return "Call";
|
||||
case BranchHint::Return: return "Return";
|
||||
case BranchHint::CheckTF: return "CheckTF";
|
||||
}
|
||||
return "<Unknown Branch Hint>";
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, const std::array<uint8_t, 0x10>& Arg) {
|
||||
@@ -315,7 +359,8 @@ void Dump(fextl::stringstream* out, const IRListView* IR) {
|
||||
|
||||
++CurrentIndent;
|
||||
AddIndent();
|
||||
*out << fextl::fmt::format("(%0) IRHeader %{}, #{:#x}, #{}, #{}\n", HeaderOp->Blocks.ID(), HeaderOp->OriginalRIP, HeaderOp->BlockCount, HeaderOp->NumHostInstructions);
|
||||
*out << fextl::fmt::format("(%0) IRHeader %{}, #{:#x}, #{}, #{}\n", HeaderOp->Blocks.ID(), HeaderOp->OriginalRIP, HeaderOp->BlockCount,
|
||||
HeaderOp->NumHostInstructions);
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
|
||||
{
|
||||
@@ -352,17 +397,17 @@ void Dump(fextl::stringstream* out, const IRListView* IR) {
|
||||
|
||||
auto PhyReg = PhysicalRegister(CodeNode);
|
||||
if (!PhyReg.IsInvalid()) {
|
||||
switch (PhyReg.Class) {
|
||||
case FEXCore::IR::GPRClass.Val: *out << "(r"; break;
|
||||
case FEXCore::IR::GPRFixedClass.Val: *out << "(R"; break;
|
||||
case FEXCore::IR::FPRClass.Val: *out << "(v"; break;
|
||||
case FEXCore::IR::FPRFixedClass.Val: *out << "(V"; break;
|
||||
case FEXCore::IR::ComplexClass.Val: *out << "(complex"; break;
|
||||
case FEXCore::IR::InvalidClass.Val: *out << "(invalid"; break;
|
||||
switch (PhyReg.AsRegClass()) {
|
||||
case RegClass::GPR: *out << "(r"; break;
|
||||
case RegClass::GPRFixed: *out << "(R"; break;
|
||||
case RegClass::FPR: *out << "(v"; break;
|
||||
case RegClass::FPRFixed: *out << "(V"; break;
|
||||
case RegClass::Complex: *out << "(complex"; break;
|
||||
case RegClass::Invalid: *out << "(invalid"; break;
|
||||
default: *out << "(unknown"; break;
|
||||
}
|
||||
if (PhyReg.Class != FEXCore::IR::InvalidClass.Val) {
|
||||
*out << std::dec << (uint32_t)PhyReg.Reg << ")";
|
||||
if (PhyReg.AsRegClass() != RegClass::Invalid) {
|
||||
*out << std::dec << uint32_t(PhyReg.Reg) << ")";
|
||||
} else {
|
||||
*out << ")";
|
||||
}
|
||||
|
||||
@@ -33,14 +33,14 @@ bool IsBlockExit(FEXCore::IR::IROps Op) {
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::IR::RegisterClassType IREmitter::WalkFindRegClass(Ref Node) {
|
||||
RegClass IREmitter::WalkFindRegClass(Ref Node) {
|
||||
auto Class = GetOpRegClass(Node);
|
||||
switch (Class) {
|
||||
case GPRClass:
|
||||
case FPRClass:
|
||||
case GPRFixedClass:
|
||||
case FPRFixedClass:
|
||||
case InvalidClass: return Class;
|
||||
case RegClass::GPR:
|
||||
case RegClass::FPR:
|
||||
case RegClass::GPRFixed:
|
||||
case RegClass::FPRFixed:
|
||||
case RegClass::Invalid: return Class;
|
||||
default: break;
|
||||
}
|
||||
|
||||
@@ -82,7 +82,7 @@ FEXCore::IR::RegisterClassType IREmitter::WalkFindRegClass(Ref Node) {
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled op type: {} {} in argument class validation", ToUnderlying(IROp->Op), GetOpName(Node)); break;
|
||||
}
|
||||
return InvalidClass;
|
||||
return RegClass::Invalid;
|
||||
}
|
||||
|
||||
void IREmitter::ResetWorkingList() {
|
||||
|
||||
@@ -16,13 +16,8 @@
|
||||
#include <string.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class Pass;
|
||||
class PassManager;
|
||||
|
||||
class IREmitter {
|
||||
friend class FEXCore::IR::Pass;
|
||||
friend class FEXCore::IR::PassManager;
|
||||
|
||||
public:
|
||||
IREmitter(FEXCore::Utils::IntrusivePooledAllocator& ThreadAllocator, bool SupportsTSOImm9)
|
||||
: DualListData {ThreadAllocator, 8 * 1024 * 1024}
|
||||
@@ -51,12 +46,12 @@ public:
|
||||
*
|
||||
* @{ */
|
||||
|
||||
FEXCore::IR::RegisterClassType WalkFindRegClass(Ref Node);
|
||||
RegClass WalkFindRegClass(Ref Node);
|
||||
|
||||
// These inlining helpers are used by IRDefines.inc so define first.
|
||||
Ref InlineMem(OpSize Size, Ref Offset, MemOffsetType OffsetType, uint8_t& OffsetScale, bool TSO = false) {
|
||||
uint64_t Imm {};
|
||||
if (OffsetType != MEM_OFFSET_SXTX || !IsValueConstant(WrapNode(Offset), &Imm)) {
|
||||
if (OffsetType != MemOffsetType::SXTX || !IsValueConstant(WrapNode(Offset), &Imm)) {
|
||||
return Offset;
|
||||
}
|
||||
|
||||
@@ -113,36 +108,86 @@ public:
|
||||
IRPair<IROp_Jump> _Jump() {
|
||||
return _Jump(InvalidNode);
|
||||
}
|
||||
IRPair<IROp_CondJump> _CondJump(Ref ssa0, CondClassType cond = {COND_NEQ}) {
|
||||
IRPair<IROp_CondJump> _CondJump(Ref ssa0, CondClass cond = CondClass::NEQ) {
|
||||
return _CondJump(ssa0, _Constant(0), InvalidNode, InvalidNode, cond, GetOpSize(ssa0));
|
||||
}
|
||||
IRPair<IROp_CondJump> _CondJump(Ref ssa0, Ref ssa1, Ref ssa2, CondClassType cond = {COND_NEQ}) {
|
||||
IRPair<IROp_CondJump> _CondJump(Ref ssa0, Ref ssa1, Ref ssa2, CondClass cond = CondClass::NEQ) {
|
||||
return _CondJump(ssa0, _Constant(0), ssa1, ssa2, cond, GetOpSize(ssa0));
|
||||
}
|
||||
// TODO: Work to remove this implicit sized Select implementation.
|
||||
IRPair<IROp_Select> _Select(uint8_t Cond, Ref ssa0, Ref ssa1, Ref ssa2, Ref ssa3, IR::OpSize CompareSize = OpSize::iUnsized) {
|
||||
if (CompareSize == OpSize::iUnsized) {
|
||||
CompareSize = std::max(OpSize::i32Bit, std::max(GetOpSize(ssa0), GetOpSize(ssa1)));
|
||||
}
|
||||
|
||||
return _Select(std::max(OpSize::i32Bit, std::max(GetOpSize(ssa2), GetOpSize(ssa3))), CompareSize, CondClassType {Cond}, ssa0, ssa1, ssa2, ssa3);
|
||||
IRPair<IROp_LoadContext> _LoadContextGPR(OpSize ByteSize, uint32_t Offset) {
|
||||
return _LoadContext(ByteSize, RegClass::GPR, Offset);
|
||||
}
|
||||
IRPair<IROp_LoadMem> _LoadMem(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, Ref ssa0, IR::OpSize Align = OpSize::i8Bit) {
|
||||
return _LoadMem(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
IRPair<IROp_LoadContext> _LoadContextFPR(OpSize ByteSize, uint32_t Offset) {
|
||||
return _LoadContext(ByteSize, RegClass::FPR, Offset);
|
||||
}
|
||||
IRPair<IROp_StoreMem> _StoreMem(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, Ref Addr, Ref Value, IR::OpSize Align = OpSize::i8Bit) {
|
||||
return _StoreMem(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
IRPair<IROp_StoreContext> _StoreContextGPR(OpSize ByteSize, Ref Value, uint32_t Offset) {
|
||||
return _StoreContext(ByteSize, RegClass::GPR, Value, Offset);
|
||||
}
|
||||
IRPair<IROp_StoreContext> _StoreContextFPR(OpSize ByteSize, Ref Value, uint32_t Offset) {
|
||||
return _StoreContext(ByteSize, RegClass::FPR, Value, Offset);
|
||||
}
|
||||
|
||||
IRPair<IROp_Select> Select01(FEXCore::IR::OpSize CompareSize, CondClassType Cond, OrderedNode* Cmp1, OrderedNode* Cmp2) {
|
||||
IRPair<IROp_LoadContextIndexed> _LoadContextGPRIndexed(Ref Index, OpSize ByteSize, uint32_t BaseOffset, uint32_t Stride) {
|
||||
return _LoadContextIndexed(Index, ByteSize, BaseOffset, Stride, RegClass::GPR);
|
||||
}
|
||||
IRPair<IROp_LoadContextIndexed> _LoadContextFPRIndexed(Ref Index, OpSize ByteSize, uint32_t BaseOffset, uint32_t Stride) {
|
||||
return _LoadContextIndexed(Index, ByteSize, BaseOffset, Stride, RegClass::FPR);
|
||||
}
|
||||
IRPair<IROp_StoreContextIndexed> _StoreContextGPRIndexed(Ref Value, Ref Index, OpSize ByteSize, uint32_t BaseOffset, uint32_t Stride) {
|
||||
return _StoreContextIndexed(Value, Index, ByteSize, BaseOffset, Stride, RegClass::GPR);
|
||||
}
|
||||
IRPair<IROp_StoreContextIndexed> _StoreContextFPRIndexed(Ref Value, Ref Index, OpSize ByteSize, uint32_t BaseOffset, uint32_t Stride) {
|
||||
return _StoreContextIndexed(Value, Index, ByteSize, BaseOffset, Stride, RegClass::FPR);
|
||||
}
|
||||
|
||||
IRPair<IROp_LoadMem> _LoadMem(RegClass Class, OpSize Size, Ref ssa0, OpSize Align = OpSize::i8Bit) {
|
||||
return _LoadMem(Class, Size, ssa0, Invalid(), Align, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
IRPair<IROp_LoadMem> _LoadMemGPR(OpSize Size, Ref ssa0, OpSize Align = OpSize::i8Bit) {
|
||||
return _LoadMem(RegClass::GPR, Size, ssa0, Invalid(), Align, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
IRPair<IROp_LoadMem> _LoadMemGPR(OpSize Size, Ref Addr, Ref Offset, OpSize Align, MemOffsetType OffsetType, uint8_t OffsetScale) {
|
||||
return _LoadMem(RegClass::GPR, Size, Addr, Offset, Align, OffsetType, OffsetScale);
|
||||
}
|
||||
IRPair<IROp_LoadMem> _LoadMemFPR(OpSize Size, Ref ssa0, OpSize Align = OpSize::i8Bit) {
|
||||
return _LoadMem(RegClass::FPR, Size, ssa0, Invalid(), Align, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
IRPair<IROp_LoadMem> _LoadMemFPR(OpSize Size, Ref Addr, Ref Offset, OpSize Align, MemOffsetType OffsetType, uint8_t OffsetScale) {
|
||||
return _LoadMem(RegClass::FPR, Size, Addr, Offset, Align, OffsetType, OffsetScale);
|
||||
}
|
||||
IRPair<IROp_StoreMem> _StoreMem(RegClass Class, OpSize Size, Ref Addr, Ref Value, OpSize Align = OpSize::i8Bit) {
|
||||
return _StoreMem(Class, Size, Value, Addr, Invalid(), Align, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
IRPair<IROp_StoreMem> _StoreMemGPR(OpSize Size, Ref Addr, Ref Value, OpSize Align = OpSize::i8Bit) {
|
||||
return _StoreMem(RegClass::GPR, Size, Value, Addr, Invalid(), Align, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
IRPair<IROp_StoreMem> _StoreMemGPR(OpSize Size, Ref Value, Ref Addr, Ref Offset, OpSize Align, MemOffsetType OffsetType, uint8_t OffsetScale) {
|
||||
return _StoreMem(RegClass::GPR, Size, Value, Addr, Offset, Align, OffsetType, OffsetScale);
|
||||
}
|
||||
IRPair<IROp_StoreMem> _StoreMemFPR(OpSize Size, Ref Addr, Ref Value, OpSize Align = OpSize::i8Bit) {
|
||||
return _StoreMem(RegClass::FPR, Size, Value, Addr, Invalid(), Align, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
IRPair<IROp_StoreMem> _StoreMemFPR(OpSize Size, Ref Value, Ref Addr, Ref Offset, OpSize Align, MemOffsetType OffsetType, uint8_t OffsetScale) {
|
||||
return _StoreMem(RegClass::FPR, Size, Value, Addr, Offset, Align, OffsetType, OffsetScale);
|
||||
}
|
||||
|
||||
IRPair<IROp_StoreMemPair> _StoreMemPairGPR(OpSize Size, Ref Value1, Ref Value2, Ref Addr, uint32_t Offset) {
|
||||
return _StoreMemPair(RegClass::GPR, Size, Value1, Value2, Addr, Offset);
|
||||
}
|
||||
IRPair<IROp_StoreMemPair> _StoreMemPairFPR(OpSize Size, Ref Value1, Ref Value2, Ref Addr, uint32_t Offset) {
|
||||
return _StoreMemPair(RegClass::FPR, Size, Value1, Value2, Addr, Offset);
|
||||
}
|
||||
|
||||
IRPair<IROp_Select> Select01(FEXCore::IR::OpSize CompareSize, CondClass Cond, OrderedNode* Cmp1, OrderedNode* Cmp2) {
|
||||
return _Select(OpSize::i64Bit, CompareSize, Cond, Cmp1, Cmp2, _InlineConstant(1), _InlineConstant(0));
|
||||
}
|
||||
|
||||
IRPair<IROp_Select> To01(FEXCore::IR::OpSize CompareSize, OrderedNode* Cmp1) {
|
||||
return Select01(CompareSize, CondClassType {COND_NEQ}, Cmp1, Constant(0));
|
||||
return Select01(CompareSize, CondClass::NEQ, Cmp1, Constant(0));
|
||||
}
|
||||
|
||||
IRPair<IROp_NZCVSelect> _NZCVSelect01(CondClassType Cond) {
|
||||
IRPair<IROp_NZCVSelect> _NZCVSelect01(CondClass Cond) {
|
||||
return _NZCVSelect(OpSize::i64Bit, Cond, _InlineConstant(1), _InlineConstant(0));
|
||||
}
|
||||
|
||||
@@ -255,7 +300,7 @@ public:
|
||||
}
|
||||
|
||||
/** @} */
|
||||
FEXCore::IR::RegisterClassType WalkFindRegClass(OrderedNodeWrapper ssa) {
|
||||
RegClass WalkFindRegClass(OrderedNodeWrapper ssa) {
|
||||
Ref RealNode = ssa.GetNode(DualListData.ListBegin());
|
||||
return WalkFindRegClass(RealNode);
|
||||
}
|
||||
|
||||
@@ -93,11 +93,11 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
// After RA, the destination needs to be assigned a register and class
|
||||
auto PhyReg = PhysicalRegister(CodeNode);
|
||||
|
||||
FEXCore::IR::RegisterClassType ExpectedClass = IR::GetRegClass(IROp->Op);
|
||||
FEXCore::IR::RegisterClassType AssignedClass = FEXCore::IR::RegisterClassType {PhyReg.Class};
|
||||
const auto ExpectedClass = IR::GetRegClass(IROp->Op);
|
||||
const auto AssignedClass = PhyReg.AsRegClass();
|
||||
|
||||
// If no register class was assigned
|
||||
if (AssignedClass == IR::InvalidClass) {
|
||||
if (AssignedClass == IR::RegClass::Invalid) {
|
||||
HadError |= true;
|
||||
Errors << "%" << ID << ": Had destination but with no register class assigned" << std::endl;
|
||||
}
|
||||
@@ -109,10 +109,10 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
}
|
||||
|
||||
// Assigned class wasn't the expected class and it is a non-complex op
|
||||
if (AssignedClass != ExpectedClass && ExpectedClass != IR::ComplexClass) {
|
||||
if (AssignedClass != ExpectedClass && ExpectedClass != IR::RegClass::Complex) {
|
||||
HadWarning |= true;
|
||||
Warnings << "%" << ID << ": Destination had register class " << AssignedClass.Val << " When register class "
|
||||
<< ExpectedClass.Val << " Was expected" << std::endl;
|
||||
Warnings << "%" << ID << ": Destination had register class " << uint32_t(AssignedClass) << " When register class "
|
||||
<< uint32_t(ExpectedClass) << " Was expected" << std::endl;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2,22 +2,21 @@
|
||||
/*
|
||||
$info$
|
||||
tags: ir|opts
|
||||
desc: This is not used right now, possibly broken
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "FEXCore/Core/X86Enums.h"
|
||||
#include "FEXCore/Utils/CompilerDefs.h"
|
||||
#include "FEXCore/Utils/MathUtils.h"
|
||||
#include "FEXCore/fextl/deque.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/fextl/deque.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
// Flag bit flags
|
||||
#define FLAG_V (1U << 0)
|
||||
#define FLAG_C (1U << 1)
|
||||
@@ -62,36 +61,36 @@ struct FlagInfo {
|
||||
return {.Raw = R};
|
||||
}
|
||||
|
||||
bool Trivial() {
|
||||
bool Trivial() const {
|
||||
return Raw == 0;
|
||||
}
|
||||
|
||||
unsigned Read() {
|
||||
unsigned Read() const {
|
||||
return Bits(0, 8);
|
||||
}
|
||||
|
||||
unsigned Write() {
|
||||
unsigned Write() const {
|
||||
return Bits(8, 8);
|
||||
}
|
||||
|
||||
bool CanEliminate() {
|
||||
bool CanEliminate() const {
|
||||
return Bits(16, 1);
|
||||
}
|
||||
|
||||
bool Special() {
|
||||
bool Special() const {
|
||||
return Bits(63, 1);
|
||||
}
|
||||
|
||||
IROps Replacement() {
|
||||
IROps Replacement() const {
|
||||
return (IROps)Bits(32, 16);
|
||||
}
|
||||
|
||||
IROps ReplacementNoWrite() {
|
||||
IROps ReplacementNoWrite() const {
|
||||
return (IROps)Bits(48, 16);
|
||||
}
|
||||
|
||||
private:
|
||||
unsigned Bits(unsigned Start, unsigned Count) {
|
||||
unsigned Bits(unsigned Start, unsigned Count) const {
|
||||
return (Raw >> Start) & ((1u << Count) - 1);
|
||||
}
|
||||
};
|
||||
@@ -154,45 +153,44 @@ public:
|
||||
|
||||
private:
|
||||
FlagInfo Classify(IROp_Header* Node);
|
||||
unsigned FlagForReg(unsigned Reg);
|
||||
unsigned FlagsForCondClassType(CondClassType Cond);
|
||||
unsigned FlagsForCondClassType(CondClass Cond);
|
||||
bool EliminateDeadCode(IREmitter* IREmit, Ref CodeNode, IROp_Header* IROp);
|
||||
void FoldBranch(IREmitter* IREmit, IRListView& CurrentIR, IROp_CondJump* Op, Ref CodeNode);
|
||||
CondClassType X86ToArmFloatCond(CondClassType X86);
|
||||
CondClass X86ToArmFloatCond(CondClass X86);
|
||||
bool ProcessBlock(IREmitter* IREmit, IRListView& CurrentIR, Ref Block, ControlFlowGraph& CFG);
|
||||
void OptimizeParity(IREmitter* IREmit, IRListView& CurrentIR, ControlFlowGraph& CFG);
|
||||
};
|
||||
|
||||
unsigned DeadFlagCalculationEliminination::FlagsForCondClassType(CondClassType Cond) {
|
||||
unsigned DeadFlagCalculationEliminination::FlagsForCondClassType(CondClass Cond) {
|
||||
switch (Cond) {
|
||||
case COND_AL: return 0;
|
||||
case CondClass::AL: return 0;
|
||||
|
||||
case COND_MI:
|
||||
case COND_PL: return FLAG_N;
|
||||
case CondClass::MI:
|
||||
case CondClass::PL: return FLAG_N;
|
||||
|
||||
case COND_EQ:
|
||||
case COND_NEQ: return FLAG_Z;
|
||||
case CondClass::EQ:
|
||||
case CondClass::NEQ: return FLAG_Z;
|
||||
|
||||
case COND_UGE:
|
||||
case COND_ULT: return FLAG_C;
|
||||
case CondClass::UGE:
|
||||
case CondClass::ULT: return FLAG_C;
|
||||
|
||||
case COND_VS:
|
||||
case COND_VC:
|
||||
case COND_FU:
|
||||
case COND_FNU: return FLAG_V;
|
||||
case CondClass::VS:
|
||||
case CondClass::VC:
|
||||
case CondClass::FU:
|
||||
case CondClass::FNU: return FLAG_V;
|
||||
|
||||
case COND_UGT:
|
||||
case COND_ULE: return FLAG_Z | FLAG_C;
|
||||
case CondClass::UGT:
|
||||
case CondClass::ULE: return FLAG_Z | FLAG_C;
|
||||
|
||||
case COND_SGE:
|
||||
case COND_SLT:
|
||||
case COND_FLU:
|
||||
case COND_FGE: return FLAG_N | FLAG_V;
|
||||
case CondClass::SGE:
|
||||
case CondClass::SLT:
|
||||
case CondClass::FLU:
|
||||
case CondClass::FGE: return FLAG_N | FLAG_V;
|
||||
|
||||
case COND_SGT:
|
||||
case COND_SLE:
|
||||
case COND_FLEU:
|
||||
case COND_FGT: return FLAG_N | FLAG_Z | FLAG_V;
|
||||
case CondClass::SGT:
|
||||
case CondClass::SLE:
|
||||
case CondClass::FLEU:
|
||||
case CondClass::FGT: return FLAG_N | FLAG_Z | FLAG_V;
|
||||
|
||||
default: LOGMAN_THROW_A_FMT(false, "unknown cond class type"); return FLAG_NZCV;
|
||||
}
|
||||
@@ -456,7 +454,7 @@ bool DeadFlagCalculationEliminination::EliminateDeadCode(IREmitter* IREmit, Ref
|
||||
return true;
|
||||
}
|
||||
|
||||
CondClassType DeadFlagCalculationEliminination::X86ToArmFloatCond(CondClassType X86) {
|
||||
CondClass DeadFlagCalculationEliminination::X86ToArmFloatCond(CondClass X86) {
|
||||
// Table of x86 condition codes that map to arm64 condition codes, in the
|
||||
// sense that fcmp+axflag+branch(x86) is equivalent to fcmp+branch(arm).
|
||||
//
|
||||
@@ -465,12 +463,12 @@ CondClassType DeadFlagCalculationEliminination::X86ToArmFloatCond(CondClassType
|
||||
//
|
||||
// SF/OF conditions are trivial and therefore shouldn't actually be generated
|
||||
switch (X86) {
|
||||
case COND_UGE /* A */: return {COND_FGE} /* GE */;
|
||||
case COND_UGT /* AE */: return {COND_FGT} /* GT */;
|
||||
case COND_ULT /* B */: return {COND_SLT} /* LT */;
|
||||
case COND_ULE /* BE */: return {COND_SLE} /* LE */;
|
||||
case COND_SLE /* LE */: return {COND_SLE} /* LE */;
|
||||
default: return {COND_AL};
|
||||
case CondClass::UGE /* A */: return CondClass::FGE /* GE */;
|
||||
case CondClass::UGT /* AE */: return CondClass::FGT /* GT */;
|
||||
case CondClass::ULT /* B */: return CondClass::SLT /* LT */;
|
||||
case CondClass::ULE /* BE */: return CondClass::SLE /* LE */;
|
||||
case CondClass::SLE /* LE */: return CondClass::SLE /* LE */;
|
||||
default: return CondClass::AL;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -485,8 +483,8 @@ void DeadFlagCalculationEliminination::FoldBranch(IREmitter* IREmit, IRListView&
|
||||
auto Prev = CurrentIR.GetOp<IR::IROp_Header>(PrevWrap);
|
||||
if (Prev->Op == OP_AXFLAG) {
|
||||
// Pattern match a branch fed by AXFLAG.
|
||||
CondClassType ArmCond = X86ToArmFloatCond(Op->Cond);
|
||||
if (ArmCond == COND_AL) {
|
||||
CondClass ArmCond = X86ToArmFloatCond(Op->Cond);
|
||||
if (ArmCond == CondClass::AL) {
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -495,7 +493,7 @@ void DeadFlagCalculationEliminination::FoldBranch(IREmitter* IREmit, IRListView&
|
||||
// Pattern match a branch fed by a compare. We could also handle bit tests
|
||||
// here, but tbz/tbnz has a limited offset range which we don't have a way to
|
||||
// deal with yet. Let's hope that's not a big deal.
|
||||
if (!(Op->Cond == COND_NEQ || Op->Cond == COND_EQ) || (Prev->Size < OpSize::i32Bit)) {
|
||||
if (!(Op->Cond == CondClass::NEQ || Op->Cond == CondClass::EQ) || (Prev->Size < OpSize::i32Bit)) {
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -629,9 +627,9 @@ void DeadFlagCalculationEliminination::OptimizeParity(IREmitter* IREmit, IRListV
|
||||
}
|
||||
|
||||
for (auto [Block, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
auto ID = BlockHeader->C<IROp_CodeBlock>()->ID;
|
||||
const auto ID = BlockHeader->C<IROp_CodeBlock>()->ID;
|
||||
const auto& Predecessors = CFG.Get(ID)->Predecessors;
|
||||
bool Full = false;
|
||||
auto Predecessors = CFG.Get(ID)->Predecessors;
|
||||
|
||||
if (Predecessors.empty()) {
|
||||
// Conservatively assume there was full parity before the start block
|
||||
|
||||
@@ -12,6 +12,7 @@ $end_info$
|
||||
#include "Interface/IR/Passes.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
@@ -22,7 +23,7 @@ using namespace FEXCore;
|
||||
|
||||
namespace FEXCore::IR {
|
||||
namespace {
|
||||
struct RegisterClass {
|
||||
struct RegisterClassData {
|
||||
uint32_t Available;
|
||||
uint32_t Count;
|
||||
|
||||
@@ -32,9 +33,9 @@ namespace {
|
||||
Ref RegToSSA[32];
|
||||
};
|
||||
|
||||
IR::RegisterClassType GetRegClassFromNode(IR::IRListView* IR, IR::IROp_Header* IROp) {
|
||||
IR::RegisterClassType Class = IR::GetRegClass(IROp->Op);
|
||||
if (Class != IR::ComplexClass) {
|
||||
IR::RegClass GetRegClassFromNode(IR::IRListView* IR, IR::IROp_Header* IROp) {
|
||||
const auto Class = IR::GetRegClass(IROp->Op);
|
||||
if (Class != IR::RegClass::Complex) {
|
||||
return Class;
|
||||
}
|
||||
|
||||
@@ -46,7 +47,7 @@ namespace {
|
||||
case IR::OP_LOADMEM:
|
||||
case IR::OP_LOADMEMTSO: return IROp->C<IR::IROp_LoadMem>()->Class;
|
||||
case IR::OP_FILLREGISTER: return IROp->C<IR::IROp_FillRegister>()->Class;
|
||||
default: return IR::InvalidClass;
|
||||
default: return IR::RegClass::Invalid;
|
||||
}
|
||||
};
|
||||
} // Anonymous namespace
|
||||
@@ -56,11 +57,11 @@ public:
|
||||
explicit ConstrainedRAPass(const FEXCore::CPUIDEmu* CPUID)
|
||||
: CPUID {CPUID} {}
|
||||
void Run(IREmitter* IREmit) override;
|
||||
void AddRegisters(IR::RegisterClassType Class, uint32_t RegisterCount) override;
|
||||
void AddRegisters(IR::RegClass Class, uint32_t RegisterCount) override;
|
||||
bool TryPostRAMerge(Ref LastNode, Ref CodeNode, IROp_Header* IROp);
|
||||
|
||||
private:
|
||||
RegisterClass Classes[IR::NumClasses];
|
||||
RegisterClassData Classes[IR::NumClasses];
|
||||
|
||||
IREmitter* IREmit;
|
||||
IRListView* IR;
|
||||
@@ -101,7 +102,7 @@ private:
|
||||
uint32_t SlotPlusOne = SpillSlots[IR->GetID(Node).Value];
|
||||
LOGMAN_THROW_A_FMT(SlotPlusOne >= 1, "Node must have been spilled");
|
||||
|
||||
RegisterClassType RegClass = GetRegClassFromNode(IR, IROp);
|
||||
const auto RegClass = GetRegClassFromNode(IR, IROp);
|
||||
return IREmit->_FillRegister(IROp->Size, IROp->ElementSize, SlotPlusOne - 1, RegClass);
|
||||
};
|
||||
|
||||
@@ -120,7 +121,7 @@ private:
|
||||
return Op != OP_INLINECONSTANT && Op != OP_INLINEENTRYPOINTOFFSET;
|
||||
};
|
||||
|
||||
RegisterClass* GetClass(PhysicalRegister Reg) {
|
||||
RegisterClassData* GetClass(PhysicalRegister Reg) {
|
||||
return &Classes[Reg.Class];
|
||||
};
|
||||
|
||||
@@ -133,13 +134,13 @@ private:
|
||||
LOGMAN_THROW_A_FMT(ID < SSAToReg.size(), "Only old nodes looked up");
|
||||
|
||||
PhysicalRegister Reg = SSAToReg[ID];
|
||||
RegisterClass* Class = GetClass(Reg);
|
||||
RegisterClassData* Class = GetClass(Reg);
|
||||
|
||||
return (Class->Available & GetRegBits(Reg)) == 0 && Class->RegToSSA[Reg.Reg] == Node;
|
||||
};
|
||||
|
||||
void FreeReg(PhysicalRegister Reg) {
|
||||
RegisterClass* Class = GetClass(Reg);
|
||||
RegisterClassData* Class = GetClass(Reg);
|
||||
uint32_t RegBits = GetRegBits(Reg);
|
||||
|
||||
LOGMAN_THROW_A_FMT(!(Class->Available & RegBits), "Register double-free");
|
||||
@@ -187,22 +188,22 @@ private:
|
||||
};
|
||||
|
||||
PhysicalRegister DecodeSRAReg(const IROp_Header* IROp, Ref Node) {
|
||||
uint8_t FlagOffset = Classes[GPRFixedClass.Val].Count - 2;
|
||||
uint8_t FlagOffset = Classes[FEXCore::ToUnderlying(RegClass::GPRFixed)].Count - 2;
|
||||
|
||||
if (IROp->Op == OP_STOREREGISTER) {
|
||||
return PhysicalRegister(Node);
|
||||
} else if (IROp->Op == OP_LOADPF || IROp->Op == OP_STOREPF) {
|
||||
return PhysicalRegister {GPRFixedClass, FlagOffset};
|
||||
return PhysicalRegister {RegClass::GPRFixed, FlagOffset};
|
||||
} else if (IROp->Op == OP_LOADAF || IROp->Op == OP_STOREAF) {
|
||||
return PhysicalRegister {GPRFixedClass, (uint8_t)(FlagOffset + 1)};
|
||||
return PhysicalRegister {RegClass::GPRFixed, uint8_t(FlagOffset + 1)};
|
||||
} else {
|
||||
const IROp_LoadRegister* Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
|
||||
LOGMAN_THROW_A_FMT(Op->Class == GPRClass || Op->Class == FPRClass, "SRA classes");
|
||||
if (Op->Class == FPRClass) {
|
||||
return PhysicalRegister {FPRFixedClass, (uint8_t)Op->Reg};
|
||||
LOGMAN_THROW_A_FMT(Op->Class == RegClass::GPR || Op->Class == RegClass::FPR, "SRA classes");
|
||||
if (Op->Class == RegClass::FPR) {
|
||||
return PhysicalRegister {RegClass::FPRFixed, uint8_t(Op->Reg)};
|
||||
} else {
|
||||
return PhysicalRegister {GPRFixedClass, (uint8_t)Op->Reg};
|
||||
return PhysicalRegister {RegClass::GPRFixed, uint8_t(Op->Reg)};
|
||||
}
|
||||
}
|
||||
};
|
||||
@@ -267,7 +268,7 @@ private:
|
||||
SourceIndex = SourcesNextUses.size();
|
||||
}
|
||||
|
||||
void SpillReg(RegisterClass* Class, IROp_CodeBlock* Block, IROp_Header* Exclude) {
|
||||
void SpillReg(RegisterClassData* Class, IROp_CodeBlock* Block, IROp_Header* Exclude) {
|
||||
// We're about to use next-use information, so calculate it.
|
||||
if (!AnySpilled) {
|
||||
CalculateNextUses(Block, Exclude);
|
||||
@@ -318,7 +319,7 @@ private:
|
||||
// If we already spilled the Candidate, we don't need to spill again.
|
||||
// Similarly, if we can rematerialize the instruction, we don't spill it.
|
||||
if (!Spilled && Header->Op != OP_CONSTANT) {
|
||||
LOGMAN_THROW_A_FMT(Reg.Class == GetRegClassFromNode(IR, Header), "Consistent");
|
||||
LOGMAN_THROW_A_FMT(Reg.AsRegClass() == GetRegClassFromNode(IR, Header), "Consistent");
|
||||
|
||||
// SpillSlots allocation is deferred.
|
||||
if (SpillSlots.empty()) {
|
||||
@@ -329,7 +330,7 @@ private:
|
||||
uint32_t Slot = IR->GetHeader()->SpillSlots++;
|
||||
|
||||
// We must map here in case we're spilling something we shuffled.
|
||||
auto SpillOp = IREmit->_SpillRegister(OrderedNodeWrapper::FromImmediate(Reg.Raw), Slot, RegisterClassType {Reg.Class});
|
||||
auto SpillOp = IREmit->_SpillRegister(OrderedNodeWrapper::FromImmediate(Reg.Raw), Slot, Reg.AsRegClass());
|
||||
SpillOp.first->Header.Size = Header->Size;
|
||||
SpillOp.first->Header.ElementSize = Header->ElementSize;
|
||||
SpillSlots[Value] = Slot + 1;
|
||||
@@ -341,7 +342,7 @@ private:
|
||||
};
|
||||
|
||||
void RemapReg(Ref Node, PhysicalRegister Reg) {
|
||||
RegisterClass* Class = GetClass(Reg);
|
||||
RegisterClassData* Class = GetClass(Reg);
|
||||
Class->RegToSSA[Reg.Reg] = Node;
|
||||
|
||||
uint32_t Index = IR->GetID(Node).Value;
|
||||
@@ -352,7 +353,7 @@ private:
|
||||
|
||||
// Record a given assignment of register Reg to Node.
|
||||
void SetReg(Ref Node, PhysicalRegister Reg) {
|
||||
RegisterClass* Class = GetClass(Reg);
|
||||
RegisterClassData* Class = GetClass(Reg);
|
||||
uint32_t RegBits = GetRegBits(Reg);
|
||||
|
||||
LOGMAN_THROW_A_FMT((Class->Available & RegBits) == RegBits, "Precondition");
|
||||
@@ -370,7 +371,7 @@ private:
|
||||
// Prioritize preferred registers.
|
||||
if (Node < PreferredReg.size()) {
|
||||
if (PhysicalRegister Reg = PreferredReg[Node]; !Reg.IsInvalid()) {
|
||||
RegisterClass* Class = GetClass(Reg);
|
||||
RegisterClassData* Class = GetClass(Reg);
|
||||
uint32_t RegBits = GetRegBits(Reg);
|
||||
|
||||
if ((Class->Available & RegBits) == RegBits) {
|
||||
@@ -383,10 +384,10 @@ private:
|
||||
// Try to handle tied registers. This can fail, the JIT will insert moves.
|
||||
if (int TiedIdx = IR::TiedSource(IROp->Op); TiedIdx >= 0) {
|
||||
auto Reg = PhysicalRegister(IROp->Args[TiedIdx]);
|
||||
RegisterClass* Class = GetClass(Reg);
|
||||
RegisterClassData* Class = GetClass(Reg);
|
||||
uint32_t RegBits = GetRegBits(Reg);
|
||||
|
||||
if (Reg.Class != GPRFixedClass && Reg.Class != FPRFixedClass && (Class->Available & RegBits) == RegBits) {
|
||||
if (Reg.AsRegClass() != RegClass::GPRFixed && Reg.AsRegClass() != RegClass::FPRFixed && (Class->Available & RegBits) == RegBits) {
|
||||
SetReg(CodeNode, Reg);
|
||||
return;
|
||||
}
|
||||
@@ -394,7 +395,7 @@ private:
|
||||
|
||||
// Try to coalesce reserved pairs. Just a heuristic to remove some moves.
|
||||
if (IROp->Op == OP_ALLOCATEGPR && IROp->C<IROp_AllocateGPR>()->ForPair) {
|
||||
uint32_t Available = Classes[GPRClass].Available;
|
||||
uint32_t Available = Classes[FEXCore::ToUnderlying(RegClass::GPR)].Available;
|
||||
|
||||
// Only choose base register R if R and R + 1 are both free
|
||||
Available &= (Available >> 1);
|
||||
@@ -405,20 +406,20 @@ private:
|
||||
|
||||
if (Available) {
|
||||
unsigned Reg = std::countr_zero(Available);
|
||||
SetReg(CodeNode, PhysicalRegister(GPRClass, Reg));
|
||||
SetReg(CodeNode, PhysicalRegister(RegClass::GPR, Reg));
|
||||
return;
|
||||
}
|
||||
} else if (IROp->Op == OP_ALLOCATEGPRAFTER) {
|
||||
uint32_t Available = Classes[GPRClass].Available;
|
||||
uint32_t Available = Classes[FEXCore::ToUnderlying(RegClass::GPR)].Available;
|
||||
auto After = PhysicalRegister(IROp->Args[0]);
|
||||
if ((After.Reg & 1) == 0 && Available & (1ull << (After.Reg + 1))) {
|
||||
SetReg(CodeNode, PhysicalRegister(GPRClass, After.Reg + 1));
|
||||
SetReg(CodeNode, PhysicalRegister(RegClass::GPR, After.Reg + 1));
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
RegisterClassType ClassType = GetRegClassFromNode(IR, IROp);
|
||||
RegisterClass* Class = &Classes[ClassType];
|
||||
RegClass ClassType = GetRegClassFromNode(IR, IROp);
|
||||
RegisterClassData* Class = &Classes[FEXCore::ToUnderlying(ClassType)];
|
||||
|
||||
// Spill to make room in the register file.
|
||||
if (!Class->Available) {
|
||||
@@ -433,10 +434,10 @@ private:
|
||||
};
|
||||
};
|
||||
|
||||
void ConstrainedRAPass::AddRegisters(IR::RegisterClassType Class, uint32_t RegisterCount) {
|
||||
void ConstrainedRAPass::AddRegisters(IR::RegClass Class, uint32_t RegisterCount) {
|
||||
LOGMAN_THROW_A_FMT(RegisterCount <= 31, "Up to 31 regs supported");
|
||||
|
||||
Classes[Class].Count = RegisterCount;
|
||||
Classes[FEXCore::ToUnderlying(Class)].Count = RegisterCount;
|
||||
}
|
||||
|
||||
inline bool KillMove(IROp_Header* LastOp, IROp_Header* IROp, Ref LastNode, Ref CodeNode) {
|
||||
@@ -530,7 +531,7 @@ bool ConstrainedRAPass::TryPostRAMerge(Ref LastNode, Ref CodeNode, IROp_Header*
|
||||
const auto Result = CPUID->RunFunction(ConstantFunction, 0 /* leaf */);
|
||||
|
||||
IREmit->SetWriteCursorBefore(CodeNode);
|
||||
IREmit->_Fence({FEXCore::IR::Fence_Inst});
|
||||
IREmit->_Fence(IR::FenceType::Inst);
|
||||
IREmit->_Constant(Result.eax).Node->Reg = PhysicalRegister(Op->OutEAX).Raw;
|
||||
IREmit->_Constant(Result.ebx).Node->Reg = PhysicalRegister(Op->OutEBX).Raw;
|
||||
IREmit->_Constant(Result.ecx).Node->Reg = PhysicalRegister(Op->OutECX).Raw;
|
||||
@@ -663,7 +664,7 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
// Static registers must be consistent at SRA load/store. Evict to ensure.
|
||||
if (auto Node = DecodeSRANode(IROp, CodeNode); Node != nullptr) {
|
||||
auto Reg = DecodeSRAReg(IROp, CodeNode);
|
||||
RegisterClass* Class = &Classes[Reg.Class];
|
||||
RegisterClassData* Class = &Classes[Reg.Class];
|
||||
|
||||
if (!(Class->Available & (1u << Reg.Reg))) {
|
||||
Ref Old = Class->RegToSSA[Reg.Reg];
|
||||
@@ -678,7 +679,7 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
|
||||
Ref Copy;
|
||||
|
||||
if (Reg.Class == FPRFixedClass) {
|
||||
if (Reg.AsRegClass() == RegClass::FPRFixed) {
|
||||
IROp_Header* Header = IR->GetOp<IROp_Header>(Old);
|
||||
Copy = IREmit->_VMov(Header->Size, OrderedNodeWrapper::FromImmediate(Reg.Raw));
|
||||
} else {
|
||||
|
||||
@@ -12,11 +12,11 @@ $end_info$
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
struct RegisterClassType;
|
||||
enum class RegClass : uint32_t;
|
||||
|
||||
class RegisterAllocationPass : public FEXCore::IR::Pass {
|
||||
public:
|
||||
virtual void AddRegisters(FEXCore::IR::RegisterClassType Class, uint32_t RegisterCount) = 0;
|
||||
virtual void AddRegisters(RegClass Class, uint32_t RegisterCount) = 0;
|
||||
|
||||
// Number of GPRs usable for pairs at start of GPR set. Must be even.
|
||||
uint32_t PairRegs;
|
||||
|
||||
@@ -158,7 +158,9 @@ public:
|
||||
: Features(Features)
|
||||
, GPROpSize(GPROpSize) {
|
||||
FEX_CONFIG_OPT(ReducedPrecision, X87REDUCEDPRECISION);
|
||||
FEX_CONFIG_OPT(StrictReducedPrecision, X87STRICTREDUCEDPRECISION);
|
||||
ReducedPrecisionMode = ReducedPrecision;
|
||||
StrictReducedPrecisionMode = StrictReducedPrecision;
|
||||
}
|
||||
void Run(IREmitter* Emit) override;
|
||||
|
||||
@@ -166,10 +168,12 @@ private:
|
||||
const FEXCore::HostFeatures& Features;
|
||||
const OpSize GPROpSize;
|
||||
bool ReducedPrecisionMode;
|
||||
bool StrictReducedPrecisionMode;
|
||||
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
|
||||
|
||||
// Helpers
|
||||
Ref RotateRight8(uint32_t V, Ref Amount);
|
||||
Ref SilenceNaN(Ref Value);
|
||||
|
||||
void F80SplitStore_Helper(const IROp_StoreStackMem* Op, Ref StackNode) {
|
||||
Ref AddrNode = IR->GetNode(Op->Addr);
|
||||
@@ -178,18 +182,18 @@ private:
|
||||
MemOffsetType OffsetType = Op->OffsetType;
|
||||
uint8_t OffsetScale = Op->OffsetScale;
|
||||
|
||||
IREmit->_StoreMem(FPRClass, OpSize::i64Bit, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
IREmit->_StoreMemFPR(OpSize::i64Bit, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
auto Upper = IREmit->_VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, StackNode, 1);
|
||||
|
||||
// Store the Upper part of the register (the remaining 2 bytes) into memory.
|
||||
AddressMode A {.Base = AddrNode,
|
||||
.Index = Op->Offset.IsInvalid() ? nullptr : Offset,
|
||||
.IndexType = MEM_OFFSET_SXTX,
|
||||
.IndexScale = OffsetScale,
|
||||
.Offset = 8,
|
||||
.IndexType = MemOffsetType::SXTX,
|
||||
.IndexScale = OffsetScale,
|
||||
.AddrSize = OpSize::i64Bit};
|
||||
A = SelectAddressMode(IREmit, A, GPROpSize, Features.SupportsTSOImm9, false, false, OpSize::i16Bit);
|
||||
IREmit->_StoreMem(GPRClass, OpSize::i16Bit, Upper, A.Base, A.Index, OpSize::i64Bit, MEM_OFFSET_SXTX, A.IndexScale);
|
||||
IREmit->_StoreMemGPR(OpSize::i16Bit, Upper, A.Base, A.Index, OpSize::i64Bit, MemOffsetType::SXTX, A.IndexScale);
|
||||
}
|
||||
|
||||
void StoreStackMem_Helper(const IROp_StoreStackMem* Op, Ref StackNode) {
|
||||
@@ -204,7 +208,10 @@ private:
|
||||
case OpSize::i32Bit:
|
||||
case OpSize::i64Bit: {
|
||||
StackNode = IREmit->_F80CVT(Op->StoreSize, StackNode);
|
||||
IREmit->_StoreMem(FPRClass, Op->StoreSize, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
if (!ReducedPrecisionMode || StrictReducedPrecisionMode) {
|
||||
StackNode = SilenceNaN(StackNode);
|
||||
}
|
||||
IREmit->_StoreMemFPR(Op->StoreSize, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -212,7 +219,7 @@ private:
|
||||
if (Features.SupportsSVE128 || Features.SupportsSVE256) {
|
||||
AddressMode A {.Base = AddrNode,
|
||||
.Index = Op->Offset.IsInvalid() ? nullptr : Offset,
|
||||
.IndexType = MEM_OFFSET_SXTX,
|
||||
.IndexType = MemOffsetType::SXTX,
|
||||
.IndexScale = OffsetScale,
|
||||
.AddrSize = OpSize::i64Bit};
|
||||
AddrNode = LoadEffectiveAddress(IREmit, A, GPROpSize, false);
|
||||
@@ -235,13 +242,17 @@ private:
|
||||
MemOffsetType OffsetType = Op->OffsetType;
|
||||
uint8_t OffsetScale = Op->OffsetScale;
|
||||
|
||||
if ((!ReducedPrecisionMode || StrictReducedPrecisionMode) && Op->StoreSize != OpSize::f80Bit) {
|
||||
StackNode = SilenceNaN(StackNode);
|
||||
}
|
||||
|
||||
switch (Op->StoreSize) {
|
||||
case OpSize::i32Bit: {
|
||||
StackNode = IREmit->_Float_FToF(OpSize::i32Bit, OpSize::i64Bit, StackNode);
|
||||
[[fallthrough]];
|
||||
}
|
||||
case OpSize::i64Bit: {
|
||||
IREmit->_StoreMem(FPRClass, Op->StoreSize, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
IREmit->_StoreMemFPR(Op->StoreSize, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -255,18 +266,6 @@ private:
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
// Helper to check if a Ref is a Zero constant
|
||||
bool IsZero(Ref Node) {
|
||||
auto Header = IR->GetOp<IR::IROp_Header>(Node);
|
||||
if (Header->Op != OP_CONSTANT) {
|
||||
return false;
|
||||
}
|
||||
|
||||
auto Const = Header->C<IROp_Constant>();
|
||||
return Const->Constant == 0;
|
||||
}
|
||||
|
||||
// Handles a Unary operation.
|
||||
// Takes the op we are handling, the Node for the reduced precision case and the node for the normal case.
|
||||
// Depending on the type of Op64, we might need to pass a couple of extra constant arguments, this happens
|
||||
@@ -279,7 +278,7 @@ private:
|
||||
|
||||
// Top Management Helpers
|
||||
/// Set the valid tag for Value as valid (if Valid is true), or invalid (if Valid is false).
|
||||
void SetX87ValidTag(Ref Value, bool Valid);
|
||||
void SetX87ValidTag(uint8_t Offset, bool Valid);
|
||||
// Generates slow code to load/store a value from an offset from the top of the stack
|
||||
Ref LoadStackValueAtOffset_Slow(uint8_t Offset = 0);
|
||||
void StoreStackValueAtOffset_Slow(Ref Value, uint8_t Offset = 0, bool SetValid = true);
|
||||
@@ -294,11 +293,12 @@ private:
|
||||
void MigrateToSlowPathIf(bool ShouldMigrate);
|
||||
// Top Cache Management
|
||||
Ref GetTopWithCache_Slow();
|
||||
Ref GetOffsetTopWithCache_Slow(uint8_t Offset);
|
||||
Ref GetOffsetTopWithCache_Slow(uint8_t Offset, bool Reverse = false);
|
||||
Ref GetOffsetTopAddressWithCache_Slow(uint8_t Offset);
|
||||
void SetTopWithCache_Slow(Ref Value);
|
||||
Ref GetX87ValidTag_Slow(uint8_t Offset);
|
||||
// Resets fields to initial values
|
||||
void Reset(bool AlsoSlowPath = true);
|
||||
void Reset();
|
||||
|
||||
struct StackMemberInfo {
|
||||
StackMemberInfo() {}
|
||||
@@ -330,7 +330,7 @@ private:
|
||||
FixedSizeStack<StackMemberInfo> StackData;
|
||||
|
||||
void InvalidateCaches();
|
||||
void InvalidateTopOffsetCache();
|
||||
void InvalidateCachedRegs();
|
||||
|
||||
// Path Migration helper management
|
||||
std::optional<StackMemberInfo> MigrateToSlowPath_IfInvalid(uint8_t Offset = 0);
|
||||
@@ -345,7 +345,19 @@ private:
|
||||
|
||||
// Cached value for Top
|
||||
// If slowpath is false, then TopCache is nullptr.
|
||||
bool FlushTopPending = false;
|
||||
std::array<bool, 8> FlushValuesPending {};
|
||||
bool FlushValidPending = false;
|
||||
void FlushCachedRegs();
|
||||
|
||||
Ref GetFTW();
|
||||
|
||||
Ref FTWCached {};
|
||||
std::array<Ref, 8> TopOffsetCache {};
|
||||
std::array<Ref, 8> TopOffsetAddressCache {};
|
||||
std::array<Ref, 8> TopValueCache {};
|
||||
std::array<StackSlot, 8> TopValidCache {};
|
||||
|
||||
// Are we on the slow path?
|
||||
// Once we enter the slow path, we never come out.
|
||||
// This just simplifies the code atm. If there's a need to return to the fast path in the future
|
||||
@@ -359,18 +371,21 @@ private:
|
||||
};
|
||||
|
||||
inline void X87StackOptimization::InvalidateCaches() {
|
||||
InvalidateTopOffsetCache();
|
||||
InvalidateCachedRegs();
|
||||
ConstantPool.fill(nullptr);
|
||||
}
|
||||
|
||||
inline void X87StackOptimization::InvalidateTopOffsetCache() {
|
||||
inline void X87StackOptimization::InvalidateCachedRegs() {
|
||||
FlushCachedRegs();
|
||||
FTWCached = {};
|
||||
TopOffsetCache.fill(nullptr);
|
||||
TopOffsetAddressCache.fill(nullptr);
|
||||
TopValueCache.fill(nullptr);
|
||||
TopValidCache.fill(StackSlot::UNUSED);
|
||||
}
|
||||
|
||||
inline void X87StackOptimization::Reset(bool AlsoSlowPath) {
|
||||
if (AlsoSlowPath) {
|
||||
SlowPath = false;
|
||||
}
|
||||
inline void X87StackOptimization::Reset() {
|
||||
SlowPath = false;
|
||||
StackData.clear();
|
||||
InvalidateCaches();
|
||||
}
|
||||
@@ -390,20 +405,25 @@ inline Ref X87StackOptimization::GetConstant(ssize_t Offset) {
|
||||
inline void X87StackOptimization::MigrateToSlowPathIf(bool ShouldMigrate) {
|
||||
if (ShouldMigrate && !SlowPath) {
|
||||
SynchronizeStackValues();
|
||||
Reset(false); // Reset everything but no need to change slowpath
|
||||
StackData.clear();
|
||||
SlowPath = true;
|
||||
}
|
||||
}
|
||||
|
||||
inline Ref X87StackOptimization::GetTopWithCache_Slow() {
|
||||
if (!TopOffsetCache[0]) {
|
||||
TopOffsetCache[0] =
|
||||
IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
TopOffsetCache[0] = IREmit->_LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
}
|
||||
return TopOffsetCache[0];
|
||||
}
|
||||
|
||||
inline Ref X87StackOptimization::GetOffsetTopWithCache_Slow(uint8_t Offset) {
|
||||
inline Ref X87StackOptimization::GetOffsetTopWithCache_Slow(uint8_t Offset, bool Reverse) {
|
||||
if (Reverse) {
|
||||
Offset = 8 - Offset;
|
||||
}
|
||||
|
||||
Offset &= 7;
|
||||
|
||||
if (TopOffsetCache[Offset]) {
|
||||
return TopOffsetCache[Offset];
|
||||
}
|
||||
@@ -418,38 +438,60 @@ inline Ref X87StackOptimization::GetOffsetTopWithCache_Slow(uint8_t Offset) {
|
||||
return OffsetTop;
|
||||
}
|
||||
|
||||
inline Ref X87StackOptimization::GetOffsetTopAddressWithCache_Slow(uint8_t Offset) {
|
||||
if (TopOffsetAddressCache[Offset]) {
|
||||
return TopOffsetAddressCache[Offset];
|
||||
}
|
||||
|
||||
inline void X87StackOptimization::SetTopWithCache_Slow(Ref Value) {
|
||||
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
InvalidateTopOffsetCache();
|
||||
TopOffsetCache[0] = Value;
|
||||
Ref OffsetRef = GetOffsetTopWithCache_Slow(Offset);
|
||||
TopOffsetAddressCache[Offset] = IREmit->_FormContextAddress(OpSize::i64Bit, OffsetRef, 16);
|
||||
|
||||
return TopOffsetAddressCache[Offset];
|
||||
}
|
||||
|
||||
inline void X87StackOptimization::SetX87ValidTag(Ref Value, bool Valid) {
|
||||
Ref AbridgedFTW = IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
Ref RegMask = IREmit->_Lshl(OpSize::i32Bit, GetConstant(1), Value);
|
||||
Ref NewAbridgedFTW = Valid ? IREmit->_Or(OpSize::i32Bit, AbridgedFTW, RegMask) : IREmit->_Andn(OpSize::i32Bit, AbridgedFTW, RegMask);
|
||||
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, NewAbridgedFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
inline void X87StackOptimization::SetTopWithCache_Slow(Ref Value) {
|
||||
InvalidateCachedRegs();
|
||||
TopOffsetCache[0] = Value;
|
||||
FlushTopPending = true;
|
||||
}
|
||||
|
||||
inline Ref X87StackOptimization::GetFTW() {
|
||||
if (!FTWCached) {
|
||||
FTWCached = IREmit->_LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
}
|
||||
return FTWCached;
|
||||
}
|
||||
|
||||
inline void X87StackOptimization::SetX87ValidTag(uint8_t Offset, bool Valid) {
|
||||
TopValidCache[Offset] = Valid ? StackSlot::VALID : StackSlot::INVALID;
|
||||
FlushValidPending = true;
|
||||
}
|
||||
|
||||
inline Ref X87StackOptimization::GetX87ValidTag_Slow(uint8_t Offset) {
|
||||
Ref AbridgedFTW = IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
return IREmit->_And(OpSize::i32Bit, IREmit->_Lshr(OpSize::i32Bit, AbridgedFTW, GetOffsetTopWithCache_Slow(Offset)), GetConstant(1));
|
||||
switch (TopValidCache[Offset]) {
|
||||
case StackSlot::UNUSED:
|
||||
return IREmit->_And(OpSize::i32Bit, IREmit->_Lshr(OpSize::i32Bit, GetFTW(), GetOffsetTopWithCache_Slow(Offset)), GetConstant(1));
|
||||
case StackSlot::INVALID: return GetConstant(0);
|
||||
case StackSlot::VALID: return GetConstant(1);
|
||||
}
|
||||
}
|
||||
|
||||
inline Ref X87StackOptimization::LoadStackValueAtOffset_Slow(uint8_t Offset) {
|
||||
return IREmit->_LoadContextIndexed(GetOffsetTopWithCache_Slow(Offset), ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit,
|
||||
MMBaseOffset(), 16, FPRClass);
|
||||
OrderedNode* TopOffsetAddress = GetOffsetTopAddressWithCache_Slow(Offset);
|
||||
auto Size = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
if (!TopValueCache[Offset]) {
|
||||
TopValueCache[Offset] = IREmit->_LoadMemFPR(Size, TopOffsetAddress, IREmit->_InlineConstant(MMBaseOffset()), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
return TopValueCache[Offset];
|
||||
}
|
||||
|
||||
inline void X87StackOptimization::StoreStackValueAtOffset_Slow(Ref Value, uint8_t Offset, bool SetValid) {
|
||||
OrderedNode* TopOffset = GetOffsetTopWithCache_Slow(Offset);
|
||||
// store
|
||||
IREmit->_StoreContextIndexed(Value, TopOffset, ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit, MMBaseOffset(), 16, FPRClass);
|
||||
TopValueCache[Offset] = Value;
|
||||
FlushValuesPending[Offset] = true;
|
||||
// mark it valid
|
||||
// In some cases we might already know it has been previously set as valid so we don't need to do it again
|
||||
if (SetValid) {
|
||||
SetX87ValidTag(TopOffset, true);
|
||||
SetX87ValidTag(Offset, true);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -457,6 +499,15 @@ inline Ref X87StackOptimization::RotateRight8(uint32_t V, Ref Amount) {
|
||||
return IREmit->_Lshr(OpSize::i32Bit, GetConstant(V | (V << 8)), Amount);
|
||||
}
|
||||
|
||||
inline Ref X87StackOptimization::SilenceNaN(Ref Value) {
|
||||
Ref GPRValue = IREmit->_VExtractToGPR(OpSize::i64Bit, OpSize::i64Bit, Value, 0);
|
||||
|
||||
IREmit->_FCmp(OpSize::i64Bit, Value, Value); // Comparison with itself should set VS if nan
|
||||
Ref QuietNaNGPR = IREmit->_Or(OpSize::i64Bit, GPRValue, IREmit->_Constant(0x0008000000000000ULL));
|
||||
Ref SilencedValue = IREmit->_VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, QuietNaNGPR);
|
||||
return IREmit->_NZCVSelectV(OpSize::i64Bit, CondClass::VS, SilencedValue, Value);
|
||||
}
|
||||
|
||||
inline std::optional<X87StackOptimization::StackMemberInfo> X87StackOptimization::MigrateToSlowPath_IfInvalid(uint8_t Offset) {
|
||||
const auto& [Valid, StackMember] = StackData.top(Offset);
|
||||
MigrateToSlowPathIf(Valid != StackSlot::VALID);
|
||||
@@ -541,25 +592,100 @@ void X87StackOptimization::HandleBinopStack(IROps Op64, bool VFOp64, IROps Op80,
|
||||
|
||||
inline void X87StackOptimization::UpdateTopForPop_Slow() {
|
||||
// Pop the top of the x87 stack
|
||||
auto* TopOffset = GetTopWithCache_Slow();
|
||||
TopOffset = IREmit->Add(OpSize::i32Bit, TopOffset, 1);
|
||||
TopOffset = IREmit->_And(OpSize::i32Bit, TopOffset, GetConstant(7));
|
||||
SetTopWithCache_Slow(TopOffset);
|
||||
GetOffsetTopWithCache_Slow(1);
|
||||
std::rotate(TopOffsetCache.begin(), std::next(TopOffsetCache.begin()), TopOffsetCache.end());
|
||||
std::rotate(TopOffsetAddressCache.begin(), std::next(TopOffsetAddressCache.begin()), TopOffsetAddressCache.end());
|
||||
std::rotate(TopValueCache.begin(), std::next(TopValueCache.begin()), TopValueCache.end());
|
||||
std::rotate(FlushValuesPending.begin(), std::next(FlushValuesPending.begin()), FlushValuesPending.end());
|
||||
std::rotate(TopValidCache.begin(), std::next(TopValidCache.begin()), TopValidCache.end());
|
||||
FlushTopPending = true;
|
||||
}
|
||||
|
||||
inline void X87StackOptimization::UpdateTopForPush_Slow() {
|
||||
// Pop the top of the x87 stack
|
||||
auto* TopOffset = GetTopWithCache_Slow();
|
||||
TopOffset = IREmit->Sub(OpSize::i32Bit, TopOffset, 1);
|
||||
TopOffset = IREmit->_And(OpSize::i32Bit, TopOffset, GetConstant(7));
|
||||
SetTopWithCache_Slow(TopOffset);
|
||||
GetOffsetTopWithCache_Slow(1, true);
|
||||
std::rotate(TopOffsetCache.begin(), std::prev(TopOffsetCache.end()), TopOffsetCache.end());
|
||||
std::rotate(TopOffsetAddressCache.begin(), std::prev(TopOffsetAddressCache.end()), TopOffsetAddressCache.end());
|
||||
std::rotate(TopValueCache.begin(), std::prev(TopValueCache.end()), TopValueCache.end());
|
||||
std::rotate(FlushValuesPending.begin(), std::prev(FlushValuesPending.end()), FlushValuesPending.end());
|
||||
std::rotate(TopValidCache.begin(), std::prev(TopValidCache.end()), TopValidCache.end());
|
||||
FlushTopPending = true;
|
||||
}
|
||||
|
||||
void X87StackOptimization::FlushCachedRegs() {
|
||||
if (FlushTopPending) {
|
||||
IREmit->_StoreContextGPR(OpSize::i8Bit, TopOffsetCache[0], offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
FlushTopPending = false;
|
||||
}
|
||||
|
||||
auto Size = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
for (size_t i = 0; i < FlushValuesPending.size(); i++) {
|
||||
if (FlushValuesPending[i]) {
|
||||
OrderedNode* TopOffsetAddress = GetOffsetTopAddressWithCache_Slow(i);
|
||||
IREmit->_StoreMemFPR(Size, TopValueCache[i], TopOffsetAddress, IREmit->_InlineConstant(MMBaseOffset()), Size, MemOffsetType::SXTX, 1);
|
||||
// store
|
||||
FlushValuesPending[i] = false;
|
||||
}
|
||||
}
|
||||
|
||||
if (FlushValidPending) {
|
||||
uint8_t ValidMask = 0;
|
||||
uint8_t InvalidMask = 0;
|
||||
for (auto It = TopValidCache.rbegin(); It != TopValidCache.rend(); It++) {
|
||||
ValidMask <<= 1;
|
||||
InvalidMask <<= 1;
|
||||
if (*It == StackSlot::VALID) {
|
||||
ValidMask |= 1;
|
||||
} else if (*It == StackSlot::INVALID) {
|
||||
InvalidMask |= 1;
|
||||
}
|
||||
}
|
||||
|
||||
if (ValidMask || InvalidMask) {
|
||||
Ref NewFTW = [&]() {
|
||||
if (ValidMask == 0xff || InvalidMask == 0xff) {
|
||||
// If InvalidMask == 0xff then ValidMask = 0
|
||||
return GetConstant(ValidMask);
|
||||
} else {
|
||||
Ref NewFTW = GetFTW();
|
||||
Ref RotAmount {};
|
||||
if (std::popcount(ValidMask) == 1) {
|
||||
uint8_t BitIdx = std::countr_zero(ValidMask);
|
||||
Ref RegMask = IREmit->_Lshl(OpSize::i32Bit, GetConstant(1), GetOffsetTopWithCache_Slow(BitIdx));
|
||||
NewFTW = IREmit->_Or(OpSize::i32Bit, NewFTW, RegMask);
|
||||
} else if (ValidMask) {
|
||||
RotAmount = IREmit->_Sub(OpSize::i32Bit, GetConstant(8), GetTopWithCache_Slow());
|
||||
// perform a rotate right on mask by top
|
||||
NewFTW = IREmit->_Or(OpSize::i32Bit, NewFTW, RotateRight8(ValidMask, RotAmount));
|
||||
}
|
||||
|
||||
if (std::popcount(InvalidMask) == 1) {
|
||||
uint8_t BitIdx = std::countr_zero(InvalidMask);
|
||||
Ref RegMask = IREmit->_Lshl(OpSize::i32Bit, GetConstant(1), GetOffsetTopWithCache_Slow(BitIdx));
|
||||
NewFTW = IREmit->_Andn(OpSize::i32Bit, NewFTW, RegMask);
|
||||
} else if (InvalidMask) {
|
||||
if (!RotAmount) {
|
||||
RotAmount = IREmit->_Sub(OpSize::i32Bit, GetConstant(8), GetTopWithCache_Slow());
|
||||
}
|
||||
NewFTW = IREmit->_Andn(OpSize::i32Bit, NewFTW, RotateRight8(InvalidMask, RotAmount));
|
||||
}
|
||||
return NewFTW;
|
||||
}
|
||||
}();
|
||||
|
||||
IREmit->_StoreContextGPR(OpSize::i8Bit, NewFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
FTWCached = NewFTW;
|
||||
}
|
||||
|
||||
FlushValidPending = false;
|
||||
}
|
||||
}
|
||||
|
||||
// We synchronize stack values in a few occasions but one of the most important of those,
|
||||
// is when we move from fast to a slow path and need to make sure that the context is properly
|
||||
// written.
|
||||
Ref X87StackOptimization::SynchronizeStackValues() {
|
||||
if (SlowPath) { // Nothing to do here.
|
||||
if (SlowPath) {
|
||||
return GetTopWithCache_Slow();
|
||||
}
|
||||
|
||||
@@ -568,8 +694,7 @@ Ref X87StackOptimization::SynchronizeStackValues() {
|
||||
const auto TopOffset = StackData.TopOffset;
|
||||
|
||||
if (TopOffset != 0) {
|
||||
auto* OrigTop = GetTopWithCache_Slow();
|
||||
Ref NewTop = IREmit->_And(OpSize::i32Bit, IREmit->Sub(OpSize::i32Bit, OrigTop, TopOffset), GetConstant(0x7));
|
||||
Ref NewTop = GetOffsetTopWithCache_Slow(TopOffset, true);
|
||||
SetTopWithCache_Slow(NewTop);
|
||||
}
|
||||
StackData.TopOffset = 0;
|
||||
@@ -581,51 +706,22 @@ Ref X87StackOptimization::SynchronizeStackValues() {
|
||||
for (size_t i = 0; i < StackData.size; ++i) {
|
||||
const auto& [Valid, StackMember] = StackData.top(i);
|
||||
|
||||
if (Valid == StackSlot::UNUSED) {
|
||||
continue;
|
||||
}
|
||||
Ref TopIndex = GetOffsetTopWithCache_Slow(i);
|
||||
if (Valid == StackSlot::VALID) {
|
||||
IREmit->_StoreContextIndexed(StackMember.StackDataNode, TopIndex, ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit,
|
||||
MMBaseOffset(), 16, FPRClass);
|
||||
StoreStackValueAtOffset_Slow(StackMember.StackDataNode, i, false);
|
||||
}
|
||||
}
|
||||
{ // Set valid tags
|
||||
uint8_t Mask = StackData.getValidMask();
|
||||
if (Mask == 0xff) {
|
||||
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, GetConstant(Mask), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
} else if (Mask != 0) {
|
||||
if (std::popcount(Mask) == 1) {
|
||||
uint8_t BitIdx = __builtin_ctz(Mask);
|
||||
SetX87ValidTag(GetOffsetTopWithCache_Slow(BitIdx), true);
|
||||
} else {
|
||||
// perform a rotate right on mask by top
|
||||
auto* TopValue = GetTopWithCache_Slow();
|
||||
Ref RotAmount = IREmit->_Sub(OpSize::i32Bit, GetConstant(8), TopValue);
|
||||
Ref AbridgedFTW = IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
Ref NewAbridgedFTW = IREmit->_Or(OpSize::i32Bit, AbridgedFTW, RotateRight8(Mask, RotAmount));
|
||||
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, NewAbridgedFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
}
|
||||
}
|
||||
}
|
||||
{ // Set invalid tags
|
||||
uint8_t Mask = StackData.getInvalidMask();
|
||||
if (Mask == 0xff) {
|
||||
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, GetConstant(0), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
} else if (Mask != 0) {
|
||||
if (std::popcount(Mask)) {
|
||||
uint8_t BitIdx = __builtin_ctz(Mask);
|
||||
SetX87ValidTag(GetOffsetTopWithCache_Slow(BitIdx), false);
|
||||
} else {
|
||||
// Same rotate right as above but this time on the invalid mask
|
||||
auto* TopValue = GetTopWithCache_Slow();
|
||||
Ref RotAmount = IREmit->_Sub(OpSize::i32Bit, GetConstant(8), TopValue);
|
||||
Ref AbridgedFTW = IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
Ref NewAbridgedFTW = IREmit->_Andn(OpSize::i32Bit, AbridgedFTW, RotateRight8(Mask, RotAmount));
|
||||
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, NewAbridgedFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
}
|
||||
uint8_t ValidMask = StackData.getValidMask();
|
||||
uint8_t InvalidMask = StackData.getInvalidMask();
|
||||
for (auto& Elem : TopValidCache) {
|
||||
Elem = (ValidMask & 1) ? StackSlot::VALID : ((InvalidMask & 1) ? StackSlot::INVALID : StackSlot::UNUSED);
|
||||
|
||||
ValidMask >>= 1;
|
||||
InvalidMask >>= 1;
|
||||
}
|
||||
FlushValidPending = true;
|
||||
}
|
||||
|
||||
return TopValue;
|
||||
}
|
||||
|
||||
@@ -815,6 +911,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
|
||||
case OP_INITSTACK: {
|
||||
StackData.clear();
|
||||
InvalidateCachedRegs();
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -824,18 +921,14 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
|
||||
if (Offset != 0xff) { // invalidate single offset
|
||||
if (SlowPath) {
|
||||
auto* TopValue = GetTopWithCache_Slow();
|
||||
if (Offset != 0) {
|
||||
auto* Mask = GetConstant(7);
|
||||
TopValue = IREmit->_And(OpSize::i32Bit, IREmit->Add(OpSize::i32Bit, TopValue, Offset), Mask);
|
||||
}
|
||||
SetX87ValidTag(TopValue, false);
|
||||
SetX87ValidTag(Offset, false);
|
||||
} else {
|
||||
StackData.setTagInvalid(Offset);
|
||||
}
|
||||
} else { // invalidate all
|
||||
if (SlowPath) {
|
||||
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, GetConstant(0), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
TopValidCache.fill(StackSlot::INVALID);
|
||||
FlushValidPending = true;
|
||||
} else {
|
||||
for (size_t i = 0; i < StackData.size; i++) {
|
||||
StackData.setTagInvalid(i);
|
||||
@@ -921,8 +1014,8 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
// or similar. As long as the source size and dest size are one and the same.
|
||||
// This will avoid any conversions between source and stack element size and conversion back.
|
||||
if (!SlowPath && Value->Source && Value->Source->Size == Op->StoreSize && Value->InterpretAsFloat) {
|
||||
IREmit->_StoreMem(Value->InterpretAsFloat ? FPRClass : GPRClass, Op->StoreSize, Value->Source->Node, AddrNode, Offset, Align,
|
||||
OffsetType, OffsetScale);
|
||||
const auto ClassType = Value->InterpretAsFloat ? RegClass::FPR : RegClass::GPR;
|
||||
IREmit->_StoreMem(ClassType, Op->StoreSize, Value->Source->Node, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -953,7 +1046,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
}
|
||||
case OP_POPSTACKDESTROY: {
|
||||
if (SlowPath) {
|
||||
SetX87ValidTag(GetTopWithCache_Slow(), false);
|
||||
SetX87ValidTag(0, false);
|
||||
}
|
||||
StackPop();
|
||||
break;
|
||||
@@ -1052,12 +1145,14 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
case OP_SYNCSTACKTOSLOW: {
|
||||
// This synchronizes stack values but doesn't necessarily moves us off the FastPath!
|
||||
Ref NewTop = SynchronizeStackValues();
|
||||
FlushCachedRegs();
|
||||
IREmit->ReplaceUsesWithAfter(CodeNode, NewTop, CodeNode);
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_STACKFORCESLOW: {
|
||||
MigrateToSlowPathIf(true);
|
||||
InvalidateCachedRegs();
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -1084,7 +1179,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
|
||||
Ref Value {};
|
||||
if (ReducedPrecisionMode) {
|
||||
Value = IREmit->_Vector_FToI(OpSize::i64Bit, OpSize::i64Bit, St0, Round_Host);
|
||||
Value = IREmit->_Vector_FToI(OpSize::i64Bit, OpSize::i64Bit, St0, RoundMode::Host);
|
||||
} else {
|
||||
Value = IREmit->_F80Round(St0);
|
||||
}
|
||||
@@ -1116,6 +1211,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
LOGMAN_THROW_A_FMT(IsBlockExit(LastIROp->Op), "must be exit");
|
||||
IREmit->SetWriteCursorBefore(LastCodeNode);
|
||||
SynchronizeStackValues();
|
||||
FlushCachedRegs();
|
||||
}
|
||||
|
||||
return;
|
||||
|
||||
@@ -19,9 +19,9 @@ union PhysicalRegister {
|
||||
return Raw == Other.Raw;
|
||||
}
|
||||
|
||||
PhysicalRegister(RegisterClassType Class, uint8_t Reg)
|
||||
PhysicalRegister(RegClass Class, uint8_t Reg)
|
||||
: Reg(Reg)
|
||||
, Class(Class.Val) {}
|
||||
, Class(uint8_t(Class)) {}
|
||||
|
||||
PhysicalRegister(OrderedNodeWrapper Arg)
|
||||
: Raw(Arg.GetImmediate()) {}
|
||||
@@ -29,12 +29,16 @@ union PhysicalRegister {
|
||||
PhysicalRegister(Ref Node)
|
||||
: Raw(Node->Reg) {}
|
||||
|
||||
RegClass AsRegClass() const {
|
||||
return RegClass {Class};
|
||||
}
|
||||
|
||||
static const PhysicalRegister Invalid() {
|
||||
return PhysicalRegister(InvalidClass, 0);
|
||||
return PhysicalRegister(RegClass::Invalid, 0);
|
||||
}
|
||||
|
||||
bool IsInvalid() const {
|
||||
static_assert(InvalidClass == 0);
|
||||
static_assert(uint8_t(RegClass::Invalid) == 0);
|
||||
return Raw == 0;
|
||||
}
|
||||
};
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/PrctlUtils.h>
|
||||
#include <FEXCore/Utils/TypeDefines.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
@@ -50,6 +51,10 @@ void* FEX_mmap(void* addr, size_t length, int prot, int flags, int fd, off_t off
|
||||
errno = -(uint64_t)Result;
|
||||
return (void*)-1;
|
||||
}
|
||||
|
||||
if (flags & MAP_ANONYMOUS) {
|
||||
prctl(PR_SET_VMA, PR_SET_VMA_ANON_NAME, Result, length, "FEXMem");
|
||||
}
|
||||
return Result;
|
||||
}
|
||||
int FEX_munmap(void* addr, size_t length) {
|
||||
|
||||
@@ -165,7 +165,8 @@ private:
|
||||
size_t SizePlusManagedData = UsedSize + SizeOfLiveRegion;
|
||||
|
||||
auto Res = mprotect(reinterpret_cast<void*>(ReservedRegion->Base), SizePlusManagedData, PROT_READ | PROT_WRITE);
|
||||
LOGMAN_THROW_A_FMT(Res != -1, "Couldn't mprotect region: {} '{}' Likely occurs when running out of memory or Maximum VMAs", errno, strerror(errno));
|
||||
LOGMAN_THROW_A_FMT(Res != -1, "Couldn't mprotect region: {} '{}' Likely occurs when running out of memory or Maximum VMAs", errno,
|
||||
strerror(errno));
|
||||
|
||||
LiveVMARegion* LiveRange = new (reinterpret_cast<void*>(ReservedRegion->Base)) LiveVMARegion();
|
||||
|
||||
@@ -273,8 +274,8 @@ void* OSAllocator_64Bit::Mmap(void* addr, size_t length, int prot, int flags, in
|
||||
again:
|
||||
|
||||
struct RangeResult final {
|
||||
LiveVMARegion *RegionInsertedInto;
|
||||
void *Ptr;
|
||||
LiveVMARegion* RegionInsertedInto;
|
||||
void* Ptr;
|
||||
};
|
||||
|
||||
auto CheckIfRangeFits = [&AllocatedOffset](LiveVMARegion* Region, uint64_t length, int prot, int flags, int fd, off_t offset,
|
||||
|
||||
+3
-8
@@ -78,14 +78,9 @@ When generating IR inside of the `OpDispatchBuilder` it is straight forward, jus
|
||||
This is an intrusive allocator that is used by the `OpDispatchBuilder` for storing IR data. It is a simple linear arena allocator without resizing capabilities.
|
||||
|
||||
### OpDispatchBuilder
|
||||
OpDispatchBuilder provides two routines for handling the IR outside of the class
|
||||
* `IRListView ViewIR();`
|
||||
* Returns a wrapper container class the allows you to view the IR. This doesn't take ownership of the IR data.
|
||||
* If the OpDispatcherBuilder changes its IR then changes are also visible to this class
|
||||
* `IRListView *CreateIRCopy()`
|
||||
* As the name says, it creates a new copy of the IR that is in the OpDispatchBuilder
|
||||
* Copying the IR only copies the memory used and doesn't have any free space for optimizations after this copy operation
|
||||
* Useful for tiered recompilers, AOT, and offline analysis
|
||||
OpDispatchBuilder provides `IRListView ViewIR()` for handling the IR outside of the class:
|
||||
* Returns a wrapper container class the allows you to view the IR. This doesn't take ownership of the IR data.
|
||||
* If the OpDispatcherBuilder changes its IR then changes are also visible to this class
|
||||
|
||||
This class uses two IntrusiveAllocator objects for tracking IR data. `ListData` and `Data` are the object names.
|
||||
* `ListData` is for tracking the doubly linked list of nodes
|
||||
|
||||
@@ -0,0 +1,58 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
namespace Core {
|
||||
struct InternalThreadState;
|
||||
} // namespace Core
|
||||
|
||||
namespace HLE {
|
||||
struct SourcecodeMap;
|
||||
} // namespace HLE
|
||||
|
||||
// Generic information associated with an executable file.
|
||||
struct ExecutableFileInfo {
|
||||
~ExecutableFileInfo();
|
||||
|
||||
fextl::unique_ptr<HLE::SourcecodeMap> SourcecodeMap;
|
||||
fextl::string FileId;
|
||||
fextl::string Filename;
|
||||
};
|
||||
|
||||
// Information associated with a specific section of an executable file
|
||||
struct ExecutableFileSectionInfo {
|
||||
ExecutableFileInfo& FileInfo;
|
||||
|
||||
// Start address that the file is mapped to.
|
||||
uintptr_t FileStartVA;
|
||||
};
|
||||
|
||||
class AbstractCodeCache {
|
||||
public:
|
||||
virtual ~AbstractCodeCache() = default;
|
||||
|
||||
/**
|
||||
* Loads a code cache from mapped memory and appends it to the current Core state.
|
||||
* TODO: Optionally recompiles all contained code blocks at runtime for validation.
|
||||
*/
|
||||
virtual void LoadData(Core::InternalThreadState&, std::byte* MappedCacheFile, const ExecutableFileSectionInfo&) = 0;
|
||||
|
||||
/**
|
||||
* Bundles the current Core state (CodeBuffer, GuestToHostMapping, ...) to a code cache and writes it to the given file descriptor.
|
||||
* Returns true on success.
|
||||
*/
|
||||
virtual bool SaveData(Core::InternalThreadState&, int TargetFD, const ExecutableFileSectionInfo&, uint64_t SerializedBaseAddress) = 0;
|
||||
|
||||
/**
|
||||
* Function to be called before compiling any code for caching purposes
|
||||
*/
|
||||
virtual void InitiateCacheGeneration() = 0;
|
||||
};
|
||||
|
||||
} // namespace FEXCore
|
||||
@@ -4,6 +4,7 @@
|
||||
#include <stdint.h>
|
||||
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Core/CodeCache.h>
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
@@ -13,12 +14,7 @@
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <istream>
|
||||
#include <ostream>
|
||||
#include <span>
|
||||
|
||||
namespace FEXCore {
|
||||
class CodeLoader;
|
||||
struct HostFeatures;
|
||||
class ForkableSharedMutex;
|
||||
class ThunkHandler;
|
||||
@@ -29,17 +25,11 @@ struct CPUState;
|
||||
struct InternalThreadState;
|
||||
} // namespace FEXCore::Core
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
class CPUBackend;
|
||||
}
|
||||
|
||||
namespace FEXCore::HLE {
|
||||
struct SyscallArguments;
|
||||
class SyscallHandler;
|
||||
} // namespace FEXCore::HLE
|
||||
|
||||
namespace FEXCore::IR {
|
||||
struct AOTIRCacheEntry;
|
||||
class IREmitter;
|
||||
} // namespace FEXCore::IR
|
||||
|
||||
@@ -148,18 +138,13 @@ public:
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::IR::AOTIRCacheEntry* LoadAOTIRCacheEntry(const fextl::string& Name) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void UnloadAOTIRCacheEntry(FEXCore::IR::AOTIRCacheEntry* Entry) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void FinalizeAOTIRCache() = 0;
|
||||
virtual AbstractCodeCache& GetCodeCache() = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer = true) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void InvalidateGuestCodeRange(
|
||||
FEXCore::Core::InternalThreadState* Thread, InvalidatedEntryAccumulator& Accumulator, uint64_t Start, uint64_t Length) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::ForkableSharedMutex& GetCodeInvalidationMutex() = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void
|
||||
ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) = 0;
|
||||
|
||||
|
||||
@@ -146,15 +146,15 @@ struct CPUState {
|
||||
// - Three are reserved for user-space to setup TLS segments in
|
||||
// LDT segments are entirely controlled by userspace.
|
||||
// - Kernel allocates up to 8192 ldt segments.
|
||||
gdt_segment *segment_arrays[2] {};
|
||||
gdt_segment* segment_arrays[2] {};
|
||||
|
||||
static gdt_segment* GetSegmentFromIndex(CPUState &State, uint16_t Selector) {
|
||||
static gdt_segment* GetSegmentFromIndex(CPUState& State, uint16_t Selector) {
|
||||
auto base = State.segment_arrays[(Selector >> 2) & 1];
|
||||
return &base[Selector >> 3];
|
||||
}
|
||||
|
||||
static uint32_t CalculateGDTBase(gdt_segment GDT) {
|
||||
uint32_t Base{};
|
||||
uint32_t Base {};
|
||||
Base |= GDT.Base2 << 24;
|
||||
Base |= GDT.Base1 << 16;
|
||||
Base |= GDT.Base0;
|
||||
@@ -162,19 +162,19 @@ struct CPUState {
|
||||
}
|
||||
|
||||
static uint32_t CalculateGDTLimit(gdt_segment GDT) {
|
||||
uint32_t Limit{};
|
||||
uint32_t Limit {};
|
||||
Limit |= GDT.Limit1 << 16;
|
||||
Limit |= GDT.Limit0;
|
||||
return Limit;
|
||||
}
|
||||
|
||||
static void SetGDTBase(gdt_segment *GDT, uint32_t Base) {
|
||||
static void SetGDTBase(gdt_segment* GDT, uint32_t Base) {
|
||||
GDT->Base0 = Base;
|
||||
GDT->Base1 = Base >> 16;
|
||||
GDT->Base2 = Base >> 24;
|
||||
}
|
||||
|
||||
static void SetGDTLimit(gdt_segment *GDT, uint32_t Limit) {
|
||||
static void SetGDTLimit(gdt_segment* GDT, uint32_t Limit) {
|
||||
GDT->Limit0 = Limit;
|
||||
GDT->Limit1 = Limit >> 16;
|
||||
}
|
||||
|
||||
@@ -40,6 +40,7 @@ struct HostFeatures {
|
||||
bool SupportsECV {};
|
||||
bool SupportsWFXT {};
|
||||
bool Supports3DNow {};
|
||||
bool SupportsSSE4a {};
|
||||
|
||||
// Float exception behaviour
|
||||
bool SupportsAFP {};
|
||||
|
||||
@@ -9,10 +9,6 @@
|
||||
#include <memory>
|
||||
#include <filesystem>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
struct AOTIRCacheEntry;
|
||||
}
|
||||
|
||||
namespace FEXCore::HLE {
|
||||
|
||||
struct SourcecodeLineMapping {
|
||||
@@ -88,6 +84,6 @@ private:
|
||||
|
||||
class SourcecodeResolver {
|
||||
public:
|
||||
virtual fextl::unique_ptr<SourcecodeMap> GenerateMap(const std::string_view& GuestBinaryFile, const std::string_view& GuestBinaryFileId) = 0;
|
||||
virtual fextl::unique_ptr<SourcecodeMap> GenerateMap(std::string_view GuestBinaryFile, std::string_view GuestBinaryFileId) = 0;
|
||||
};
|
||||
} // namespace FEXCore::HLE
|
||||
@@ -1,13 +1,11 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <cstdint>
|
||||
#include <shared_mutex>
|
||||
#include <optional>
|
||||
|
||||
#include <FEXCore/Core/CodeCache.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
struct AOTIRCacheEntry;
|
||||
}
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class Context;
|
||||
@@ -51,19 +49,6 @@ struct ExecutableRangeInfo {
|
||||
class SyscallHandler;
|
||||
class SourcecodeResolver;
|
||||
|
||||
struct AOTIRCacheEntryLookupResult {
|
||||
AOTIRCacheEntryLookupResult(FEXCore::IR::AOTIRCacheEntry* Entry, uintptr_t VAFileStart)
|
||||
: Entry(Entry)
|
||||
, VAFileStart(VAFileStart) {}
|
||||
|
||||
AOTIRCacheEntryLookupResult(AOTIRCacheEntryLookupResult&&) = default;
|
||||
|
||||
FEXCore::IR::AOTIRCacheEntry* Entry;
|
||||
uintptr_t VAFileStart;
|
||||
|
||||
friend class SyscallHandler;
|
||||
};
|
||||
|
||||
class SyscallHandler {
|
||||
public:
|
||||
virtual ~SyscallHandler() = default;
|
||||
@@ -82,7 +67,7 @@ public:
|
||||
virtual void MarkOvercommitRange(uint64_t Start, uint64_t Length) {}
|
||||
virtual void UnmarkOvercommitRange(uint64_t Start, uint64_t Length) {}
|
||||
virtual ExecutableRangeInfo QueryGuestExecutableRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Address) = 0;
|
||||
virtual AOTIRCacheEntryLookupResult LookupAOTIRCacheEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestAddr) = 0;
|
||||
virtual std::optional<ExecutableFileSectionInfo> LookupExecutableFileSection(Core::InternalThreadState& Thread, uint64_t GuestAddr) = 0;
|
||||
|
||||
virtual void PreCompile() {}
|
||||
|
||||
|
||||
Loaded 100 of 296 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user