mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-07 08:00:21 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e80e2bdafe | ||
|
|
ac23bce0ba | ||
|
|
1051cd97cf | ||
|
|
45330fdd5d | ||
|
|
bd296d7a11 | ||
|
|
dd7e1baa78 | ||
|
|
80a209d6dc | ||
|
|
ab228c1fcb | ||
|
|
0dde233b1e | ||
|
|
cf0d92d968 | ||
|
|
ce4b052242 | ||
|
|
125bb3afe1 | ||
|
|
30ee2c18ef | ||
|
|
ff1a5dd4c2 | ||
|
|
b9c9c7d671 | ||
|
|
62d9961bd1 | ||
|
|
4b7ed9599d | ||
|
|
c8951de9dd | ||
|
|
63517c377d | ||
|
|
5005ebdfc9 | ||
|
|
a2615f0e52 | ||
|
|
0317d381fe | ||
|
|
31b5181bca | ||
|
|
7ae59055ff | ||
|
|
3524117c23 | ||
|
|
e8ad1ca0a0 | ||
|
|
3ac1650001 | ||
|
|
4f8ae81562 | ||
|
|
329d624a99 | ||
|
|
de92624c9a | ||
|
|
c6a034da40 | ||
|
|
737e76968a | ||
|
|
7c6d49155c | ||
|
|
4d404ea94d | ||
|
|
dc810c7a1e | ||
|
|
e0ee4e71f8 | ||
|
|
ab7dac90a2 | ||
|
|
3a423fbc41 | ||
|
|
0982ec617d | ||
|
|
b45f27c8d0 | ||
|
|
54f62b6701 | ||
|
|
9664d98ba5 | ||
|
|
0c318834b5 | ||
|
|
4c2836f4b3 | ||
|
|
a112db169c | ||
|
|
0dc33ec893 | ||
|
|
0e93ba532f | ||
|
|
9e524a30d6 | ||
|
|
ecca4e4abc | ||
|
|
51791e9efa | ||
|
|
5eab087cc2 | ||
|
|
cd05cdaa57 | ||
|
|
f5e18ccea6 | ||
|
|
0da6b07770 | ||
|
|
05805356fb | ||
|
|
a72ebfdec0 | ||
|
|
59e5da9c7f | ||
|
|
cfd59db998 | ||
|
|
1fa6bf1cc8 | ||
|
|
a2f89de71e | ||
|
|
36a27de286 | ||
|
|
9b685ba824 | ||
|
|
8e9d5fe6ac | ||
|
|
89aa590615 | ||
|
|
3960e0f2a3 | ||
|
|
c3c52d01d3 | ||
|
|
0caef59e6a | ||
|
|
99e6924ed1 | ||
|
|
431e1629b0 | ||
|
|
ff9f702e52 | ||
|
|
6903159f30 | ||
|
|
504b7a03ad | ||
|
|
6933c2aaff | ||
|
|
5fd00e31bb | ||
|
|
a984674dae | ||
|
|
8589119725 | ||
|
|
5e0205378b | ||
|
|
bff2f2e5f9 | ||
|
|
5f9052a675 | ||
|
|
8691b3964f | ||
|
|
4dfe0a0595 | ||
|
|
164299ce25 | ||
|
|
9269ca6ebe | ||
|
|
93fe22e045 | ||
|
|
3849278d5a | ||
|
|
c0f976e200 | ||
|
|
3143749f64 | ||
|
|
1704805e32 | ||
|
|
3aabe07720 | ||
|
|
c5815ab59e | ||
|
|
f6fcfbabd5 | ||
|
|
a0835c95e4 | ||
|
|
ce0c24eee9 | ||
|
|
c9f0ecb1e8 | ||
|
|
50febb3459 | ||
|
|
601b0f96b1 | ||
|
|
018661609a | ||
|
|
98d935d972 | ||
|
|
b07660c4fa | ||
|
|
8037231376 | ||
|
|
6e422a8691 | ||
|
|
3d347ed565 | ||
|
|
13446d8b6b | ||
|
|
08c34bf573 | ||
|
|
3a64ea1650 | ||
|
|
db3439b61b | ||
|
|
7814be7467 | ||
|
|
8a7e0432f1 | ||
|
|
ff1d51c7bd | ||
|
|
f2602e3136 | ||
|
|
313ce0fed8 | ||
|
|
ba0887defa | ||
|
|
063841f491 | ||
|
|
4aa5c78150 | ||
|
|
c1688fa392 | ||
|
|
0dd03e9cb6 | ||
|
|
8d60d70553 | ||
|
|
c17340547c | ||
|
|
34aa6bf6c7 | ||
|
|
82a6ce63b4 | ||
|
|
3ac6ba0fe2 | ||
|
|
8b0fb4fa19 | ||
|
|
9d437b8863 | ||
|
|
f72469c80a | ||
|
|
eec7972808 | ||
|
|
fc9343e7a4 | ||
|
|
be8e353d12 | ||
|
|
53217acdd0 | ||
|
|
c1e0ea5e0a | ||
|
|
467afc7248 | ||
|
|
62753071f4 | ||
|
|
de32de9edd | ||
|
|
aa7d2c3f7e | ||
|
|
2a1dd68583 | ||
|
|
80909eaa84 | ||
|
|
c6a557abe2 | ||
|
|
8d7b8729f1 | ||
|
|
3f9a1c3751 | ||
|
|
0dfe617d96 | ||
|
|
235ad441b4 | ||
|
|
a9d00b3f8d | ||
|
|
fb41ba172d | ||
|
|
aec5b21d2a | ||
|
|
124097d563 | ||
|
|
e137c2edff | ||
|
|
2b35dd829c | ||
|
|
686895802a | ||
|
|
72d0228cd7 | ||
|
|
298e6cad0c | ||
|
|
ad34c228e3 | ||
|
|
66f74a431f | ||
|
|
57bae90c69 | ||
|
|
83c2e52ba5 | ||
|
|
4f8525da4d | ||
|
|
3b8491b558 | ||
|
|
f69c53d294 | ||
|
|
6b226dd6af | ||
|
|
227462e4d8 | ||
|
|
5f00ba5cae | ||
|
|
0b8a6d9599 | ||
|
|
13bf04a81e | ||
|
|
7eb8409f71 | ||
|
|
e821072cb9 | ||
|
|
9c01dd9d9f | ||
|
|
6a43db8c8f | ||
|
|
3ffc301dd0 | ||
|
|
46fcbe2fc0 | ||
|
|
b7806e47e9 | ||
|
|
a1ed54adc7 | ||
|
|
250a4ea4e3 | ||
|
|
b3e090c8ff | ||
|
|
982518d3a4 | ||
|
|
9ad1d5548d | ||
|
|
0de2558ef7 | ||
|
|
3e0e601616 | ||
|
|
8b202b0ebf | ||
|
|
6d2f98a379 | ||
|
|
84379b5fdf | ||
|
|
d005fdcd03 | ||
|
|
a97fb2f34f | ||
|
|
d48981b6b0 | ||
|
|
af32228e38 | ||
|
|
8b35275ec1 | ||
|
|
302a6c96ff | ||
|
|
f4a4b2c14b | ||
|
|
88b94bef54 | ||
|
|
751b66d45d | ||
|
|
ae6a57e667 | ||
|
|
9110546d34 | ||
|
|
1f1d0706dd | ||
|
|
a82d41ee22 | ||
|
|
e8e70828d1 | ||
|
|
b7a58fdb8a | ||
|
|
dc9fde8fae | ||
|
|
c0cf4f6a61 | ||
|
|
21fc6bedcb | ||
|
|
ad6fd5ab72 | ||
|
|
0f696c6092 | ||
|
|
9af3bc1864 | ||
|
|
b020e593a5 | ||
|
|
4771a340f5 | ||
|
|
4449b60459 | ||
|
|
4139332ad9 | ||
|
|
30a28ff1ad | ||
|
|
498d0fc145 | ||
|
|
4a547fe95f | ||
|
|
ca906589d4 | ||
|
|
8bafae2262 | ||
|
|
4ef82c82b5 | ||
|
|
e03d253310 | ||
|
|
48c45da2de | ||
|
|
04a1ac967c | ||
|
|
aa17f64593 | ||
|
|
ae5cfcc249 | ||
|
|
8f578b57f2 | ||
|
|
3871646611 | ||
|
|
48a574dfd3 | ||
|
|
ec49100c63 | ||
|
|
c04d2409da | ||
|
|
b93b713179 | ||
|
|
fe2f54fc3d | ||
|
|
599f8f99ed | ||
|
|
b1b338ed5a | ||
|
|
e4d659a619 | ||
|
|
9ce94266e1 | ||
|
|
5a19425b28 | ||
|
|
1494aac861 | ||
|
|
8a21ecabee | ||
|
|
005b2bc3db | ||
|
|
f27830bf41 | ||
|
|
5fe6afc270 | ||
|
|
4f9bc70562 | ||
|
|
835155021d | ||
|
|
e2c5c2292d | ||
|
|
072690a191 | ||
|
|
a2c9d5a398 | ||
|
|
5944342604 | ||
|
|
4dbcd34548 | ||
|
|
f02c73a2c2 | ||
|
|
158ba1ae3b | ||
|
|
a85a77e088 | ||
|
|
542ab04671 | ||
|
|
28ee2ca5a2 | ||
|
|
3913dd6c8b | ||
|
|
1ecf147e3e | ||
|
|
d8fa53a445 | ||
|
|
1c1ad876ca | ||
|
|
5cd0f6e7e7 | ||
|
|
2627ed3193 | ||
|
|
e3bb8b43ab | ||
|
|
8ba6a3bda9 | ||
|
|
d288249609 | ||
|
|
33d7c6be9e | ||
|
|
c13acc5b60 | ||
|
|
fc9be28d51 | ||
|
|
944f400d19 | ||
|
|
5aee30a7b5 | ||
|
|
58ae49372c | ||
|
|
917e69b021 | ||
|
|
9242e59841 | ||
|
|
05d15fa052 | ||
|
|
0a62a4c571 | ||
|
|
525266e482 | ||
|
|
503881f5df | ||
|
|
04fd4f2bb4 | ||
|
|
8be1e5e260 | ||
|
|
e84e084d56 | ||
|
|
12ad05e089 | ||
|
|
472ce7c1e4 | ||
|
|
19fbc10c16 | ||
|
|
39f3406fba | ||
|
|
19b0a9cd7d | ||
|
|
eac579f714 | ||
|
|
b0a31f7105 | ||
|
|
29163cfd07 | ||
|
|
f05b24636f | ||
|
|
39466a3a8f | ||
|
|
bccf18f46c | ||
|
|
1109126e35 | ||
|
|
98dd40c3a8 | ||
|
|
b1cfb104ff | ||
|
|
c339cfe184 | ||
|
|
ca8e6a83e1 | ||
|
|
760217ab29 | ||
|
|
ee9361bf41 | ||
|
|
e62bc24b3f | ||
|
|
18074307f6 | ||
|
|
63b70ff3d4 | ||
|
|
dacdfd5c02 | ||
|
|
d5c039cdc7 | ||
|
|
e2e6f2a92b | ||
|
|
d7d8244593 | ||
|
|
b05e5ce14e | ||
|
|
08730c817a | ||
|
|
34e086eec2 | ||
|
|
83c8a9b675 | ||
|
|
a3db629fd6 | ||
|
|
64b3cef126 | ||
|
|
7100a2eaee | ||
|
|
ffcde1823b | ||
|
|
4c73c715ad | ||
|
|
26c3ebd6a7 | ||
|
|
790bd9747f | ||
|
|
83aa8731d1 | ||
|
|
bbd9eb5b9a | ||
|
|
cc90fa5773 | ||
|
|
30ebc6c938 | ||
|
|
c0a8984799 | ||
|
|
a6e34d301e | ||
|
|
82c88168cc | ||
|
|
d169c3ed4a | ||
|
|
563d11a702 | ||
|
|
31e2ba4696 | ||
|
|
2c3baaad1c | ||
|
|
01bd43aa2b | ||
|
|
f4d095ce0f | ||
|
|
8c5bd9a058 | ||
|
|
335ce1d056 | ||
|
|
0afb3caaed | ||
|
|
8521aaf65c | ||
|
|
0a451cfe0d | ||
|
|
cb0935c96b | ||
|
|
40ec910108 | ||
|
|
2d3c6efae2 | ||
|
|
c027acecf8 | ||
|
|
bc2840e4a7 | ||
|
|
869b472d91 | ||
|
|
81dfc21700 | ||
|
|
a3d8fe2362 | ||
|
|
d2a57f6231 | ||
|
|
9e9ceb3894 | ||
|
|
bd70088877 | ||
|
|
63c9d99b33 | ||
|
|
a2469f48e7 | ||
|
|
62f0f421a9 | ||
|
|
0b24758f41 | ||
|
|
147897e13d | ||
|
|
ade3a5275d | ||
|
|
51c5f945b4 | ||
|
|
a4e2e36243 | ||
|
|
2c5dc13201 | ||
|
|
43234939ca | ||
|
|
87b3d50899 | ||
|
|
da8dbf1777 | ||
|
|
34e1fcccf8 | ||
|
|
c99d1e48bd | ||
|
|
6a428043f6 | ||
|
|
aafe7ff10f | ||
|
|
952e157770 |
No files matched your search
@@ -29,6 +29,20 @@ jobs:
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
|
||||
- name: Set rootfs paths
|
||||
run: |
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
|
||||
|
||||
- name : submodule checkout
|
||||
# Need to update submodules
|
||||
run: |
|
||||
@@ -51,7 +65,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DENABLE_INTERPRETER=True
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DENABLE_INTERPRETER=True -DBUILD_FEX_LINUX_TESTS=True
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
@@ -153,6 +167,17 @@ jobs:
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_APITests.log || true
|
||||
|
||||
- name: FEXLinuxTests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target fex_linux_tests_all
|
||||
|
||||
- name: FEXLinuxTests Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_FEXLinuxTests.log || true
|
||||
|
||||
- name: Truncate test results
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
|
||||
@@ -10,3 +10,4 @@ out/
|
||||
.vscode/
|
||||
.vs/
|
||||
*.pyc
|
||||
.cache
|
||||
@@ -49,3 +49,7 @@
|
||||
shallow = true
|
||||
path = External/robin-map
|
||||
url = https://github.com/Tessil/robin-map.git
|
||||
[submodule "External/Vulkan-Headers"]
|
||||
shallow = true
|
||||
path = External/Vulkan-Headers
|
||||
url = https://github.com/KhronosGroup/Vulkan-Headers.git
|
||||
+24
-139
@@ -1,7 +1,11 @@
|
||||
cmake_minimum_required(VERSION 3.14)
|
||||
project(FEX)
|
||||
|
||||
INCLUDE (CheckIncludeFiles)
|
||||
CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
|
||||
|
||||
option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
|
||||
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
|
||||
option(BUILD_THUNKS "Build thunks" FALSE)
|
||||
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
|
||||
option(ENABLE_IWYU "Enables include what you use program" FALSE)
|
||||
@@ -12,10 +16,10 @@ option(ENABLE_MOLD "Enable linking with mold" FALSE)
|
||||
option(ENABLE_ASAN "Enables Clang ASAN" FALSE)
|
||||
option(ENABLE_TSAN "Enables Clang TSAN" FALSE)
|
||||
option(ENABLE_ASSERTIONS "Enables assertions in build" FALSE)
|
||||
option(ENABLE_GDB_SYMBOLS "Enables GDBSymbols integration support" ${HAVE_GDB_JIT_READER_H})
|
||||
option(ENABLE_VISUAL_DEBUGGER "Enables the visual debugger for compiling" FALSE)
|
||||
option(ENABLE_STRICT_WERROR "Enables stricter -Werror for CI" FALSE)
|
||||
option(ENABLE_WERROR "Enables -Werror" FALSE)
|
||||
option(ENABLE_STATIC_PIE "Enables static-pie build" FALSE)
|
||||
option(ENABLE_JEMALLOC "Enables jemalloc allocator" TRUE)
|
||||
option(ENABLE_OFFLINE_TELEMETRY "Enables FEX offline telemetry" TRUE)
|
||||
option(ENABLE_COMPILE_TIME_TRACE "Enables time trace compile option" FALSE)
|
||||
@@ -24,8 +28,7 @@ option(ENABLE_INTERPRETER "Enables FEX's Interpreter" FALSE)
|
||||
option(ENABLE_CCACHE "Enables ccache for compile caching" TRUE)
|
||||
option(ENABLE_TERMUX_BUILD "Forces building for Termux on a non-Termux build machine" FALSE)
|
||||
|
||||
set (X86_C_COMPILER "x86_64-linux-gnu-gcc" CACHE STRING "c compiler for compiling x86 guest libs")
|
||||
set (X86_CXX_COMPILER "x86_64-linux-gnu-g++" CACHE STRING "c++ compiler for compiling x86 guest libs")
|
||||
set (X86_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86.cmake" CACHE FILEPATH "Toolchain file for the x86 (cross-)compiler")
|
||||
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu" CACHE PATH "global data directory")
|
||||
|
||||
# These options are meant for package management
|
||||
@@ -43,6 +46,12 @@ if (ENABLE_ASSERTIONS)
|
||||
add_definitions(-DASSERTIONS_ENABLED=1)
|
||||
endif()
|
||||
|
||||
if (ENABLE_GDB_SYMBOLS)
|
||||
message(STATUS "GDBSymbols support enabled")
|
||||
add_definitions(-DGDB_SYMBOLS_ENABLED=1)
|
||||
endif()
|
||||
|
||||
|
||||
if (ENABLE_INTERPRETER)
|
||||
message(STATUS "Interpreter enabled")
|
||||
add_definitions(-DINTERPRETER_ENABLED=1)
|
||||
@@ -75,6 +84,7 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
|
||||
set(_M_X86_64 1)
|
||||
add_definitions(-D_M_X86_64=1)
|
||||
set (CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcx16")
|
||||
set (X86_TOOLCHAIN_FILE "")
|
||||
endif()
|
||||
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
|
||||
@@ -129,130 +139,6 @@ if(DEFINED ENV{TERMUX_VERSION} OR ENABLE_TERMUX_BUILD)
|
||||
set(ENABLE_JEMALLOC FALSE)
|
||||
endif()
|
||||
|
||||
if (ENABLE_STATIC_PIE)
|
||||
if (_M_ARM_64 AND ENABLE_LLD)
|
||||
message (FATAL_ERROR "Static linking does not currently work with AArch64+lld. Use GNU ld for now.")
|
||||
endif()
|
||||
|
||||
file(WRITE ${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp/Determine_iplt.c
|
||||
"int main(int argc, char* argv[])
|
||||
{
|
||||
return 0;
|
||||
}")
|
||||
|
||||
# Compile the test application with our LD_OVERRIDE and static-pie options
|
||||
try_compile(
|
||||
COMPILE_RESULT
|
||||
${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp
|
||||
${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp/Determine_iplt.c
|
||||
COMPILE_DEFINITIONS "-fPIE ${LD_OVERRIDE}"
|
||||
LINK_LIBRARIES "-static-pie ${LD_OVERRIDE}"
|
||||
COPY_FILE ${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp/Determine_iplt
|
||||
)
|
||||
|
||||
if (${COMPILE_RESULT})
|
||||
# Read the symbols from the elf
|
||||
execute_process(COMMAND
|
||||
readelf -s ${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp/Determine_iplt
|
||||
OUTPUT_FILE ${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp/plt_out.txt
|
||||
OUTPUT_VARIABLE PLT_SYMBOLS)
|
||||
|
||||
# Pull out the __rela_iplt_{start,end} symbols if they exist
|
||||
execute_process(COMMAND
|
||||
"grep" "__rela_iplt" ${PROJECT_BINARY_DIR}/CMakeFiles/CMakeTmp/plt_out.txt
|
||||
OUTPUT_VARIABLE PLT_SYMBOLS)
|
||||
|
||||
set (SYMBOLS_FINE TRUE)
|
||||
set (HAS_IPLT -1)
|
||||
# Check if we have any symbols in our grep output
|
||||
# The symbols must either not exist at all OR the symbols are zero
|
||||
if (PLT_SYMBOLS)
|
||||
string(FIND ${PLT_SYMBOLS} "__rela_iplt_start" HAS_IPLT)
|
||||
endif()
|
||||
|
||||
if (NOT HAS_IPLT EQUAL -1)
|
||||
# We have some symbols from readelf. Let's parse the results to check if they are zero
|
||||
# Format: '35: 0000000000000000 0 NOTYPE LOCAL HIDDEN UND __rela_iplt_start'
|
||||
string(REPLACE "\n" ";" SYMBOL_LIST ${PLT_SYMBOLS})
|
||||
foreach (SYMBOL ${SYMBOL_LIST})
|
||||
# strip any leading and trailing whitespace
|
||||
string (STRIP ${SYMBOL} SYMBOL)
|
||||
# Convert string to a list
|
||||
string(REPLACE " " ";" SYMBOL_VALUES ${SYMBOL}})
|
||||
# Pull out the address argument
|
||||
list(GET SYMBOL_VALUES 1 OFFSET)
|
||||
|
||||
# Check against integer zero
|
||||
if (NOT ${OFFSET} EQUAL 0)
|
||||
# Symbol wasn't zero, this now fails
|
||||
set (SYMBOLS_FINE FALSE)
|
||||
endif()
|
||||
endforeach()
|
||||
endif()
|
||||
|
||||
if (SYMBOLS_FINE)
|
||||
# We can now exnable static-pie
|
||||
set (STATIC_PIE_OPTIONS "-static-pie")
|
||||
# Pthreads has an issue with exposing symbols
|
||||
# We need to make some concessions to the pthread gods
|
||||
if (ENABLE_LLD)
|
||||
set (PTHREAD_LIB
|
||||
-Wl,--undefined-glob=pthread_*
|
||||
-Wl,--undefined=__cxa_finalize
|
||||
-Wl,--undefined=_pthread_cleanup_push_defer
|
||||
-Wl,--undefined=_pthread_cleanup_pop_restore
|
||||
-Wl,--undefined=__pthread_cleanup_upto
|
||||
pthread)
|
||||
else()
|
||||
set (PTHREAD_LIB
|
||||
-Wl,--undefined=pthread_join
|
||||
-Wl,--undefined=pthread_attr_getdetachstate
|
||||
-Wl,--undefined=pthread_sigmask
|
||||
-Wl,--undefined=pthread_mutex_lock
|
||||
-Wl,--undefined=pthread_cond_init
|
||||
-Wl,--undefined=pthread_attr_init
|
||||
-Wl,--undefined=pthread_mutex_unlock
|
||||
-Wl,--undefined=pthread_mutexattr_destroy
|
||||
-Wl,--undefined=pthread_detach
|
||||
-Wl,--undefined=pthread_mutex_init
|
||||
-Wl,--undefined=pthread_getattr_np
|
||||
-Wl,--undefined=pthread_cond_timedwait
|
||||
-Wl,--undefined=pthread_attr_destroy
|
||||
-Wl,--undefined=pthread_mutexattr_settype
|
||||
-Wl,--undefined=pthread_rwlock_unlock
|
||||
-Wl,--undefined=pthread_rwlock_wrlock
|
||||
-Wl,--undefined=pthread_setspecific
|
||||
-Wl,--undefined=pthread_create
|
||||
-Wl,--undefined=pthread_cond_clockwait
|
||||
-Wl,--undefined=pthread_key_create
|
||||
-Wl,--undefined=pthread_rwlock_rdlock
|
||||
-Wl,--undefined=pthread_setname_np
|
||||
-Wl,--undefined=pthread_cond_signal
|
||||
-Wl,--undefined=pthread_mutexattr_init
|
||||
-Wl,--undefined=pthread_attr_setstack
|
||||
-Wl,--undefined=pthread_self
|
||||
-Wl,--undefined=pthread_getaffinity_np
|
||||
-Wl,--undefined=pthread_cond_wait
|
||||
-Wl,--undefined=pthread_mutex_trylock
|
||||
-Wl,--undefined=pthread_cond_broadcast
|
||||
-Wl,--undefined=pthread_cond_destroy
|
||||
-Wl,--undefined=pthread_getspecific
|
||||
-Wl,--undefined=pthread_key_delete
|
||||
-Wl,--undefined=pthread_once
|
||||
-Wl,--undefined=__cxa_finalize
|
||||
-Wl,--undefined=_pthread_cleanup_push_defer
|
||||
-Wl,--undefined=_pthread_cleanup_pop_restore
|
||||
-Wl,--undefined=__pthread_cleanup_upto
|
||||
pthread)
|
||||
endif()
|
||||
else()
|
||||
message (FATAL_ERROR "Application has __rela_iplt_{start,end} symbols. Which means static-pie can't be enabled")
|
||||
endif()
|
||||
else()
|
||||
message (FATAL_ERROR "Couldn't compile static-pie test. Static-pie can't be enabled! Is your glibc compiled without static-pie?")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (ENABLE_ASAN)
|
||||
add_definitions(-DENABLE_ASAN=1)
|
||||
add_compile_options(-fno-omit-frame-pointer -fsanitize=address -fsanitize-address-use-after-scope)
|
||||
@@ -488,15 +374,20 @@ add_subdirectory(Source/)
|
||||
add_subdirectory(Data/AppConfig/)
|
||||
|
||||
# Install the ThunksDB file
|
||||
install(
|
||||
FILES ${CMAKE_CURRENT_SOURCE_DIR}/Data/ThunksDB.json
|
||||
DESTINATION ${DATA_DIRECTORY}/)
|
||||
file(GLOB CONFIG_SOURCES CONFIGURE_DEPENDS ${CMAKE_CURRENT_SOURCE_DIR}/Data/*.json)
|
||||
|
||||
# Any application configuration json file gets installed
|
||||
foreach(CONFIG_SRC ${CONFIG_SOURCES})
|
||||
install(FILES ${CONFIG_SRC}
|
||||
DESTINATION ${DATA_DIRECTORY}/)
|
||||
endforeach()
|
||||
|
||||
if (BUILD_TESTS)
|
||||
add_subdirectory(unittests/)
|
||||
endif()
|
||||
|
||||
if (BUILD_THUNKS)
|
||||
set (FEX_PROJECT_SOURCE_DIR ${PROJECT_SOURCE_DIR})
|
||||
add_subdirectory(ThunkLibs/Generator)
|
||||
|
||||
# Thunk targets for both host libraries and IDE integration
|
||||
@@ -513,10 +404,10 @@ if (BUILD_THUNKS)
|
||||
BINARY_DIR "Guest"
|
||||
CMAKE_ARGS
|
||||
"-DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}"
|
||||
"-DX86_C_COMPILER:STRING=${X86_C_COMPILER}"
|
||||
"-DX86_CXX_COMPILER:STRING=${X86_CXX_COMPILER}"
|
||||
"-DCMAKE_TOOLCHAIN_FILE:FILEPATH=${X86_TOOLCHAIN_FILE}"
|
||||
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
|
||||
"-DSTRUCT_VERIFIER=${CMAKE_SOURCE_DIR}/Scripts/StructPackVerifier.py"
|
||||
"-DFEX_PROJECT_SOURCE_DIR=${FEX_PROJECT_SOURCE_DIR}"
|
||||
"-DGENERATOR_EXE=$<TARGET_FILE:thunkgen>"
|
||||
INSTALL_COMMAND ""
|
||||
BUILD_ALWAYS ON
|
||||
@@ -583,13 +474,7 @@ endif()
|
||||
|
||||
# Package creation
|
||||
set (CPACK_GENERATOR "DEB")
|
||||
if (ENABLE_STATIC_PIE)
|
||||
set (CPACK_PACKAGE_NAME fex-emu-static)
|
||||
set (CPACK_DEBIAN_PACKAGE_CONFLICTS "fex-emu")
|
||||
else()
|
||||
set (CPACK_PACKAGE_NAME fex-emu)
|
||||
set (CPACK_DEBIAN_PACKAGE_CONFLICTS "fex-emu-static")
|
||||
endif()
|
||||
set (CPACK_PACKAGE_NAME fex-emu)
|
||||
set (CPACK_PACKAGE_FILE_NAME "${CPACK_PACKAGE_NAME}-${GIT_DESCRIBE_STRING}_${CMAKE_SYSTEM_PROCESSOR}")
|
||||
set (CPACK_PACKAGE_CONTACT "FEX-Emu Maintainers <team@fex-emu.org>")
|
||||
set (CPACK_PACKAGE_VERSION_MAJOR "${FEX_VERSION_MAJOR}")
|
||||
|
||||
@@ -11,15 +11,15 @@ endforeach()
|
||||
# First generate then install it
|
||||
foreach(GEN_CONFIG_SRC ${GEN_CONFIG_SOURCES})
|
||||
# Get the filename only component
|
||||
get_filename_component(CONFIG_NAME ${GEN_CONFIG_SRC} NAME_WE)
|
||||
get_filename_component(CONFIG_NAME ${GEN_CONFIG_SRC} NAME_WLE)
|
||||
|
||||
# Configure it
|
||||
configure_file(
|
||||
${GEN_CONFIG_SRC}
|
||||
${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}.json)
|
||||
${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME})
|
||||
|
||||
# Then install the configured json
|
||||
install(
|
||||
FILES ${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}.json
|
||||
FILES ${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}
|
||||
DESTINATION ${DATA_DIRECTORY}/AppConfig/)
|
||||
endforeach()
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"Config": {
|
||||
"x86dec_SynchronizeRIPOnAllBlocks": "1"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"Config": {
|
||||
"x86dec_SynchronizeRIPOnAllBlocks": "1"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"Config": {
|
||||
"x86dec_SynchronizeRIPOnAllBlocks": "1"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"Config": {
|
||||
"x86dec_SynchronizeRIPOnAllBlocks": "1"
|
||||
}
|
||||
}
|
||||
+60
-239
@@ -6,18 +6,10 @@
|
||||
"X11"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libGL.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libGL.so.1",
|
||||
"/usr/lib/x86_64-linux-gnu/libGL.so.1.2.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libGL.so.1.7.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGL.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGL.so.1",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGL.so.1.2.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGL.so.1.7.0",
|
||||
"/lib/x86_64-linux-gnu/libGL.so",
|
||||
"/lib/x86_64-linux-gnu/libGL.so.1",
|
||||
"/lib/x86_64-linux-gnu/libGL.so.1.2.0",
|
||||
"/lib/x86_64-linux-gnu/libGL.so.1.7.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGL.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGL.so.1",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGL.so.1.2.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGL.so.1.7.0"
|
||||
]
|
||||
},
|
||||
"GLESv2": {
|
||||
@@ -26,322 +18,151 @@
|
||||
"X11"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libGLESv2.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libGLESv2.so.2",
|
||||
"/usr/lib/x86_64-linux-gnu/libGLESv2.so.2.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGLESv2.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGLESv2.so.2",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGLESv2.so.2.0.0",
|
||||
"/lib/x86_64-linux-gnu/libGLESv2.so",
|
||||
"/lib/x86_64-linux-gnu/libGLESv2.so.2",
|
||||
"/lib/x86_64-linux-gnu/libGLESv2.so.2.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGLESv2.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGLESv2.so.2",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libGLESv2.so.2.0.0"
|
||||
]
|
||||
},
|
||||
"X11": {
|
||||
"Library": "libX11-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libX11.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libX11.so.6",
|
||||
"/usr/lib/x86_64-linux-gnu/libX11.so.6.4.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libX11.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libX11.so.6",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libX11.so.6.4.0",
|
||||
"/lib/x86_64-linux-gnu/libX11.so",
|
||||
"/lib/x86_64-linux-gnu/libX11.so.6",
|
||||
"/lib/x86_64-linux-gnu/libX11.so.6.4.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libX11.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libX11.so.6",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libX11.so.6.4.0"
|
||||
]
|
||||
},
|
||||
"Vulkan-radeon": {
|
||||
"Library": "libvulkan_radeon-guest.so",
|
||||
"Vulkan": {
|
||||
"Library": "libvulkan-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libvulkan_radeon.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libvulkan_radeon.so",
|
||||
"/lib/x86_64-linux-gnu/libvulkan_radeon.so"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libvulkan.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libvulkan.so.1",
|
||||
"@HOME@/.local/share/Steam/ubuntu12_32/steam-runtime/pinned_libs_64/libvulkan.so.1"
|
||||
],
|
||||
"Comment": [
|
||||
"Vulkan library relies on xcb, otherwise it crashes with jemalloc"
|
||||
]
|
||||
},
|
||||
"Vulkan-lavapipe": {
|
||||
"Library": "libvulkan_lvp-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libvulkan_lvp.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libvulkan_lvp.so",
|
||||
"/lib/x86_64-linux-gnu/libvulkan_lvp.so"
|
||||
]
|
||||
},
|
||||
"Vulkan-freedreno": {
|
||||
"Library": "libvulkan_freedreno-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libvulkan_freedreno.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libvulkan_freedreno.so",
|
||||
"/lib/x86_64-linux-gnu/libvulkan_freedreno.so"
|
||||
]
|
||||
},
|
||||
"Vulkan-intel": {
|
||||
"Library": "libvulkan_intel-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libvulkan_intel.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libvulkan_intel.so",
|
||||
"/lib/x86_64-linux-gnu/libvulkan_intel.so"
|
||||
]
|
||||
},
|
||||
"Vulkan-panfrost": {
|
||||
"Library": "libvulkan_panfrost-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libvulkan_panfrost.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libvulkan_panfrost.so",
|
||||
"/lib/x86_64-linux-gnu/libvulkan_panfrost.so"
|
||||
]
|
||||
},
|
||||
"Vulkan-nvidia": {
|
||||
"Library": "libvulkan_nvidia-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libGLX_nvidia.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libGLX_nvidia.so.0",
|
||||
"/lib/x86_64-linux-gnu/libGLX_nvidia.so.0"
|
||||
],
|
||||
"Comment": [
|
||||
"Not currently wired up"
|
||||
]
|
||||
},
|
||||
"Vulkan-virtio": {
|
||||
"Library": "libvulkan_virtio-guest.so",
|
||||
"Depends": [
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libvulkan_virtio.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libvulkan_virtio.so",
|
||||
"/lib/x86_64-linux-gnu/libvulkan_virtio.so"
|
||||
]
|
||||
},
|
||||
"xcb": {
|
||||
"Library": "libxcb-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb.so.1",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb.so.1.1.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb.so.1",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb.so.1.1.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb.so.1",
|
||||
"/lib/x86_64-linux-gnu/libxcb.so.1.1.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb.so.1",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb.so.1.1.0"
|
||||
]
|
||||
},
|
||||
"xcb-dri2": {
|
||||
"Library": "libxcb_dri2-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-dri2.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-dri2.so.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-dri2.so.0.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri2.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri2.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri2.so.0.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-dri2.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-dri2.so.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-dri2.so.0.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri2.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri2.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri2.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-dri3": {
|
||||
"Library": "libxcb_dri3-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-dri3.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-dri3.so.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-dri3.so.0.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri3.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri3.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-dri3.so.0.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-dri3.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-dri3.so.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-dri3.so.0.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri3.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri3.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-dri3.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-xfixes": {
|
||||
"Library": "libxcb_xfixes-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-xfixes.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-xfixes.so.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-xfixes.so.0.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-xfixes.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-xfixes.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-xfixes.so.0.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-xfixes.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-xfixes.so.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-xfixes.so.0.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-xfixes.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-xfixes.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-xfixes.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-shm": {
|
||||
"Library": "libxcb_shm-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-shm.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-shm.so.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-shm.so.0.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-shm.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-shm.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-shm.so.0.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-shm.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-shm.so.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-shm.so.0.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-shm.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-shm.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-shm.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-sync": {
|
||||
"Library": "libxcb_sync-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-sync.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-sync.so.1",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-sync.so.1.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-sync.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-sync.so.1",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-sync.so.1.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-sync.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-sync.so.1",
|
||||
"/lib/x86_64-linux-gnu/libxcb-sync.so.1.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-sync.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-sync.so.1",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-sync.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-randr": {
|
||||
"Library": "libxcb_randr-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-randr.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-randr.so.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-randr.so.0.1.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-randr.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-randr.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-randr.so.0.1.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-randr.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-randr.so.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-randr.so.0.1.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-randr.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-randr.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-randr.so.0.1.0"
|
||||
]
|
||||
},
|
||||
"xcb-present": {
|
||||
"Library": "libxcb_present-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-present.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-present.so.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-present.so.0.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-present.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-present.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-present.so.0.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-present.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-present.so.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-present.so.0.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-present.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-present.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-present.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-glx": {
|
||||
"Library": "libxcb_glx-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-glx.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-glx.so.0",
|
||||
"/usr/lib/x86_64-linux-gnu/libxcb-glx.so.0.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-glx.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-glx.so.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxcb-glx.so.0.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-glx.so",
|
||||
"/lib/x86_64-linux-gnu/libxcb-glx.so.0",
|
||||
"/lib/x86_64-linux-gnu/libxcb-glx.so.0.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-glx.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-glx.so.0",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxcb-glx.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xshmfence": {
|
||||
"Library": "libshmfence-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libxshmfence.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libxshmfence.so.1",
|
||||
"/usr/lib/x86_64-linux-gnu/libxshmfence.so.1.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxshmfence.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxshmfence.so.1",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libxshmfence.so.1.0.0",
|
||||
"/lib/x86_64-linux-gnu/libxshmfence.so",
|
||||
"/lib/x86_64-linux-gnu/libxshmfence.so.1",
|
||||
"/lib/x86_64-linux-gnu/libxshmfence.so.1.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxshmfence.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxshmfence.so.1",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libxshmfence.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"drm": {
|
||||
"Library": "libdrm-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libdrm.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libdrm.so.2",
|
||||
"/usr/lib/x86_64-linux-gnu/libdrm.so.2.4.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libdrm.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libdrm.so.2",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libdrm.so.2.4.0",
|
||||
"/lib/x86_64-linux-gnu/libdrm.so",
|
||||
"/lib/x86_64-linux-gnu/libdrm.so.2",
|
||||
"/lib/x86_64-linux-gnu/libdrm.so.2.4.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libdrm.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libdrm.so.2",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libdrm.so.2.4.0"
|
||||
]
|
||||
},
|
||||
"asound": {
|
||||
"Library": "libasound-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libasound.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libasound.so.2",
|
||||
"/usr/lib/x86_64-linux-gnu/libasound.so.2.0.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libasound.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libasound.so.2",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libasound.so.2.0.0",
|
||||
"/lib/x86_64-linux-gnu/libasound.so",
|
||||
"/lib/x86_64-linux-gnu/libasound.so.2",
|
||||
"/lib/x86_64-linux-gnu/libasound.so.2.0.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libasound.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libasound.so.2",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libasound.so.2.0.0"
|
||||
]
|
||||
},
|
||||
"Xrender": {
|
||||
"Library": "libXrender-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libXrender.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libXrender.so.1",
|
||||
"/usr/lib/x86_64-linux-gnu/libXrender.so.1.3.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXrender.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXrender.so.1",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXrender.so.1.3.0",
|
||||
"/lib/x86_64-linux-gnu/libXrender.so",
|
||||
"/lib/x86_64-linux-gnu/libXrender.so.1",
|
||||
"/lib/x86_64-linux-gnu/libXrender.so.1.3.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXrender.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXrender.so.1",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXrender.so.1.3.0"
|
||||
]
|
||||
},
|
||||
"Xext": {
|
||||
"Library": "libXext-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libXext.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libXext.so.6",
|
||||
"/usr/lib/x86_64-linux-gnu/libXext.so.6.4.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXext.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXext.so.6",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXext.so.6.4.0",
|
||||
"/lib/x86_64-linux-gnu/libXext.so",
|
||||
"/lib/x86_64-linux-gnu/libXext.so.6",
|
||||
"/lib/x86_64-linux-gnu/libXext.so.6.4.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXext.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXext.so.6",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXext.so.6.4.0"
|
||||
]
|
||||
},
|
||||
"Xfixes": {
|
||||
"Library": "libXfixes-guest.so",
|
||||
"Overlay": [
|
||||
"/usr/lib/x86_64-linux-gnu/libXfixes.so",
|
||||
"/usr/lib/x86_64-linux-gnu/libXfixes.so.3",
|
||||
"/usr/lib/x86_64-linux-gnu/libXfixes.so.3.1.0",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXfixes.so",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXfixes.so.3",
|
||||
"/usr/local/lib/x86_64-linux-gnu/libXfixes.so.3.1.0",
|
||||
"/lib/x86_64-linux-gnu/libXfixes.so",
|
||||
"/lib/x86_64-linux-gnu/libXfixes.so.3",
|
||||
"/lib/x86_64-linux-gnu/libXfixes.so.3.1.0"
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXfixes.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXfixes.so.3",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXfixes.so.3.1.0"
|
||||
]
|
||||
},
|
||||
"":{}
|
||||
|
||||
-10
@@ -321,8 +321,6 @@ def print_ir_structs(defines):
|
||||
|
||||
|
||||
if op.SSAArgNum > 0:
|
||||
# Add helpers for accessing SSA arguments, given how frequently they're accessed
|
||||
|
||||
output_file.write("\t// Get index of argument by name\n")
|
||||
SSAArg = 0
|
||||
for arg in op.Arguments:
|
||||
@@ -330,14 +328,6 @@ def print_ir_structs(defines):
|
||||
output_file.write("\tstatic constexpr size_t {}_Index = {};\n".format(arg.Name, SSAArg))
|
||||
SSAArg = SSAArg + 1
|
||||
|
||||
output_file.write("\n")
|
||||
output_file.write("\t[[nodiscard]] OrderedNodeWrapper& Args(size_t Index) {\n")
|
||||
output_file.write("\t\treturn Header.Args[Index];\n")
|
||||
output_file.write("\t}\n")
|
||||
output_file.write("\t[[nodiscard]] const OrderedNodeWrapper& Args(size_t Index) const {\n")
|
||||
output_file.write("\t\treturn Header.Args[Index];\n")
|
||||
output_file.write("\t}\n")
|
||||
|
||||
|
||||
output_file.write("};\n")
|
||||
|
||||
|
||||
+7
-8
@@ -81,6 +81,7 @@ set (SRCS
|
||||
Interface/Core/LookupCache.cpp
|
||||
Interface/Core/BlockSamplingData.cpp
|
||||
Interface/Core/Core.cpp
|
||||
Interface/Core/CPUBackend.cpp
|
||||
Interface/Core/CPUID.cpp
|
||||
Interface/Core/Frontend.cpp
|
||||
Interface/Core/GdbServer.cpp
|
||||
@@ -117,6 +118,7 @@ set (SRCS
|
||||
Interface/Core/X86Tables/X87Tables.cpp
|
||||
Interface/Core/X86Tables/XOPTables.cpp
|
||||
Interface/HLE/Thunks/Thunks.cpp
|
||||
Interface/GDBJIT/GDBJIT.cpp
|
||||
Interface/IR/AOTIR.cpp
|
||||
Interface/IR/IRDumper.cpp
|
||||
Interface/IR/IRParser.cpp
|
||||
@@ -206,7 +208,9 @@ if (ENABLE_JIT_ARM64)
|
||||
Interface/Core/JIT/Arm64/MemoryOps.cpp
|
||||
Interface/Core/JIT/Arm64/MiscOps.cpp
|
||||
Interface/Core/JIT/Arm64/MoveOps.cpp
|
||||
Interface/Core/JIT/Arm64/VectorOps.cpp)
|
||||
Interface/Core/JIT/Arm64/VectorOps.cpp
|
||||
Interface/Core/JIT/Arm64/Arm64Relocations.cpp
|
||||
)
|
||||
endif()
|
||||
|
||||
set (LIBS vixl dl xxhash tiny-json)
|
||||
@@ -224,13 +228,11 @@ set(OUTPUT_IR_FOLDER "${CMAKE_BINARY_DIR}/include/FEXCore/IR")
|
||||
set(OUTPUT_NAME "${OUTPUT_IR_FOLDER}/IRDefines.inc")
|
||||
set(INPUT_NAME "${CMAKE_CURRENT_SOURCE_DIR}/Interface/IR/IR.json")
|
||||
|
||||
add_custom_target(CREATE_IR_FOLDER ALL
|
||||
COMMAND ${CMAKE_COMMAND} -E make_directory "${OUTPUT_IR_FOLDER}")
|
||||
file(MAKE_DIRECTORY "${OUTPUT_IR_FOLDER}")
|
||||
|
||||
add_custom_command(
|
||||
OUTPUT "${OUTPUT_NAME}"
|
||||
DEPENDS "${INPUT_NAME}"
|
||||
DEPENDS CREATE_IR_FOLDER
|
||||
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py"
|
||||
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_generator.py" "${INPUT_NAME}" "${OUTPUT_NAME}"
|
||||
)
|
||||
@@ -244,7 +246,6 @@ set(OUTPUT_IR_DOC "${CMAKE_BINARY_DIR}/IR.md")
|
||||
add_custom_command(
|
||||
OUTPUT "${OUTPUT_IR_DOC}"
|
||||
DEPENDS "${INPUT_NAME}"
|
||||
DEPENDS CREATE_IR_FOLDER
|
||||
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_doc_generator.py"
|
||||
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/json_ir_doc_generator.py" "${INPUT_NAME}" "${OUTPUT_IR_DOC}"
|
||||
)
|
||||
@@ -265,15 +266,13 @@ set(INPUT_CONFIG_NAME "${CMAKE_BINARY_DIR}/generated/Config/Config.json")
|
||||
set(OUTPUT_MAN_NAME "${CMAKE_BINARY_DIR}/generated/FEX.1")
|
||||
set(OUTPUT_MAN_NAME_COMPRESS "${CMAKE_BINARY_DIR}/generated/FEX.1.gz")
|
||||
|
||||
add_custom_target(CREATE_CONFIG_FOLDER ALL
|
||||
COMMAND ${CMAKE_COMMAND} -E make_directory "${OUTPUT_CONFIG_FOLDER}")
|
||||
file(MAKE_DIRECTORY "${OUTPUT_CONFIG_FOLDER}")
|
||||
|
||||
add_custom_command(
|
||||
OUTPUT "${OUTPUT_CONFIG_NAME}"
|
||||
OUTPUT "${OUTPUT_CONFIG_OPTION_NAME}"
|
||||
OUTPUT "${OUTPUT_MAN_NAME}"
|
||||
DEPENDS "${INPUT_CONFIG_NAME}"
|
||||
DEPENDS CREATE_CONFIG_FOLDER
|
||||
DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/config_generator.py"
|
||||
COMMAND "python3" "${CMAKE_CURRENT_SOURCE_DIR}/../Scripts/config_generator.py" "${INPUT_CONFIG_NAME}" "${OUTPUT_CONFIG_NAME}" "${OUTPUT_MAN_NAME}"
|
||||
"${OUTPUT_CONFIG_OPTION_NAME}"
|
||||
|
||||
+3
-3
@@ -20,12 +20,12 @@ struct BitSet final {
|
||||
ElementType *Memory;
|
||||
void Allocate(size_t Elements) {
|
||||
size_t AllocateSize = AlignUp(Elements, MinimumSizeBits) / MinimumSize;
|
||||
LOGMAN_THROW_A_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
LOGMAN_THROW_AA_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
Memory = static_cast<ElementType*>(FEXCore::Allocator::malloc(AllocateSize));
|
||||
}
|
||||
void Realloc(size_t Elements) {
|
||||
size_t AllocateSize = AlignUp(Elements, MinimumSizeBits) / MinimumSize;
|
||||
LOGMAN_THROW_A_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
LOGMAN_THROW_AA_FMT((AllocateSize * MinimumSize) >= Elements, "Fail");
|
||||
Memory = static_cast<ElementType*>(FEXCore::Allocator::realloc(Memory, AllocateSize));
|
||||
}
|
||||
void Free() {
|
||||
@@ -64,7 +64,7 @@ struct BitSetView final {
|
||||
ElementType *Memory;
|
||||
|
||||
void GetView(BitSet<T> &Set, uint64_t ElementOffset) {
|
||||
LOGMAN_THROW_A_FMT((ElementOffset % MinimumSize) == 0,
|
||||
LOGMAN_THROW_AA_FMT((ElementOffset % MinimumSize) == 0,
|
||||
"Bitset view offset needs to be aligned to size of backing element");
|
||||
Memory = &Set.Memory[ElementOffset / MinimumSizeBits];
|
||||
}
|
||||
|
||||
+5
-2
@@ -7,6 +7,11 @@
|
||||
|
||||
namespace FEXCore {
|
||||
JITSymbols::JITSymbols() : fp{nullptr, std::fclose} {
|
||||
}
|
||||
|
||||
JITSymbols::~JITSymbols() = default;
|
||||
|
||||
void JITSymbols::InitFile() {
|
||||
const auto PerfMap = fmt::format("/tmp/perf-{}.map", getpid());
|
||||
|
||||
fp.reset(fopen(PerfMap.c_str(), "wb"));
|
||||
@@ -16,8 +21,6 @@ namespace FEXCore {
|
||||
}
|
||||
}
|
||||
|
||||
JITSymbols::~JITSymbols() = default;
|
||||
|
||||
void JITSymbols::Register(const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize) {
|
||||
if (!fp) return;
|
||||
|
||||
|
||||
+1
@@ -11,6 +11,7 @@ public:
|
||||
JITSymbols();
|
||||
~JITSymbols();
|
||||
|
||||
void InitFile();
|
||||
void Register(const void *HostAddr, uint64_t GuestAddr, uint32_t CodeSize);
|
||||
void Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name);
|
||||
void Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name, uintptr_t Offset);
|
||||
|
||||
+38
-12
@@ -154,15 +154,20 @@ namespace JSON {
|
||||
return ConfigDir;
|
||||
}
|
||||
|
||||
std::string GetConfigFileLocation() {
|
||||
std::string GetConfigFileLocation(bool Global) {
|
||||
std::string ConfigFile{};
|
||||
const char *AppConfig = getenv("FEX_APP_CONFIG");
|
||||
if (AppConfig) {
|
||||
// App config environment variable overwrites only the config file
|
||||
ConfigFile = AppConfig;
|
||||
if (Global) {
|
||||
ConfigFile = GetConfigDirectory(true) + "Config.json";
|
||||
}
|
||||
else {
|
||||
ConfigFile = GetConfigDirectory(false) + "Config.json";
|
||||
const char *AppConfig = getenv("FEX_APP_CONFIG");
|
||||
if (AppConfig) {
|
||||
// App config environment variable overwrites only the config file
|
||||
ConfigFile = AppConfig;
|
||||
}
|
||||
else {
|
||||
ConfigFile = GetConfigDirectory(false) + "Config.json";
|
||||
}
|
||||
}
|
||||
return ConfigFile;
|
||||
}
|
||||
@@ -206,7 +211,8 @@ namespace JSON {
|
||||
static std::map<FEXCore::Config::LayerType, std::unique_ptr<FEXCore::Config::Layer>> ConfigLayers;
|
||||
static FEXCore::Config::Layer *Meta{};
|
||||
|
||||
constexpr std::array<FEXCore::Config::LayerType, 6> LoadOrder = {
|
||||
constexpr std::array<FEXCore::Config::LayerType, 7> LoadOrder = {
|
||||
FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN,
|
||||
FEXCore::Config::LayerType::LAYER_MAIN,
|
||||
FEXCore::Config::LayerType::LAYER_GLOBAL_APP,
|
||||
FEXCore::Config::LayerType::LAYER_LOCAL_APP,
|
||||
@@ -371,6 +377,22 @@ namespace JSON {
|
||||
return {};
|
||||
}
|
||||
|
||||
|
||||
std::string FindContainer() {
|
||||
// We only support pressure-vessel at the moment
|
||||
const static std::string ContainerManager = "/run/host/container-manager";
|
||||
if (std::filesystem::exists(ContainerManager)) {
|
||||
std::vector<char> Manager{};
|
||||
if (FEXCore::FileLoading::LoadFile(Manager, ContainerManager)) {
|
||||
// Trim the whitespace, may contain a newline
|
||||
std::string ManagerStr = Manager.data();
|
||||
ManagerStr = FEXCore::StringUtils::Trim(ManagerStr);
|
||||
return ManagerStr;
|
||||
}
|
||||
}
|
||||
return {};
|
||||
}
|
||||
|
||||
std::string FindContainerPrefix() {
|
||||
// We only support pressure-vessel at the moment
|
||||
const static std::string ContainerManager = "/run/host/container-manager";
|
||||
@@ -597,7 +619,7 @@ namespace JSON {
|
||||
// Application loaders
|
||||
class MainLoader final : public FEXCore::Config::OptionMapper {
|
||||
public:
|
||||
explicit MainLoader();
|
||||
explicit MainLoader(FEXCore::Config::LayerType Type);
|
||||
explicit MainLoader(std::string ConfigFile);
|
||||
void Load() override;
|
||||
|
||||
@@ -643,9 +665,9 @@ namespace JSON {
|
||||
}
|
||||
}
|
||||
|
||||
MainLoader::MainLoader()
|
||||
: FEXCore::Config::OptionMapper(FEXCore::Config::LayerType::LAYER_MAIN)
|
||||
, Config{FEXCore::Config::GetConfigFileLocation()} {
|
||||
MainLoader::MainLoader(FEXCore::Config::LayerType Type)
|
||||
: FEXCore::Config::OptionMapper(Type)
|
||||
, Config{FEXCore::Config::GetConfigFileLocation(Type == FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN)} {
|
||||
}
|
||||
|
||||
MainLoader::MainLoader(std::string ConfigFile)
|
||||
@@ -719,12 +741,16 @@ namespace JSON {
|
||||
}
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateGlobalMainLayer() {
|
||||
return std::make_unique<FEXCore::Config::MainLoader>(FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN);
|
||||
}
|
||||
|
||||
std::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(std::string const *File) {
|
||||
if (File) {
|
||||
return std::make_unique<FEXCore::Config::MainLoader>(*File);
|
||||
}
|
||||
else {
|
||||
return std::make_unique<FEXCore::Config::MainLoader>();
|
||||
return std::make_unique<FEXCore::Config::MainLoader>(FEXCore::Config::LayerType::LAYER_MAIN);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+37
-11
@@ -49,6 +49,13 @@
|
||||
"Cache JIT object code to drive.",
|
||||
"Allows JIT code to be shared between applications"
|
||||
]
|
||||
},
|
||||
"EnableAVX": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"Determines whether or not we use the expanded register file for AVX or not"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Emulation": {
|
||||
@@ -208,6 +215,16 @@
|
||||
"Useful for determining hot blocks of code",
|
||||
"Has some file writing overhead per JIT block"
|
||||
]
|
||||
},
|
||||
"GDBSymbols": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Integrates with GDB using the JIT interface.",
|
||||
"Needs the fex jit loader in GDB, which can be loaded via `jit-reader-load libFEXGDBReader.so.`",
|
||||
"Also needs x86_64-linux-gnu-objdump in PATH.",
|
||||
"Can be very slow."
|
||||
]
|
||||
}
|
||||
},
|
||||
"Logging": {
|
||||
@@ -219,22 +236,13 @@
|
||||
"Disables logging"
|
||||
]
|
||||
},
|
||||
"OutputSocket": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"Desc": [
|
||||
"Socket to connect to",
|
||||
"eg: localhost:8087",
|
||||
"If set will override the OutputLog location"
|
||||
]
|
||||
},
|
||||
"OutputLog": {
|
||||
"Type": "str",
|
||||
"Default": "stderr",
|
||||
"Default": "server",
|
||||
"ShortArg": "o",
|
||||
"Desc": [
|
||||
"File to write FEX output to.",
|
||||
"[stdout, stderr, <Filename>]"
|
||||
"[stdout, stderr, server, <Filename>]"
|
||||
]
|
||||
}
|
||||
},
|
||||
@@ -260,6 +268,14 @@
|
||||
"Highly likely to break any multithreaded application if disabled."
|
||||
]
|
||||
},
|
||||
"TSOAutoMigration": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"Automatically enables TSO when shared memory is used.",
|
||||
"Should work without issues in most cases."
|
||||
]
|
||||
},
|
||||
"X87ReducedPrecision": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
@@ -300,6 +316,16 @@
|
||||
"Forces a process to stall out on initialization",
|
||||
"Useful for a process that keeps restarting and doesn't work"
|
||||
]
|
||||
},
|
||||
"x86dec_SynchronizeRIPOnAllBlocks": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"An application that uses try-catch or longjump extensively needs the ability to do context aware state flushing",
|
||||
"In the case of FEX's block-linking, it won't always ensure that RIP is synchronized.",
|
||||
"If an exception occurs and RIP isn't synchronized, then FEX's exception stack restore may not long jump as expected",
|
||||
"Can be useful for Wine applications that rely on stack unwinding"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Misc": {
|
||||
|
||||
+11
-2
@@ -43,8 +43,8 @@ namespace FEXCore::Context {
|
||||
delete CTX;
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState* InitCore(FEXCore::Context::Context *CTX, FEXCore::CodeLoader *Loader) {
|
||||
return CTX->InitCore(Loader);
|
||||
FEXCore::Core::InternalThreadState* InitCore(FEXCore::Context::Context *CTX, uint64_t InitialRIP, uint64_t StackPointer) {
|
||||
return CTX->InitCore(InitialRIP, StackPointer);
|
||||
}
|
||||
|
||||
void SetExitHandler(FEXCore::Context::Context *CTX, ExitHandler handler) {
|
||||
@@ -110,6 +110,10 @@ namespace FEXCore::Context {
|
||||
void RegisterExternalSyscallVisitor(FEXCore::Context::Context *CTX, [[maybe_unused]] uint64_t Syscall, [[maybe_unused]] FEXCore::HLE::SyscallVisitor *Visitor) {
|
||||
}
|
||||
|
||||
HostFeatures GetHostFeatures(const FEXCore::Context::Context *CTX) {
|
||||
return CTX->HostFeatures;
|
||||
}
|
||||
|
||||
void HandleCallback(FEXCore::Context::Context *CTX, FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) {
|
||||
CTX->HandleCallback(Thread, RIP);
|
||||
}
|
||||
@@ -156,6 +160,7 @@ namespace FEXCore::Context {
|
||||
|
||||
void SetSyscallHandler(FEXCore::Context::Context *CTX, FEXCore::HLE::SyscallHandler *Handler) {
|
||||
CTX->SyscallHandler = Handler;
|
||||
CTX->SourcecodeResolver = Handler->GetSourcecodeResolver();
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults RunCPUIDFunction(FEXCore::Context::Context *CTX, uint32_t Function, uint32_t Leaf) {
|
||||
@@ -193,6 +198,10 @@ namespace FEXCore::Context {
|
||||
return CTX->UnloadAOTIRCacheEntry(Entry);
|
||||
}
|
||||
|
||||
CustomIRResult AddCustomIREntrypoint(FEXCore::Context::Context *CTX, uintptr_t Entrypoint, std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)> Handler, void *Creator, void *Data) {
|
||||
return CTX->AddCustomIREntrypoint(Entrypoint, Handler, Creator, Data);
|
||||
}
|
||||
|
||||
namespace Debug {
|
||||
void CompileRIP(FEXCore::Context::Context *CTX, uint64_t RIP) {
|
||||
CTX->CompileRIP(CTX->ParentThread, RIP);
|
||||
|
||||
+60
-24
@@ -1,14 +1,16 @@
|
||||
#pragma once
|
||||
|
||||
#include "Common/JitSymbols.h"
|
||||
#include "FEXHeaderUtils/ScopedSignalMask.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/HostFeatures.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/IR/AOTIR.h"
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
@@ -42,10 +44,14 @@ namespace CodeSerialize {
|
||||
namespace CPU {
|
||||
class Arm64JITCore;
|
||||
class X86JITCore;
|
||||
class InterpreterCore;
|
||||
class Dispatcher;
|
||||
}
|
||||
namespace HLE {
|
||||
struct SyscallArguments;
|
||||
class SyscallHandler;
|
||||
class SourcecodeResolver;
|
||||
struct SourcecodeMap;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -72,6 +78,7 @@ namespace FEXCore::Context {
|
||||
friend class FEXCore::CPU::X86JITCore;
|
||||
#endif
|
||||
|
||||
friend class FEXCore::CPU::InterpreterCore;
|
||||
friend class FEXCore::IR::Validation::IRValidation;
|
||||
|
||||
struct {
|
||||
@@ -86,6 +93,7 @@ namespace FEXCore::Context {
|
||||
FEX_CONFIG_OPT(GdbServer, GDBSERVER);
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
|
||||
FEX_CONFIG_OPT(TSOAutoMigration, TSOAUTOMIGRATION);
|
||||
FEX_CONFIG_OPT(ABILocalFlags, ABILOCALFLAGS);
|
||||
FEX_CONFIG_OPT(ABINoPF, ABINOPF);
|
||||
FEX_CONFIG_OPT(AOTIRCapture, AOTIRCAPTURE);
|
||||
@@ -102,21 +110,21 @@ namespace FEXCore::Context {
|
||||
FEX_CONFIG_OPT(GlobalJITNaming, GLOBALJITNAMING);
|
||||
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
|
||||
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
|
||||
FEX_CONFIG_OPT(GDBSymbols, GDBSYMBOLS);
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
|
||||
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
|
||||
FEX_CONFIG_OPT(x86dec_SynchronizeRIPOnAllBlocks, X86DEC_SYNCHRONIZERIPONALLBLOCKS);
|
||||
FEX_CONFIG_OPT(EnableAVX, ENABLEAVX);
|
||||
} Config;
|
||||
|
||||
using IntCallbackReturn = FEX_NAKED void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
|
||||
IntCallbackReturn InterpreterCallbackReturn;
|
||||
|
||||
FEXCore::HostFeatures HostFeatures;
|
||||
|
||||
std::mutex ThreadCreationMutex;
|
||||
uint64_t ThreadID{};
|
||||
FEXCore::Core::InternalThreadState* ParentThread;
|
||||
std::vector<FEXCore::Core::InternalThreadState*> Threads;
|
||||
std::atomic_bool CoreShuttingDown{false};
|
||||
bool NeedToCheckXID{true};
|
||||
|
||||
std::mutex IdleWaitMutex;
|
||||
std::condition_variable IdleWaitCV;
|
||||
@@ -125,9 +133,13 @@ namespace FEXCore::Context {
|
||||
Event PauseWait;
|
||||
bool Running{};
|
||||
|
||||
std::shared_mutex CodeInvalidationMutex;
|
||||
|
||||
FEXCore::CPUIDEmu CPUID;
|
||||
FEXCore::HLE::SyscallHandler *SyscallHandler{};
|
||||
FEXCore::HLE::SourcecodeResolver *SourcecodeResolver{};
|
||||
std::unique_ptr<FEXCore::ThunkHandler> ThunkHandler;
|
||||
std::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
|
||||
CustomCPUFactoryType CustomCPUFactory;
|
||||
FEXCore::Context::ExitHandler CustomExitHandler;
|
||||
@@ -142,7 +154,7 @@ namespace FEXCore::Context {
|
||||
Context();
|
||||
~Context();
|
||||
|
||||
FEXCore::Core::InternalThreadState* InitCore(FEXCore::CodeLoader *Loader);
|
||||
FEXCore::Core::InternalThreadState* InitCore(uint64_t InitialRIP, uint64_t StackPointer);
|
||||
FEXCore::Context::ExitReason RunUntilExit();
|
||||
int GetProgramStatus() const;
|
||||
bool IsPaused() const { return !Running; }
|
||||
@@ -162,15 +174,33 @@ namespace FEXCore::Context {
|
||||
void RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
|
||||
void RegisterFrontendHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required);
|
||||
|
||||
// Must be called from owning thread
|
||||
static void RemoveThreadCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
static void ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
static void ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker);
|
||||
|
||||
// Wrapper which takes CpuStateFrame instead of InternalThreadState
|
||||
// Must be called from owning thread
|
||||
static void RemoveThreadCodeEntryFromJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
RemoveThreadCodeEntry(Frame->Thread, GuestRIP);
|
||||
template<auto Fn>
|
||||
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
|
||||
FHU::ScopedSignalMaskWithSharedLock lk(Frame->Thread->CTX->CodeInvalidationMutex);
|
||||
|
||||
return Fn(Frame, record);
|
||||
}
|
||||
|
||||
// Wrapper which takes CpuStateFrame instead of InternalThreadState and unique_locks CodeInvalidationMutex
|
||||
// Must be called from owning thread
|
||||
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), gettid());
|
||||
|
||||
FHU::ScopedSignalMaskWithUniqueLock lk(Thread->CTX->CodeInvalidationMutex);
|
||||
|
||||
ThreadRemoveCodeEntry(Thread, GuestRIP);
|
||||
}
|
||||
|
||||
// returns false if a handler was already registered
|
||||
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)> Handler, void *Creator, void *Data);
|
||||
|
||||
void RemoveCustomIREntrypoint(uintptr_t Entrypoint);
|
||||
|
||||
// Debugger interface
|
||||
void CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP);
|
||||
uint64_t GetThreadCount() const;
|
||||
@@ -180,21 +210,19 @@ namespace FEXCore::Context {
|
||||
|
||||
struct GenerateIRResult {
|
||||
FEXCore::IR::IRListView* IRList;
|
||||
// User's responsibility to deallocate this.
|
||||
FEXCore::IR::RegisterAllocationData* RAData;
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
|
||||
uint64_t TotalInstructions;
|
||||
uint64_t TotalInstructionsLength;
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
};
|
||||
[[nodiscard]] GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP);
|
||||
[[nodiscard]] GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo);
|
||||
|
||||
struct CompileCodeResult {
|
||||
void* CompiledCode;
|
||||
FEXCore::IR::IRListView* IRData;
|
||||
FEXCore::Core::DebugData* DebugData;
|
||||
// User's responsibility to deallocate this.
|
||||
FEXCore::IR::RegisterAllocationData* RAData;
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
|
||||
bool GeneratedIR;
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
@@ -207,7 +235,7 @@ namespace FEXCore::Context {
|
||||
|
||||
// Used for thread creation from syscalls
|
||||
/**
|
||||
* @brief Used to create FEX thread objects in preparation for creating a true OS thread
|
||||
* @brief Used to create FEX thread objects in preparation for creating a true OS thread. Does set a TID or PID.
|
||||
*
|
||||
* @param NewThreadState The initial thread state to setup for our state
|
||||
* @param ParentTID The PID that was the parent thread that created this
|
||||
@@ -230,7 +258,7 @@ namespace FEXCore::Context {
|
||||
FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID);
|
||||
|
||||
/**
|
||||
* @brief Initializes the TLS data for a thread
|
||||
* @brief Initializes TID, PID and TLS data for a thread
|
||||
*
|
||||
* @param Thread The internal FEX thread state object
|
||||
*/
|
||||
@@ -297,8 +325,12 @@ namespace FEXCore::Context {
|
||||
FEXCore::Utils::PooledAllocatorMMap OpDispatcherAllocator;
|
||||
FEXCore::Utils::PooledAllocatorMMap FrontendAllocator;
|
||||
|
||||
void MarkMemoryShared();
|
||||
|
||||
bool IsTSOEnabled() { return (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled; }
|
||||
|
||||
protected:
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread, bool AlsoClearIRCache);
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
private:
|
||||
/**
|
||||
@@ -317,17 +349,15 @@ namespace FEXCore::Context {
|
||||
*
|
||||
* InitializeCompiler is called inside of CreateThread, so you likely don't need this
|
||||
*/
|
||||
void InitializeCompiler(FEXCore::Core::InternalThreadState* State);
|
||||
void InitializeCompiler(FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
void WaitForIdleWithTimeout();
|
||||
|
||||
void NotifyPause();
|
||||
|
||||
void AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr, uint64_t Start, uint64_t Length);
|
||||
FEXCore::CodeLoader *LocalLoader{};
|
||||
void AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr);
|
||||
|
||||
// Entry Cache
|
||||
uint64_t StartingRIP;
|
||||
std::mutex ExitMutex;
|
||||
std::unique_ptr<GdbServer> DebugServer;
|
||||
|
||||
@@ -335,7 +365,13 @@ namespace FEXCore::Context {
|
||||
std::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
|
||||
|
||||
bool StartPaused = false;
|
||||
bool IsMemoryShared = false;
|
||||
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
|
||||
|
||||
std::shared_mutex CustomIRMutex;
|
||||
std::unordered_map<uint64_t, std::tuple<std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)>, void *, void *>> CustomIRHandlers;
|
||||
FEXCore::CPU::CPUBackendFeatures BackendFeatures;
|
||||
FEXCore::CPU::DispatcherConfig DispatcherConfig;
|
||||
};
|
||||
|
||||
uint64_t HandleSyscall(FEXCore::HLE::SyscallHandler *Handler, FEXCore::Core::CpuStateFrame *Frame, FEXCore::HLE::SyscallArguments *Args);
|
||||
|
||||
@@ -1,14 +1,15 @@
|
||||
#include "Interface/Core/ArchHelpers/Arm64.h"
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
|
||||
#include <aarch64/cpu-aarch64.h>
|
||||
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <stdint.h>
|
||||
|
||||
#include <signal.h>
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
#include <csignal>
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::ArchHelpers::Arm64 {
|
||||
FEXCORE_TELEMETRY_STATIC_INIT(SplitLock, TYPE_HAS_SPLIT_LOCKS);
|
||||
@@ -1986,7 +1987,8 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
DesiredFunction = NEGDesired;
|
||||
break;
|
||||
default:
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}", AtomicOp);
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}",
|
||||
ToUnderlying(AtomicOp));
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -2039,7 +2041,8 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
DesiredFunction = NEGDesired;
|
||||
break;
|
||||
default:
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}", AtomicOp);
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}",
|
||||
ToUnderlying(AtomicOp));
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -2092,7 +2095,8 @@ uint64_t HandleAtomicLoadstoreExclusive(void *_ucontext, void *_info) {
|
||||
DesiredFunction = NEGDesired;
|
||||
break;
|
||||
default:
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}", AtomicOp);
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}",
|
||||
ToUnderlying(AtomicOp));
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
+48
-157
@@ -7,12 +7,14 @@
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
#include "cpu-features.h"
|
||||
#include "aarch64/instructions-aarch64.h"
|
||||
#include "utils-vixl.h"
|
||||
#include <aarch64/cpu-aarch64.h>
|
||||
#include <aarch64/instructions-aarch64.h>
|
||||
#include <cpu-features.h>
|
||||
#include <utils-vixl.h>
|
||||
|
||||
#include <array>
|
||||
#include <tuple>
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define STATE x28
|
||||
@@ -207,19 +209,29 @@ void Arm64Emitter::SpillStaticRegs(bool FPRs, uint32_t GPRSpillMask, uint32_t FP
|
||||
}
|
||||
|
||||
if (FPRs) {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i+=2) {
|
||||
auto Reg1 = SRAFPR[i];
|
||||
auto Reg2 = SRAFPR[i+1];
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
const auto Reg = SRAFPR[i];
|
||||
|
||||
if (((1U << Reg1.GetCode()) & FPRSpillMask) &&
|
||||
((1U << Reg2.GetCode()) & FPRSpillMask)) {
|
||||
stp(Reg1.Q(), Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
|
||||
if (((1U << Reg.GetCode()) & FPRSpillMask) != 0) {
|
||||
str(Reg.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.avx.data[i][0])));
|
||||
}
|
||||
}
|
||||
else if (((1U << Reg1.GetCode()) & FPRSpillMask)) {
|
||||
str(Reg1.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
|
||||
}
|
||||
else if (((1U << Reg2.GetCode()) & FPRSpillMask)) {
|
||||
str(Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i+1][0])));
|
||||
} else {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i += 2) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
|
||||
if (((1U << Reg1.GetCode()) & FPRSpillMask) &&
|
||||
((1U << Reg2.GetCode()) & FPRSpillMask)) {
|
||||
stp(Reg1.Q(), Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0])));
|
||||
}
|
||||
else if (((1U << Reg1.GetCode()) & FPRSpillMask)) {
|
||||
str(Reg1.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0])));
|
||||
}
|
||||
else if (((1U << Reg2.GetCode()) & FPRSpillMask)) {
|
||||
str(Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i+1][0])));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -244,19 +256,29 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
}
|
||||
|
||||
if (FPRs) {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i+=2) {
|
||||
auto Reg1 = SRAFPR[i];
|
||||
auto Reg2 = SRAFPR[i+1];
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
const auto Reg = SRAFPR[i];
|
||||
|
||||
if (((1U << Reg1.GetCode()) & FPRFillMask) &&
|
||||
((1U << Reg2.GetCode()) & FPRFillMask)) {
|
||||
ldp(Reg1.Q(), Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
|
||||
if (((1U << Reg.GetCode()) & FPRFillMask) != 0) {
|
||||
ldr(Reg.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.avx.data[i][0])));
|
||||
}
|
||||
}
|
||||
else if (((1U << Reg1.GetCode()) & FPRFillMask)) {
|
||||
ldr(Reg1.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i][0])));
|
||||
}
|
||||
else if (((1U << Reg2.GetCode()) & FPRFillMask)) {
|
||||
ldr(Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm[i+1][0])));
|
||||
} else {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i += 2) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
|
||||
if (((1U << Reg1.GetCode()) & FPRFillMask) &&
|
||||
((1U << Reg2.GetCode()) & FPRFillMask)) {
|
||||
ldp(Reg1.Q(), Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0])));
|
||||
}
|
||||
else if (((1U << Reg1.GetCode()) & FPRFillMask)) {
|
||||
ldr(Reg1.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i][0])));
|
||||
}
|
||||
else if (((1U << Reg2.GetCode()) & FPRFillMask)) {
|
||||
ldr(Reg2.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[i+1][0])));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -309,19 +331,6 @@ void Arm64Emitter::PopDynamicRegsAndLR() {
|
||||
add(sp, sp, SPOffset);
|
||||
}
|
||||
|
||||
void Arm64Emitter::ResetStack() {
|
||||
if (SpillSlots == 0)
|
||||
return;
|
||||
|
||||
if (IsImmAddSub(SpillSlots * 16)) {
|
||||
add(sp, sp, SpillSlots * 16);
|
||||
} else {
|
||||
// Too big to fit in a 12bit immediate
|
||||
LoadConstant(x0, SpillSlots * 16);
|
||||
add(sp, sp, x0);
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::Align16B() {
|
||||
uint64_t CurrentOffset = GetCursorAddress<uint64_t>();
|
||||
for (uint64_t i = (16 - (CurrentOffset & 0xF)); i != 0; i -= 4) {
|
||||
@@ -329,122 +338,4 @@ void Arm64Emitter::Align16B() {
|
||||
}
|
||||
}
|
||||
|
||||
uint64_t Arm64Emitter::GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
switch (Op) {
|
||||
case FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER:
|
||||
return Dispatcher->ExitFunctionLinkerAddress;
|
||||
break;
|
||||
default:
|
||||
ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op));
|
||||
break;
|
||||
}
|
||||
return ~0ULL;
|
||||
}
|
||||
|
||||
void Arm64Emitter::InsertNamedThunkRelocation(vixl::aarch64::Register Reg, const IR::SHA256Sum &Sum) {
|
||||
Relocation MoveABI{};
|
||||
MoveABI.NamedThunkMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE;
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint64_t>();
|
||||
MoveABI.NamedThunkMove.Offset = CurrentCursor - GuestEntry;
|
||||
MoveABI.NamedThunkMove.Symbol = Sum;
|
||||
MoveABI.NamedThunkMove.RegisterIndex = Reg.GetCode();
|
||||
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Sum));
|
||||
|
||||
LoadConstant(Reg, Pointer, EmitterCTX->Config.CacheObjectCodeCompilation());
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
Arm64Emitter::NamedSymbolLiteralPair Arm64Emitter::InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Op);
|
||||
|
||||
Arm64Emitter::NamedSymbolLiteralPair Lit {
|
||||
.Lit = Literal(Pointer),
|
||||
.MoveABI = {
|
||||
.NamedSymbolLiteral = {
|
||||
.Header = {
|
||||
.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL,
|
||||
},
|
||||
.Symbol = Op,
|
||||
.Offset = 0,
|
||||
},
|
||||
},
|
||||
};
|
||||
return Lit;
|
||||
}
|
||||
|
||||
void Arm64Emitter::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit) {
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint64_t>();
|
||||
Lit.MoveABI.NamedSymbolLiteral.Offset = CurrentCursor - GuestEntry;
|
||||
|
||||
place(&Lit.Lit);
|
||||
Relocations.emplace_back(Lit.MoveABI);
|
||||
}
|
||||
|
||||
void Arm64Emitter::InsertGuestRIPMove(vixl::aarch64::Register Reg, uint64_t Constant) {
|
||||
Relocation MoveABI{};
|
||||
MoveABI.GuestRIPMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE;
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint64_t>();
|
||||
MoveABI.GuestRIPMove.Offset = CurrentCursor - GuestEntry;
|
||||
MoveABI.GuestRIPMove.GuestRIP = Constant;
|
||||
MoveABI.GuestRIPMove.RegisterIndex = Reg.GetCode();
|
||||
|
||||
LoadConstant(Reg, Constant, EmitterCTX->Config.CacheObjectCodeCompilation());
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
bool Arm64Emitter::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations) {
|
||||
size_t DataIndex{};
|
||||
for (size_t j = 0; j < NumRelocations; ++j) {
|
||||
const FEXCore::CPU::Relocation *Reloc = reinterpret_cast<const FEXCore::CPU::Relocation *>(&EntryRelocations[DataIndex]);
|
||||
LOGMAN_THROW_A_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
|
||||
|
||||
switch (Reloc->Header.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Reloc->NamedSymbolLiteral.Symbol);
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor
|
||||
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->NamedSymbolLiteral.Offset);
|
||||
|
||||
// Generate a literal so we can place it
|
||||
Literal<uint64_t> Lit(Pointer);
|
||||
place(&Lit);
|
||||
|
||||
DataIndex += sizeof(Reloc->NamedSymbolLiteral);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc->NamedThunkMove.Symbol));
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->NamedThunkMove.Offset);
|
||||
LoadConstant(vixl::aarch64::XRegister(Reloc->NamedThunkMove.RegisterIndex), Pointer, true);
|
||||
DataIndex += sizeof(Reloc->NamedThunkMove);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
|
||||
// XXX: Reenable once the JIT Object Cache is upstream
|
||||
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
|
||||
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->GuestRIPMove.Offset);
|
||||
LoadConstant(vixl::aarch64::XRegister(Reloc->GuestRIPMove.RegisterIndex), Pointer, true);
|
||||
DataIndex += sizeof(Reloc->GuestRIPMove);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
}
|
||||
@@ -3,16 +3,17 @@
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/ObjectCache/Relocations.h"
|
||||
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
#include "aarch64/constants-aarch64.h"
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
#include "aarch64/operands-aarch64.h"
|
||||
#include "platform-vixl.h"
|
||||
#include "FEXCore/Config/Config.h"
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
#include <aarch64/constants-aarch64.h>
|
||||
#include <aarch64/cpu-aarch64.h>
|
||||
#include <aarch64/operands-aarch64.h>
|
||||
#include <platform-vixl.h>
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
|
||||
#include <array>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
@@ -63,8 +64,6 @@ class Arm64Emitter : public vixl::aarch64::Assembler {
|
||||
protected:
|
||||
Arm64Emitter(FEXCore::Context::Context *ctx, size_t size);
|
||||
|
||||
std::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
|
||||
FEXCore::Context::Context *EmitterCTX;
|
||||
vixl::aarch64::CPU CPU;
|
||||
void LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant, bool NOPPad = false);
|
||||
@@ -83,68 +82,7 @@ protected:
|
||||
void PushCalleeSavedRegisters();
|
||||
void PopCalleeSavedRegisters();
|
||||
|
||||
void ResetStack();
|
||||
void Align16B();
|
||||
/**
|
||||
* @name Relocations
|
||||
* @{ */
|
||||
|
||||
uint64_t GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
|
||||
|
||||
/**
|
||||
* @brief A literal pair relocation object for named symbol literals
|
||||
*/
|
||||
struct NamedSymbolLiteralPair {
|
||||
Literal<uint64_t> Lit;
|
||||
Relocation MoveABI{};
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Inserts a thunk relocation
|
||||
*
|
||||
* @param Reg - The GPR to move the thunk handler in to
|
||||
* @param Sum - The hash of the thunk
|
||||
*/
|
||||
void InsertNamedThunkRelocation(vixl::aarch64::Register Reg, const IR::SHA256Sum &Sum);
|
||||
|
||||
/**
|
||||
* @brief Inserts a guest GPR move relocation
|
||||
*
|
||||
* @param Reg - The GPR to move the guest RIP in to
|
||||
* @param Constant - The guest RIP that will be relocated
|
||||
*/
|
||||
void InsertGuestRIPMove(vixl::aarch64::Register Reg, uint64_t Constant);
|
||||
|
||||
/**
|
||||
* @brief Inserts a named symbol as a literal in memory
|
||||
*
|
||||
* Need to use `PlaceNamedSymbolLiteral` with the return value to place the literal in the desired location
|
||||
*
|
||||
* @param Op The named symbol to place
|
||||
*
|
||||
* @return A temporary `NamedSymbolLiteralPair`
|
||||
*/
|
||||
NamedSymbolLiteralPair InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
|
||||
|
||||
/**
|
||||
* @brief Place the named symbol literal relocation in memory
|
||||
*
|
||||
* @param Lit - Which literal to place
|
||||
*/
|
||||
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit);
|
||||
|
||||
std::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
|
||||
///< Relocation code loading
|
||||
bool ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations);
|
||||
|
||||
/** @} */
|
||||
|
||||
uint32_t SpillSlots{};
|
||||
/**
|
||||
* @brief Current guest RIP entrypoint
|
||||
*/
|
||||
uint64_t GuestEntry{};
|
||||
|
||||
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
|
||||
};
|
||||
|
||||
@@ -124,7 +124,7 @@ static inline void SetArmReg(void* ucontext, uint32_t id, uint64_t val) {
|
||||
static inline __uint128_t GetArmFPR(void* ucontext, uint32_t id) {
|
||||
auto MContext = GetMContext(ucontext);
|
||||
HostFPRState *HostState = reinterpret_cast<HostFPRState*>(&MContext->__reserved[0]);
|
||||
LOGMAN_THROW_A_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
LOGMAN_THROW_AA_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
|
||||
return HostState->FPRs[id];
|
||||
}
|
||||
@@ -143,7 +143,7 @@ static inline void BackupContext(void* ucontext, T *Backup) {
|
||||
|
||||
// Host FPR state starts at _mcontext->reserved[0];
|
||||
HostFPRState *HostState = reinterpret_cast<HostFPRState*>(&_mcontext->__reserved[0]);
|
||||
LOGMAN_THROW_A_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
LOGMAN_THROW_AA_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
Backup->FPSR = HostState->FPSR;
|
||||
Backup->FPCR = HostState->FPCR;
|
||||
memcpy(&Backup->FPRs[0], &HostState->FPRs[0], 32 * sizeof(__uint128_t));
|
||||
@@ -163,7 +163,7 @@ static inline void RestoreContext(void* ucontext, T *Backup) {
|
||||
auto _mcontext = GetMContext(ucontext);
|
||||
|
||||
HostFPRState *HostState = reinterpret_cast<HostFPRState*>(&_mcontext->__reserved[0]);
|
||||
LOGMAN_THROW_A_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
LOGMAN_THROW_AA_FMT(HostState->Head.Magic == FPR_MAGIC, "Wrong FPR Magic: 0x{:08x}", HostState->Head.Magic);
|
||||
memcpy(&HostState->FPRs[0], &Backup->FPRs[0], 32 * sizeof(__uint128_t));
|
||||
HostState->FPCR = Backup->FPCR;
|
||||
HostState->FPSR = Backup->FPSR;
|
||||
|
||||
@@ -0,0 +1,87 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
|
||||
namespace FEXCore {
|
||||
namespace CPU {
|
||||
|
||||
CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState *ThreadState, size_t InitialCodeSize, size_t MaxCodeSize)
|
||||
: ThreadState(ThreadState), InitialCodeSize(InitialCodeSize), MaxCodeSize(MaxCodeSize) {}
|
||||
|
||||
CPUBackend::~CPUBackend() {
|
||||
for (auto CodeBuffer : CodeBuffers) {
|
||||
FreeCodeBuffer(CodeBuffer);
|
||||
}
|
||||
CodeBuffers.clear();
|
||||
}
|
||||
|
||||
auto CPUBackend::GetEmptyCodeBuffer() -> CodeBuffer * {
|
||||
if (ThreadState->CurrentFrame->SignalHandlerRefCounter == 0) {
|
||||
if (CodeBuffers.empty()) {
|
||||
auto NewCodeBuffer = AllocateNewCodeBuffer(InitialCodeSize);
|
||||
EmplaceNewCodeBuffer(NewCodeBuffer);
|
||||
} else {
|
||||
if (CodeBuffers.size() > 1) {
|
||||
// If we have more than one code buffer we are tracking then walk them and delete
|
||||
// This is a cleanup step
|
||||
for (size_t i = 1; i < CodeBuffers.size(); i++) {
|
||||
FreeCodeBuffer(CodeBuffers[i]);
|
||||
}
|
||||
CodeBuffers.resize(1);
|
||||
}
|
||||
// Set the current code buffer to the initial
|
||||
CurrentCodeBuffer = &CodeBuffers[0];
|
||||
|
||||
if (CurrentCodeBuffer->Size != MaxCodeSize) {
|
||||
FreeCodeBuffer(*CurrentCodeBuffer);
|
||||
|
||||
// Resize the code buffer and reallocate our code size
|
||||
CurrentCodeBuffer->Size *= 1.5;
|
||||
CurrentCodeBuffer->Size = std::min(CurrentCodeBuffer->Size, MaxCodeSize);
|
||||
|
||||
*CurrentCodeBuffer = AllocateNewCodeBuffer(CurrentCodeBuffer->Size);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// We have signal handlers that have generated code
|
||||
// This means that we can not safely clear the code at this point in time
|
||||
// Allocate some new code buffers that we can switch over to instead
|
||||
auto NewCodeBuffer = AllocateNewCodeBuffer(InitialCodeSize);
|
||||
EmplaceNewCodeBuffer(NewCodeBuffer);
|
||||
}
|
||||
|
||||
return CurrentCodeBuffer;
|
||||
}
|
||||
|
||||
auto CPUBackend::AllocateNewCodeBuffer(size_t Size) -> CodeBuffer {
|
||||
CodeBuffer Buffer;
|
||||
Buffer.Size = Size;
|
||||
Buffer.Ptr = static_cast<uint8_t *>(
|
||||
FEXCore::Allocator::mmap(nullptr, Buffer.Size, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
|
||||
LOGMAN_THROW_AA_FMT(!!Buffer.Ptr, "Couldn't allocate code buffer");
|
||||
|
||||
if (ThreadState->CTX->Config.GlobalJITNaming()) {
|
||||
ThreadState->CTX->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
return Buffer;
|
||||
}
|
||||
|
||||
void CPUBackend::FreeCodeBuffer(CodeBuffer Buffer) {
|
||||
FEXCore::Allocator::munmap(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
|
||||
bool CPUBackend::IsAddressInCodeBuffer(uintptr_t Address) const {
|
||||
for (auto &Buffer: CodeBuffers) {
|
||||
auto start = (uintptr_t)Buffer.Ptr;
|
||||
auto end = start + Buffer.Size;
|
||||
|
||||
if (Address >= start && Address < end) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
}
|
||||
}
|
||||
+9
-4
@@ -8,11 +8,11 @@ $end_info$
|
||||
#include "Common/StringConv.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/HostFeatures.h"
|
||||
#include "Utils/FileLoading.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include "git_version.h"
|
||||
@@ -39,6 +39,7 @@ namespace ProductNames {
|
||||
static const char ARM_A78C[] = "Cortex-A78C";
|
||||
static const char ARM_A710[] = "Cortex-A710";
|
||||
static const char ARM_X1[] = "Cortex-X1";
|
||||
static const char ARM_X1C[] = "Cortex-X1C";
|
||||
static const char ARM_X2[] = "Cortex-X2";
|
||||
static const char ARM_N1[] = "Neoverse N1";
|
||||
static const char ARM_N2[] = "Neoverse N2";
|
||||
@@ -83,7 +84,10 @@ static uint32_t CalculateNumberOfCPUs() {
|
||||
return CPUs;
|
||||
}
|
||||
|
||||
// TODO: Replace usages with CTX->HostFeatures.EnableAVX
|
||||
// when AVX implementations are further along.
|
||||
constexpr uint32_t SUPPORTS_AVX = 0;
|
||||
|
||||
// #define CPUID_AMD
|
||||
#ifdef CPUID_AMD
|
||||
constexpr uint32_t FAMILY_IDENTIFIER =
|
||||
@@ -151,7 +155,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
// CPU priority order
|
||||
// This is mostly arbitrary but will sort by some sort of CPU priority by performance
|
||||
// Relative list so things they will commonly end up in big.little configurations sort of relate
|
||||
static constexpr std::array<CPUMIDR, 35> CPUMIDRs = {{
|
||||
static constexpr std::array<CPUMIDR, 36> CPUMIDRs = {{
|
||||
// Typically big CPU cores
|
||||
{0x61, 0x023, 1, ProductNames::ARM_Firestorm}, // Apple M1 Firestorm
|
||||
|
||||
@@ -161,6 +165,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
|
||||
{0x41, 0xd48, 1, ProductNames::ARM_X2}, // X2
|
||||
{0x41, 0xd47, 1, ProductNames::ARM_A710}, // A710
|
||||
{0x41, 0xd4C, 1, ProductNames::ARM_X1C}, // X1C
|
||||
{0x41, 0xd44, 1, ProductNames::ARM_X1}, // X1
|
||||
{0x41, 0xd42, 1, ProductNames::ARM_A78AE}, // A78AE
|
||||
{0x41, 0xd41, 1, ProductNames::ARM_A78}, // A78
|
||||
@@ -416,7 +421,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
|
||||
Res.ecx =
|
||||
(1 << 0) | // SSE3
|
||||
(0 << 1) | // PCLMULQDQ
|
||||
(1 << 1) | // PCLMULQDQ
|
||||
(1 << 2) | // DS area supports 64bit layout
|
||||
(1 << 3) | // MWait
|
||||
(0 << 4) | // DS-CPL
|
||||
@@ -446,7 +451,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
(SUPPORTS_AVX << 28) | // AVX
|
||||
(0 << 29) | // F16C
|
||||
(CTX->HostFeatures.SupportsRAND << 30) | // RDRAND
|
||||
(0 << 31); // Hypervisor always returns zero
|
||||
(1 << 31); // Hypervisor always returns one
|
||||
|
||||
Res.edx =
|
||||
(1 << 0) | // FPU
|
||||
|
||||
+353
-232
@@ -7,6 +7,7 @@ desc: Glues Frontend, OpDispatcher and IR Opts & Compilation, LookupCache, Dispa
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include <cstdint>
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/Core.h"
|
||||
@@ -17,6 +18,7 @@ $end_info$
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterCore.h"
|
||||
#include "Interface/Core/JIT/JITCore.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
#include "Interface/IR/Passes.h"
|
||||
@@ -32,6 +34,7 @@ $end_info$
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Debug/X86Tables.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/HLE/SourcecodeResolver.h>
|
||||
#include <FEXCore/HLE/Linux/ThreadManagement.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
@@ -49,7 +52,6 @@ $end_info$
|
||||
#include <atomic>
|
||||
#include <chrono>
|
||||
#include <condition_variable>
|
||||
#include <cstdint>
|
||||
#include <filesystem>
|
||||
#include <functional>
|
||||
#include <fstream>
|
||||
@@ -74,6 +76,7 @@ $end_info$
|
||||
#include <vector>
|
||||
#include <xxhash.h>
|
||||
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
bool CreateCPUCore(FEXCore::Context::Context *CTX) {
|
||||
// This should be used for generating things that are shared between threads
|
||||
@@ -151,6 +154,16 @@ namespace FEXCore::Context {
|
||||
if (Config.CacheObjectCodeCompilation() != FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE) {
|
||||
CodeObjectCacheService = std::make_unique<FEXCore::CodeSerialize::CodeObjectSerializeService>(this);
|
||||
}
|
||||
if (!Config.EnableAVX) {
|
||||
HostFeatures.SupportsAVX = false;
|
||||
}
|
||||
|
||||
if (Config.BlockJITNaming() ||
|
||||
Config.GlobalJITNaming() ||
|
||||
Config.LibraryJITNaming()) {
|
||||
// Only initialize symbols file if enabled. Ensures we don't pollute /tmp with empty files.
|
||||
Symbols.InitFile();
|
||||
}
|
||||
}
|
||||
|
||||
Context::~Context() {
|
||||
@@ -177,15 +190,17 @@ namespace FEXCore::Context {
|
||||
|
||||
// Initialize default CPU state
|
||||
NewThreadState.rip = ~0ULL;
|
||||
for (int i = 0; i < 16; ++i) {
|
||||
NewThreadState.gregs[i] = 0;
|
||||
for (auto& greg : NewThreadState.gregs) {
|
||||
greg = 0;
|
||||
}
|
||||
|
||||
for (int i = 0; i < 16; ++i) {
|
||||
NewThreadState.xmm[i][0] = 0xDEADBEEFULL;
|
||||
NewThreadState.xmm[i][1] = 0xBAD0DAD1ULL;
|
||||
for (auto& xmm : NewThreadState.xmm.avx.data) {
|
||||
xmm[0] = 0xDEADBEEFULL;
|
||||
xmm[1] = 0xBAD0DAD1ULL;
|
||||
xmm[2] = 0xDEADCAFEULL;
|
||||
xmm[3] = 0xBAD2CAD3ULL;
|
||||
}
|
||||
memset(NewThreadState.flags, 0, 32);
|
||||
memset(NewThreadState.flags, 0, Core::CPUState::NUM_EFLAG_BITS);
|
||||
NewThreadState.flags[1] = 1;
|
||||
NewThreadState.flags[9] = 1;
|
||||
NewThreadState.FCW = 0x37F;
|
||||
@@ -193,19 +208,22 @@ namespace FEXCore::Context {
|
||||
return NewThreadState;
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState* Context::InitCore(FEXCore::CodeLoader *Loader) {
|
||||
// Initialize the CPU core signal handlers
|
||||
FEXCore::Core::InternalThreadState* Context::InitCore(uint64_t InitialRIP, uint64_t StackPointer) {
|
||||
// Initialize the CPU core signal handlers & DispatcherConfig
|
||||
switch (Config.Core) {
|
||||
#ifdef INTERPRETER_ENABLED
|
||||
case FEXCore::Config::CONFIG_INTERPRETER:
|
||||
FEXCore::CPU::InitializeInterpreterSignalHandlers(this);
|
||||
BackendFeatures = FEXCore::CPU::GetInterpreterBackendFeatures();
|
||||
break;
|
||||
#endif
|
||||
case FEXCore::Config::CONFIG_IRJIT:
|
||||
#if (_M_X86_64 && JIT_X86_64)
|
||||
FEXCore::CPU::InitializeX86JITSignalHandlers(this);
|
||||
BackendFeatures = FEXCore::CPU::GetX86JITBackendFeatures();
|
||||
#elif (_M_ARM_64 && JIT_ARM64)
|
||||
FEXCore::CPU::InitializeArm64JITSignalHandlers(this);
|
||||
BackendFeatures = FEXCore::CPU::GetArm64JITBackendFeatures();
|
||||
#else
|
||||
ERROR_AND_DIE_FMT("FEXCore has been compiled without a viable JIT core");
|
||||
#endif
|
||||
@@ -218,6 +236,32 @@ namespace FEXCore::Context {
|
||||
break;
|
||||
}
|
||||
|
||||
DispatcherConfig.StaticRegisterAllocation = Config.StaticRegisterAllocation && BackendFeatures.SupportsStaticRegisterAllocation;
|
||||
|
||||
#if (_M_X86_64)
|
||||
Dispatcher = FEXCore::CPU::Dispatcher::CreateX86(this, DispatcherConfig);
|
||||
#elif (_M_ARM_64)
|
||||
Dispatcher = FEXCore::CPU::Dispatcher::CreateArm64(this, DispatcherConfig);
|
||||
#else
|
||||
ERROR_AND_DIE_FMT("FEXCore has been compiled with an unknown target");
|
||||
#endif
|
||||
|
||||
// Initialize common signal handlers
|
||||
|
||||
auto PauseHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
return Thread->CTX->Dispatcher->HandleSignalPause(Thread, Signal, info, ucontext);
|
||||
};
|
||||
|
||||
SignalDelegation->RegisterHostSignalHandler(SignalDelegator::SIGNAL_FOR_PAUSE, PauseHandler, true);
|
||||
|
||||
auto GuestSignalHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) -> bool {
|
||||
return Thread->CTX->Dispatcher->HandleGuestSignal(Thread, Signal, info, ucontext, GuestAction, GuestStack);
|
||||
};
|
||||
|
||||
for (uint32_t Signal = 0; Signal <= SignalDelegator::MAX_SIGNALS; ++Signal) {
|
||||
SignalDelegation->RegisterHostSignalHandlerForGuest(Signal, GuestSignalHandler);
|
||||
}
|
||||
|
||||
// Initialize GDBServer after the signal handlers are installed
|
||||
// It may install its own handlers that need to be executed AFTER the CPU cores
|
||||
if (Config.GdbServer) {
|
||||
@@ -229,7 +273,6 @@ namespace FEXCore::Context {
|
||||
|
||||
ThunkHandler.reset(FEXCore::ThunkHandler::Create());
|
||||
|
||||
LocalLoader = Loader;
|
||||
using namespace FEXCore::Core;
|
||||
|
||||
FEXCore::Core::CPUState NewThreadState = CreateDefaultCPUState();
|
||||
@@ -238,9 +281,9 @@ namespace FEXCore::Context {
|
||||
// We are the parent thread
|
||||
ParentThread = Thread;
|
||||
|
||||
Thread->CurrentFrame->State.gregs[X86State::REG_RSP] = Loader->GetStackPointer();
|
||||
Thread->CurrentFrame->State.gregs[X86State::REG_RSP] = StackPointer;
|
||||
|
||||
Thread->CurrentFrame->State.rip = StartingRIP = Loader->DefaultRIP();
|
||||
Thread->CurrentFrame->State.rip = InitialRIP;
|
||||
|
||||
InitializeThreadData(Thread);
|
||||
return Thread;
|
||||
@@ -258,7 +301,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
void Context::HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) {
|
||||
Thread->CPUBackend->CallbackPtr(Thread->CurrentFrame, RIP);
|
||||
Thread->CTX->Dispatcher->ExecuteJITCallback(Thread->CurrentFrame, RIP);
|
||||
}
|
||||
|
||||
void Context::RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
@@ -352,10 +395,7 @@ namespace FEXCore::Context {
|
||||
// Walk the threads and tell them to clear their caches
|
||||
// Useful when our block size is set to a large number and we need to step a single instruction
|
||||
for (auto &Thread : Threads) {
|
||||
// Wait for thread to be fully constructed
|
||||
// XXX: Look into thread partial construction issues
|
||||
while(Thread->RunningEvents.WaitingToStart.load()) ;
|
||||
ClearCodeCache(Thread, true);
|
||||
ClearCodeCache(Thread);
|
||||
}
|
||||
}
|
||||
CoreRunningMode PreviousRunningMode = this->Config.RunningMode;
|
||||
@@ -448,22 +488,6 @@ namespace FEXCore::Context {
|
||||
|
||||
void Context::InitializeThreadData(FEXCore::Core::InternalThreadState *Thread) {
|
||||
Thread->CPUBackend->Initialize();
|
||||
|
||||
auto IRHandler = [Thread](uint64_t Addr, IR::IREmitter *IR) -> void {
|
||||
// Run the passmanager over the IR from the dispatcher
|
||||
Thread->PassManager->Run(IR);
|
||||
Core::LocalIREntry Entry = {Addr, 0ULL,
|
||||
decltype(Entry.IR)(IR->CreateIRCopy()),
|
||||
decltype(Entry.RAData)(Thread->PassManager->HasPass("RA")
|
||||
? Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA")->PullAllocationData()
|
||||
: nullptr),
|
||||
decltype(Entry.DebugData)(new Core::DebugData())
|
||||
};
|
||||
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
|
||||
Thread->LocalIRCache.insert({Addr, std::move(Entry)});
|
||||
};
|
||||
|
||||
LocalLoader->AddIR(IRHandler);
|
||||
}
|
||||
|
||||
struct ExecutionThreadHandler {
|
||||
@@ -483,10 +507,21 @@ namespace FEXCore::Context {
|
||||
ExecutionThreadHandler *Arg = reinterpret_cast<ExecutionThreadHandler*>(FEXCore::Allocator::malloc(sizeof(ExecutionThreadHandler)));
|
||||
Arg->This = this;
|
||||
Arg->Thread = Thread;
|
||||
Thread->StartPaused = NeedToCheckXID;
|
||||
Thread->ExecutionThread = FEXCore::Threads::Thread::Create(ThreadHandler, Arg);
|
||||
|
||||
// Wait for the thread to have started
|
||||
Thread->ThreadWaiting.Wait();
|
||||
|
||||
if (NeedToCheckXID) {
|
||||
// The first time an application creates a thread, GLIBC installs their SETXID signal handler.
|
||||
// FEX needs to capture all signals and defer them to the guest.
|
||||
// Once FEX creates its first guest thread, overwrite the GLIBC SETXID handler *again* to ensure
|
||||
// FEX maintains control of the signal handler on this signal.
|
||||
NeedToCheckXID = false;
|
||||
SignalDelegation->CheckXIDHandler();
|
||||
Thread->StartRunning.NotifyAll();
|
||||
}
|
||||
}
|
||||
|
||||
void Context::InitializeThreadTLSData(FEXCore::Core::InternalThreadState *Thread) {
|
||||
@@ -502,50 +537,51 @@ namespace FEXCore::Context {
|
||||
Thread->StartRunning.NotifyAll();
|
||||
}
|
||||
|
||||
void Context::InitializeCompiler(FEXCore::Core::InternalThreadState* State) {
|
||||
State->OpDispatcher = std::make_unique<FEXCore::IR::OpDispatchBuilder>(this);
|
||||
State->OpDispatcher->SetMultiblock(Config.Multiblock);
|
||||
State->LookupCache = std::make_unique<FEXCore::LookupCache>(this);
|
||||
State->FrontendDecoder = std::make_unique<FEXCore::Frontend::Decoder>(this);
|
||||
State->PassManager = std::make_unique<FEXCore::IR::PassManager>();
|
||||
State->PassManager->RegisterExitHandler([this]() {
|
||||
void Context::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread) {
|
||||
Thread->OpDispatcher = std::make_unique<FEXCore::IR::OpDispatchBuilder>(this);
|
||||
Thread->OpDispatcher->SetMultiblock(Config.Multiblock);
|
||||
Thread->LookupCache = std::make_unique<FEXCore::LookupCache>(this);
|
||||
Thread->FrontendDecoder = std::make_unique<FEXCore::Frontend::Decoder>(this);
|
||||
Thread->PassManager = std::make_unique<FEXCore::IR::PassManager>();
|
||||
Thread->PassManager->RegisterExitHandler([this]() {
|
||||
Stop(false /* Ignore current thread */);
|
||||
});
|
||||
|
||||
State->CTX = this;
|
||||
Thread->CurrentFrame->Pointers.Common.L1Pointer = Thread->LookupCache->GetL1Pointer();
|
||||
Thread->CurrentFrame->Pointers.Common.L2Pointer = Thread->LookupCache->GetPagePointer();
|
||||
|
||||
#if _M_ARM_64
|
||||
bool DoSRA = State->CTX->Config.StaticRegisterAllocation;
|
||||
#else
|
||||
bool DoSRA = false;
|
||||
#endif
|
||||
Dispatcher->InitThreadPointers(Thread);
|
||||
|
||||
State->PassManager->AddDefaultPasses(this, Config.Core == FEXCore::Config::CONFIG_IRJIT, DoSRA);
|
||||
State->PassManager->AddDefaultValidationPasses();
|
||||
Thread->CTX = this;
|
||||
|
||||
State->PassManager->RegisterSyscallHandler(SyscallHandler);
|
||||
bool DoSRA = DispatcherConfig.StaticRegisterAllocation;
|
||||
|
||||
Thread->PassManager->AddDefaultPasses(this, Config.Core == FEXCore::Config::CONFIG_IRJIT, DoSRA);
|
||||
Thread->PassManager->AddDefaultValidationPasses();
|
||||
|
||||
Thread->PassManager->RegisterSyscallHandler(SyscallHandler);
|
||||
|
||||
// Create CPU backend
|
||||
switch (Config.Core) {
|
||||
#ifdef INTERPRETER_ENABLED
|
||||
case FEXCore::Config::CONFIG_INTERPRETER:
|
||||
State->CPUBackend = FEXCore::CPU::CreateInterpreterCore(this, State);
|
||||
Thread->CPUBackend = FEXCore::CPU::CreateInterpreterCore(this, Thread);
|
||||
break;
|
||||
#endif
|
||||
case FEXCore::Config::CONFIG_IRJIT:
|
||||
State->PassManager->InsertRegisterAllocationPass(DoSRA);
|
||||
Thread->PassManager->InsertRegisterAllocationPass(DoSRA, HostFeatures.SupportsAVX);
|
||||
|
||||
#if (_M_X86_64 && JIT_X86_64)
|
||||
State->CPUBackend = FEXCore::CPU::CreateX86JITCore(this, State);
|
||||
Thread->CPUBackend = FEXCore::CPU::CreateX86JITCore(this, Thread);
|
||||
#elif (_M_ARM_64 && JIT_ARM64)
|
||||
State->CPUBackend = FEXCore::CPU::CreateArm64JITCore(this, State);
|
||||
Thread->CPUBackend = FEXCore::CPU::CreateArm64JITCore(this, Thread);
|
||||
#else
|
||||
ERROR_AND_DIE_FMT("FEXCore has been compiled without a viable JIT core");
|
||||
#endif
|
||||
|
||||
break;
|
||||
case FEXCore::Config::CONFIG_CUSTOM:
|
||||
State->CPUBackend = CustomCPUFactory(this, State);
|
||||
Thread->CPUBackend = CustomCPUFactory(this, Thread);
|
||||
break;
|
||||
default:
|
||||
ERROR_AND_DIE_FMT("Unknown core configuration");
|
||||
@@ -554,14 +590,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState* Context::CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) {
|
||||
FEXCore::Core::InternalThreadState *Thread{};
|
||||
|
||||
// Grab the new thread object
|
||||
{
|
||||
std::lock_guard<std::mutex> lk(ThreadCreationMutex);
|
||||
Thread = Threads.emplace_back(new FEXCore::Core::InternalThreadState{});
|
||||
Thread->ThreadManager.TID = ++ThreadID;
|
||||
}
|
||||
FEXCore::Core::InternalThreadState *Thread = new FEXCore::Core::InternalThreadState{};
|
||||
|
||||
// Copy over the new thread state to the new object
|
||||
memcpy(Thread->CurrentFrame, NewThreadState, sizeof(FEXCore::Core::CPUState));
|
||||
@@ -573,13 +602,19 @@ namespace FEXCore::Context {
|
||||
InitializeCompiler(Thread);
|
||||
InitializeThreadData(Thread);
|
||||
|
||||
// Insert after the Thread object has been fully initialized
|
||||
{
|
||||
std::lock_guard lk(ThreadCreationMutex);
|
||||
Threads.push_back(Thread);
|
||||
}
|
||||
|
||||
return Thread;
|
||||
}
|
||||
|
||||
void Context::DestroyThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// remove new thread object
|
||||
{
|
||||
std::lock_guard<std::mutex> lk(ThreadCreationMutex);
|
||||
std::lock_guard lk(ThreadCreationMutex);
|
||||
|
||||
auto It = std::find(Threads.begin(), Threads.end(), Thread);
|
||||
LOGMAN_THROW_A_FMT(It != Threads.end(), "Thread wasn't in Threads");
|
||||
@@ -634,14 +669,11 @@ namespace FEXCore::Context {
|
||||
FEXCore::Threads::Thread::CleanupAfterFork();
|
||||
}
|
||||
|
||||
void Context::AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr, uint64_t Start, uint64_t Length) {
|
||||
// Only call MarkGuestExecutableRange if new pages are marked as containing code
|
||||
if (Thread->LookupCache->AddBlockMapping(Address, Ptr, Start, Length)) {
|
||||
Thread->CTX->SyscallHandler->MarkGuestExecutableRange(Start, Length);
|
||||
}
|
||||
void Context::AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr) {
|
||||
Thread->LookupCache->AddBlockMapping(Address, Ptr);
|
||||
}
|
||||
|
||||
void Context::ClearCodeCache(FEXCore::Core::InternalThreadState *Thread, bool AlsoClearIRCache) {
|
||||
void Context::ClearCodeCache(FEXCore::Core::InternalThreadState *Thread) {
|
||||
{
|
||||
// Ensure the Code Object Serialization service has fully serialized this thread's data before clearing the cache
|
||||
// Use the thread's object cache ref counter for this
|
||||
@@ -651,18 +683,16 @@ namespace FEXCore::Context {
|
||||
|
||||
Thread->LookupCache->ClearCache();
|
||||
Thread->CPUBackend->ClearCache();
|
||||
|
||||
if (AlsoClearIRCache) {
|
||||
Thread->LocalIRCache.clear();
|
||||
}
|
||||
Thread->DebugStore.clear();
|
||||
}
|
||||
|
||||
static void IRDumper(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, IR::RegisterAllocationData* RA) {
|
||||
static void IRDumper(FEXCore::Core::InternalThreadState *Thread, IR::IREmitter *IREmitter, uint64_t GuestRIP, IR::RegisterAllocationData* RA) {
|
||||
FILE* f = nullptr;
|
||||
bool CloseAfter = false;
|
||||
const auto DumpIRStr = Thread->CTX->Config.DumpIR();
|
||||
|
||||
if (DumpIRStr =="stderr") {
|
||||
// DumpIRStr might be no if not dumping but ShouldDump is set in OpDisp
|
||||
if (DumpIRStr =="stderr" || DumpIRStr =="no") {
|
||||
f = stderr;
|
||||
}
|
||||
else if (DumpIRStr =="stdout") {
|
||||
@@ -676,7 +706,7 @@ namespace FEXCore::Context {
|
||||
|
||||
if (f) {
|
||||
std::stringstream out;
|
||||
auto NewIR = Thread->OpDispatcher->ViewIR();
|
||||
auto NewIR = IREmitter->ViewIR();
|
||||
FEXCore::IR::Dump(&out, &NewIR, RA);
|
||||
fmt::print(f,"IR-{} 0x{:x}:\n{}\n@@@@@\n", RA ? "post" : "pre", GuestRIP, out.str());
|
||||
|
||||
@@ -686,12 +716,12 @@ namespace FEXCore::Context {
|
||||
}
|
||||
};
|
||||
|
||||
static void ValidateIR(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
static void ValidateIR(FEXCore::Context::Context *ctx, IR::IREmitter *IREmitter) {
|
||||
// Convert to text, Parse, Convert to text again and make sure the texts match
|
||||
std::stringstream out;
|
||||
static auto compaction = IR::CreateIRCompaction(ctx->OpDispatcherAllocator);
|
||||
compaction->Run(Thread->OpDispatcher.get());
|
||||
auto NewIR = Thread->OpDispatcher->ViewIR();
|
||||
compaction->Run(IREmitter);
|
||||
auto NewIR = IREmitter->ViewIR();
|
||||
Dump(&out, &NewIR, nullptr);
|
||||
out.seekg(0);
|
||||
FEXCore::Utils::PooledAllocatorMalloc Allocator;
|
||||
@@ -710,151 +740,176 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
Context::GenerateIRResult Context::GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
uint8_t const *GuestCode{};
|
||||
GuestCode = reinterpret_cast<uint8_t const*>(GuestRIP);
|
||||
|
||||
bool HadDispatchError {false};
|
||||
Context::GenerateIRResult Context::GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo) {
|
||||
Thread->OpDispatcher->ReownOrClaimBuffer();
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
|
||||
uint64_t TotalInstructions {0};
|
||||
uint64_t TotalInstructionsLength {0};
|
||||
|
||||
Thread->FrontendDecoder->DecodeInstructionsAtEntry(GuestCode, GuestRIP);
|
||||
|
||||
auto CodeBlocks = Thread->FrontendDecoder->GetDecodedBlocks();
|
||||
std::shared_lock lk(CustomIRMutex);
|
||||
|
||||
auto Handler = CustomIRHandlers.find(GuestRIP);
|
||||
if (Handler != CustomIRHandlers.end()) {
|
||||
TotalInstructions = 1;
|
||||
TotalInstructionsLength = 1;
|
||||
std::get<0>(Handler->second)(GuestRIP, Thread->OpDispatcher.get());
|
||||
lk.unlock();
|
||||
} else {
|
||||
lk.unlock();
|
||||
uint8_t const *GuestCode{};
|
||||
GuestCode = reinterpret_cast<uint8_t const*>(GuestRIP);
|
||||
|
||||
Thread->OpDispatcher->ReownOrClaimBuffer();
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
Thread->OpDispatcher->BeginFunction(GuestRIP, CodeBlocks);
|
||||
bool HadDispatchError {false};
|
||||
|
||||
const uint8_t GPRSize = GetGPRSize();
|
||||
Thread->FrontendDecoder->DecodeInstructionsAtEntry(GuestCode, GuestRIP, [Thread](uint64_t BlockEntry, uint64_t Start, uint64_t Length) {
|
||||
if (Thread->LookupCache->AddBlockExecutableRange(BlockEntry, Start, Length)) {
|
||||
Thread->CTX->SyscallHandler->MarkGuestExecutableRange(Start, Length);
|
||||
}
|
||||
});
|
||||
|
||||
for (size_t j = 0; j < CodeBlocks->size(); ++j) {
|
||||
FEXCore::Frontend::Decoder::DecodedBlocks const &Block = CodeBlocks->at(j);
|
||||
// Set the block entry point
|
||||
Thread->OpDispatcher->SetNewBlockIfChanged(Block.Entry);
|
||||
auto CodeBlocks = Thread->FrontendDecoder->GetDecodedBlocks();
|
||||
|
||||
uint64_t BlockInstructionsLength {};
|
||||
Thread->OpDispatcher->BeginFunction(GuestRIP, CodeBlocks);
|
||||
|
||||
// Reset any block-specific state
|
||||
Thread->OpDispatcher->StartNewBlock();
|
||||
const uint8_t GPRSize = GetGPRSize();
|
||||
|
||||
uint64_t InstsInBlock = Block.NumInstructions;
|
||||
for (size_t j = 0; j < CodeBlocks->size(); ++j) {
|
||||
FEXCore::Frontend::Decoder::DecodedBlocks const &Block = CodeBlocks->at(j);
|
||||
// Set the block entry point
|
||||
Thread->OpDispatcher->SetNewBlockIfChanged(Block.Entry);
|
||||
|
||||
for (size_t i = 0; i < InstsInBlock; ++i) {
|
||||
FEXCore::X86Tables::X86InstInfo const* TableInfo {nullptr};
|
||||
FEXCore::X86Tables::DecodedInst const* DecodedInfo {nullptr};
|
||||
uint64_t BlockInstructionsLength {};
|
||||
|
||||
TableInfo = Block.DecodedInstructions[i].TableInfo;
|
||||
DecodedInfo = &Block.DecodedInstructions[i];
|
||||
bool IsLocked = DecodedInfo->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK;
|
||||
// Reset any block-specific state
|
||||
Thread->OpDispatcher->StartNewBlock();
|
||||
|
||||
if (Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL) {
|
||||
auto ExistingCodePtr = reinterpret_cast<uint64_t*>(Block.Entry + BlockInstructionsLength);
|
||||
|
||||
auto CodeChanged = Thread->OpDispatcher->_ValidateCode(ExistingCodePtr[0], ExistingCodePtr[1], (uintptr_t)ExistingCodePtr - GuestRIP, DecodedInfo->InstSize);
|
||||
|
||||
auto InvalidateCodeCond = Thread->OpDispatcher->_CondJump(CodeChanged);
|
||||
|
||||
auto CurrentBlock = Thread->OpDispatcher->GetCurrentBlock();
|
||||
auto CodeWasChangedBlock = Thread->OpDispatcher->CreateNewCodeBlockAtEnd();
|
||||
Thread->OpDispatcher->SetTrueJumpTarget(InvalidateCodeCond, CodeWasChangedBlock);
|
||||
|
||||
Thread->OpDispatcher->SetCurrentCodeBlock(CodeWasChangedBlock);
|
||||
Thread->OpDispatcher->_RemoveThreadCodeEntry();
|
||||
Thread->OpDispatcher->_ExitFunction(Thread->OpDispatcher->_EntrypointOffset(Block.Entry + BlockInstructionsLength - GuestRIP, GPRSize));
|
||||
|
||||
auto NextOpBlock = Thread->OpDispatcher->CreateNewCodeBlockAfter(CurrentBlock);
|
||||
|
||||
Thread->OpDispatcher->SetFalseJumpTarget(InvalidateCodeCond, NextOpBlock);
|
||||
Thread->OpDispatcher->SetCurrentCodeBlock(NextOpBlock);
|
||||
if (Config.x86dec_SynchronizeRIPOnAllBlocks) {
|
||||
// Ensure the RIP is synchronized to the context on block entry.
|
||||
// In the case of block linking, the RIP may not have synchronized.
|
||||
auto NewRIP = Thread->OpDispatcher->_EntrypointOffset(Block.Entry - GuestRIP, GPRSize);
|
||||
Thread->OpDispatcher->_StoreContext(GPRSize, IR::GPRClass, NewRIP, offsetof(FEXCore::Core::CPUState, rip));
|
||||
}
|
||||
|
||||
if (TableInfo && TableInfo->OpcodeDispatcher) {
|
||||
auto Fn = TableInfo->OpcodeDispatcher;
|
||||
Thread->OpDispatcher->HandledLock = false;
|
||||
Thread->OpDispatcher->ResetDecodeFailure();
|
||||
std::invoke(Fn, Thread->OpDispatcher, DecodedInfo);
|
||||
if (Thread->OpDispatcher->HadDecodeFailure()) {
|
||||
HadDispatchError = true;
|
||||
}
|
||||
else {
|
||||
if (Thread->OpDispatcher->HandledLock != IsLocked) {
|
||||
HadDispatchError = true;
|
||||
LogMan::Msg::EFmt("Missing LOCK HANDLER at 0x{:x}{{'{}'}}", Block.Entry + BlockInstructionsLength, TableInfo->Name ?: "UND");
|
||||
}
|
||||
BlockInstructionsLength += DecodedInfo->InstSize;
|
||||
TotalInstructionsLength += DecodedInfo->InstSize;
|
||||
++TotalInstructions;
|
||||
}
|
||||
}
|
||||
else {
|
||||
// Invalid instruction
|
||||
Thread->OpDispatcher->InvalidOp(DecodedInfo);
|
||||
Thread->OpDispatcher->_ExitFunction(Thread->OpDispatcher->_EntrypointOffset(Block.Entry - GuestRIP, GPRSize));
|
||||
}
|
||||
uint64_t InstsInBlock = Block.NumInstructions;
|
||||
|
||||
// If we had a dispatch error then leave early
|
||||
if (HadDispatchError) {
|
||||
if (TotalInstructions == 0) {
|
||||
// Couldn't handle any instruction in op dispatcher
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
return { nullptr, nullptr, 0, 0, 0, 0 };
|
||||
}
|
||||
else {
|
||||
const uint8_t GPRSize = GetGPRSize();
|
||||
for (size_t i = 0; i < InstsInBlock; ++i) {
|
||||
FEXCore::X86Tables::X86InstInfo const* TableInfo {nullptr};
|
||||
FEXCore::X86Tables::DecodedInst const* DecodedInfo {nullptr};
|
||||
|
||||
// We had some instructions. Early exit
|
||||
TableInfo = Block.DecodedInstructions[i].TableInfo;
|
||||
DecodedInfo = &Block.DecodedInstructions[i];
|
||||
bool IsLocked = DecodedInfo->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK;
|
||||
|
||||
if (ExtendedDebugInfo) {
|
||||
Thread->OpDispatcher->_GuestOpcode(Block.Entry + BlockInstructionsLength - GuestRIP);
|
||||
}
|
||||
|
||||
if (Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL) {
|
||||
auto ExistingCodePtr = reinterpret_cast<uint64_t*>(Block.Entry + BlockInstructionsLength);
|
||||
|
||||
auto CodeChanged = Thread->OpDispatcher->_ValidateCode(ExistingCodePtr[0], ExistingCodePtr[1], (uintptr_t)ExistingCodePtr - GuestRIP, DecodedInfo->InstSize);
|
||||
|
||||
auto InvalidateCodeCond = Thread->OpDispatcher->_CondJump(CodeChanged);
|
||||
|
||||
auto CurrentBlock = Thread->OpDispatcher->GetCurrentBlock();
|
||||
auto CodeWasChangedBlock = Thread->OpDispatcher->CreateNewCodeBlockAtEnd();
|
||||
Thread->OpDispatcher->SetTrueJumpTarget(InvalidateCodeCond, CodeWasChangedBlock);
|
||||
|
||||
Thread->OpDispatcher->SetCurrentCodeBlock(CodeWasChangedBlock);
|
||||
Thread->OpDispatcher->_ThreadRemoveCodeEntry();
|
||||
Thread->OpDispatcher->_ExitFunction(Thread->OpDispatcher->_EntrypointOffset(Block.Entry + BlockInstructionsLength - GuestRIP, GPRSize));
|
||||
|
||||
auto NextOpBlock = Thread->OpDispatcher->CreateNewCodeBlockAfter(CurrentBlock);
|
||||
|
||||
Thread->OpDispatcher->SetFalseJumpTarget(InvalidateCodeCond, NextOpBlock);
|
||||
Thread->OpDispatcher->SetCurrentCodeBlock(NextOpBlock);
|
||||
}
|
||||
|
||||
if (TableInfo && TableInfo->OpcodeDispatcher) {
|
||||
auto Fn = TableInfo->OpcodeDispatcher;
|
||||
Thread->OpDispatcher->HandledLock = false;
|
||||
Thread->OpDispatcher->ResetDecodeFailure();
|
||||
std::invoke(Fn, Thread->OpDispatcher, DecodedInfo);
|
||||
if (Thread->OpDispatcher->HadDecodeFailure()) {
|
||||
HadDispatchError = true;
|
||||
}
|
||||
else {
|
||||
if (Thread->OpDispatcher->HandledLock != IsLocked) {
|
||||
HadDispatchError = true;
|
||||
LogMan::Msg::EFmt("Missing LOCK HANDLER at 0x{:x}{{'{}'}}", Block.Entry + BlockInstructionsLength, TableInfo->Name ?: "UND");
|
||||
}
|
||||
BlockInstructionsLength += DecodedInfo->InstSize;
|
||||
TotalInstructionsLength += DecodedInfo->InstSize;
|
||||
++TotalInstructions;
|
||||
}
|
||||
}
|
||||
else {
|
||||
// Invalid instruction
|
||||
Thread->OpDispatcher->InvalidOp(DecodedInfo);
|
||||
Thread->OpDispatcher->_ExitFunction(Thread->OpDispatcher->_EntrypointOffset(Block.Entry - GuestRIP, GPRSize));
|
||||
}
|
||||
|
||||
// If we had a dispatch error then leave early
|
||||
if (HadDispatchError) {
|
||||
if (TotalInstructions == 0) {
|
||||
// Couldn't handle any instruction in op dispatcher
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
return { nullptr, nullptr, 0, 0, 0, 0 };
|
||||
}
|
||||
else {
|
||||
const uint8_t GPRSize = GetGPRSize();
|
||||
|
||||
// We had some instructions. Early exit
|
||||
Thread->OpDispatcher->_ExitFunction(Thread->OpDispatcher->_EntrypointOffset(Block.Entry + BlockInstructionsLength - GuestRIP, GPRSize));
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (Thread->OpDispatcher->FinishOp(DecodedInfo->PC + DecodedInfo->InstSize, i + 1 == InstsInBlock)) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (Thread->OpDispatcher->FinishOp(DecodedInfo->PC + DecodedInfo->InstSize, i + 1 == InstsInBlock)) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
Thread->OpDispatcher->Finalize();
|
||||
|
||||
Thread->FrontendDecoder->DelayedDisownBuffer();
|
||||
}
|
||||
|
||||
Thread->OpDispatcher->Finalize();
|
||||
IR::IREmitter *IREmitter = Thread->OpDispatcher.get();
|
||||
|
||||
auto ShouldDump = Thread->CTX->Config.DumpIR() != "no" || Thread->OpDispatcher->ShouldDump;
|
||||
// Debug
|
||||
{
|
||||
if (Thread->CTX->Config.DumpIR() != "no") {
|
||||
IRDumper(Thread, GuestRIP, nullptr);
|
||||
if (ShouldDump) {
|
||||
IRDumper(Thread, IREmitter, GuestRIP, nullptr);
|
||||
}
|
||||
|
||||
if (Thread->CTX->Config.ValidateIRarser) {
|
||||
ValidateIR(this, Thread);
|
||||
ValidateIR(this, IREmitter);
|
||||
}
|
||||
}
|
||||
|
||||
// Run the passmanager over the IR from the dispatcher
|
||||
Thread->PassManager->Run(Thread->OpDispatcher.get());
|
||||
Thread->PassManager->Run(IREmitter);
|
||||
|
||||
// Debug
|
||||
{
|
||||
if (Thread->CTX->Config.DumpIR() != "no") {
|
||||
IRDumper(Thread, GuestRIP, Thread->PassManager->HasPass("RA") ? Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA")->GetAllocationData() : nullptr);
|
||||
}
|
||||
|
||||
if (Thread->OpDispatcher->ShouldDump) {
|
||||
std::stringstream out;
|
||||
auto NewIR = Thread->OpDispatcher->ViewIR();
|
||||
FEXCore::IR::Dump(&out, &NewIR, Thread->PassManager->HasPass("RA") ? Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA")->GetAllocationData() : nullptr);
|
||||
LogMan::Msg::IFmt("IR 0x{:x}:\n{}\n@@@@@\n", GuestRIP, out.str());
|
||||
if (ShouldDump) {
|
||||
IRDumper(Thread, IREmitter, GuestRIP, Thread->PassManager->HasPass("RA") ? Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA")->GetAllocationData() : nullptr);
|
||||
}
|
||||
}
|
||||
|
||||
auto RAData = Thread->PassManager->HasPass("RA") ? Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA")->PullAllocationData() : nullptr;
|
||||
auto IRList = Thread->OpDispatcher->CreateIRCopy();
|
||||
auto IRList = IREmitter->CreateIRCopy();
|
||||
|
||||
Thread->OpDispatcher->DelayedDisownBuffer();
|
||||
Thread->FrontendDecoder->DelayedDisownBuffer();
|
||||
IREmitter->DelayedDisownBuffer();
|
||||
|
||||
return {
|
||||
.IRList = IRList,
|
||||
.RAData = RAData.release(),
|
||||
.RAData = std::move(RAData),
|
||||
.TotalInstructions = TotalInstructions,
|
||||
.TotalInstructionsLength = TotalInstructionsLength,
|
||||
.StartAddr = Thread->FrontendDecoder->DecodedMinAddress,
|
||||
@@ -865,27 +920,11 @@ namespace FEXCore::Context {
|
||||
Context::CompileCodeResult Context::CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
FEXCore::IR::IRListView *IRList {};
|
||||
FEXCore::Core::DebugData *DebugData {};
|
||||
FEXCore::IR::RegisterAllocationData *RAData {};
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData {};
|
||||
bool GeneratedIR {};
|
||||
uint64_t StartAddr {};
|
||||
uint64_t Length {};
|
||||
|
||||
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
|
||||
// Do we already have this in the IR cache?
|
||||
auto LocalEntry = Thread->LocalIRCache.find(GuestRIP);
|
||||
|
||||
if (LocalEntry != Thread->LocalIRCache.end()) {
|
||||
// Entry already exists
|
||||
// pull in the data
|
||||
IRList = LocalEntry->second.IR.get();
|
||||
DebugData = LocalEntry->second.DebugData.get();
|
||||
RAData = LocalEntry->second.RAData.get();
|
||||
StartAddr = LocalEntry->second.StartAddr;
|
||||
Length = LocalEntry->second.Length;
|
||||
|
||||
GeneratedIR = false;
|
||||
}
|
||||
|
||||
// JIT Code object cache lookup
|
||||
if (CodeObjectCacheService) {
|
||||
auto CodeCacheEntry = CodeObjectCacheService->FetchCodeObjectFromCache(GuestRIP);
|
||||
@@ -893,25 +932,33 @@ namespace FEXCore::Context {
|
||||
auto CompiledCode = Thread->CPUBackend->RelocateJITObjectCode(GuestRIP, CodeCacheEntry);
|
||||
if (CompiledCode) {
|
||||
return {
|
||||
.CompiledCode = CompiledCode,
|
||||
.IRData = nullptr, // No IR data generated
|
||||
.DebugData = nullptr, // nullptr here ensures that code serialization doesn't occur on from cache read
|
||||
.RAData = nullptr, // No RA data generated
|
||||
.GeneratedIR = false, // nullptr here ensures IR cache mechanisms won't run
|
||||
.StartAddr = 0, // Unused
|
||||
.Length = 0, // Unused
|
||||
.CompiledCode = CompiledCode,
|
||||
.IRData = nullptr, // No IR data generated
|
||||
.DebugData = nullptr, // nullptr here ensures that code serialization doesn't occur on from cache read
|
||||
.RAData = nullptr, // No RA data generated
|
||||
.GeneratedIR = false, // nullptr here ensures IR cache mechanisms won't run
|
||||
.StartAddr = 0, // Unused
|
||||
.Length = 0, // Unused
|
||||
};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (SourcecodeResolver && Config.GDBSymbols()) {
|
||||
auto AOTIRCacheEntry = SyscallHandler->LookupAOTIRCacheEntry(GuestRIP);
|
||||
if (AOTIRCacheEntry.Entry && !AOTIRCacheEntry.Entry->ContainsCode) {
|
||||
AOTIRCacheEntry.Entry->SourcecodeMap =
|
||||
SourcecodeResolver->GenerateMap(AOTIRCacheEntry.Entry->Filename, AOTIRCacheEntry.Entry->FileId);
|
||||
}
|
||||
}
|
||||
|
||||
// AOT IR bookkeeping and cache
|
||||
{
|
||||
auto [IRCopy, RACopy, DebugDataCopy, _StartAddr, _Length, _GeneratedIR] = IRCaptureCache.PreGenerateIRFetch(GuestRIP, IRList);
|
||||
if (_GeneratedIR) {
|
||||
// Setup pointers to internal structures
|
||||
IRList = IRCopy;
|
||||
RAData = RACopy;
|
||||
RAData = std::move(RACopy);
|
||||
DebugData = DebugDataCopy;
|
||||
StartAddr = _StartAddr;
|
||||
Length = _Length;
|
||||
@@ -921,11 +968,11 @@ namespace FEXCore::Context {
|
||||
|
||||
if (IRList == nullptr) {
|
||||
// Generate IR + Meta Info
|
||||
auto [IRCopy, RACopy, TotalInstructions, TotalInstructionsLength, _StartAddr, _Length] = GenerateIR(Thread, GuestRIP);
|
||||
auto [IRCopy, RACopy, TotalInstructions, TotalInstructionsLength, _StartAddr, _Length] = GenerateIR(Thread, GuestRIP, Config.GDBSymbols());
|
||||
|
||||
// Setup pointers to internal structures
|
||||
IRList = IRCopy;
|
||||
RAData = RACopy;
|
||||
RAData = std::move(RACopy);
|
||||
DebugData = new FEXCore::Core::DebugData();
|
||||
StartAddr = _StartAddr;
|
||||
Length = _Length;
|
||||
@@ -942,10 +989,10 @@ namespace FEXCore::Context {
|
||||
}
|
||||
// Attempt to get the CPU backend to compile this code
|
||||
return {
|
||||
.CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, IRList, DebugData, RAData),
|
||||
.CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, IRList, DebugData, RAData.get(), GetGdbServerStatus()),
|
||||
.IRData = IRList,
|
||||
.DebugData = DebugData,
|
||||
.RAData = RAData,
|
||||
.RAData = std::move(RAData),
|
||||
.GeneratedIR = GeneratedIR,
|
||||
.StartAddr = StartAddr,
|
||||
.Length = Length,
|
||||
@@ -966,8 +1013,8 @@ namespace FEXCore::Context {
|
||||
uintptr_t Context::CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
// Needs to be held for SMC interactions around concurrent compile and invalidation hazards
|
||||
auto InvalidationLk = Thread->CTX->SyscallHandler->CompileCodeLock(GuestRIP);
|
||||
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
|
||||
std::shared_lock lk(CodeInvalidationMutex);
|
||||
|
||||
// Is the code in the cache?
|
||||
// The backends only check L1 and L2, not L3
|
||||
@@ -978,16 +1025,14 @@ namespace FEXCore::Context {
|
||||
void *CodePtr {};
|
||||
FEXCore::IR::IRListView *IRList {};
|
||||
FEXCore::Core::DebugData *DebugData {};
|
||||
FEXCore::IR::RegisterAllocationData *RAData {};
|
||||
|
||||
bool GeneratedIR {};
|
||||
uint64_t StartAddr {}, Length {};
|
||||
|
||||
auto [Code, IR, Data, RA, Generated, _StartAddr, _Length] = CompileCode(Thread, GuestRIP);
|
||||
auto [Code, IR, Data, RAData, Generated, _StartAddr, _Length] = CompileCode(Thread, GuestRIP);
|
||||
CodePtr = Code;
|
||||
IRList = IR;
|
||||
DebugData = Data;
|
||||
RAData = RA;
|
||||
GeneratedIR = Generated;
|
||||
StartAddr = _StartAddr;
|
||||
Length = _Length;
|
||||
@@ -998,22 +1043,25 @@ namespace FEXCore::Context {
|
||||
|
||||
// The core managed to compile the code.
|
||||
if (Config.BlockJITNaming()) {
|
||||
auto FragmentBasePtr = reinterpret_cast<uint8_t *>(CodePtr);
|
||||
|
||||
if (DebugData) {
|
||||
auto GuestRIPLookup = this->SyscallHandler->LookupAOTIRCacheEntry(GuestRIP);
|
||||
auto GuestRIPLookup = SyscallHandler->LookupAOTIRCacheEntry(GuestRIP);
|
||||
|
||||
if (DebugData->Subblocks.size()) {
|
||||
for (auto& Subblock: DebugData->Subblocks) {
|
||||
auto BlockBasePtr = FragmentBasePtr + Subblock.HostCodeOffset;
|
||||
if (GuestRIPLookup.Entry) {
|
||||
Symbols.Register(CodePtr, DebugData->HostCodeSize, GuestRIPLookup.Entry->Filename, GuestRIP - GuestRIPLookup.Offset);
|
||||
Symbols.Register(BlockBasePtr, DebugData->HostCodeSize, GuestRIPLookup.Entry->Filename, GuestRIP - GuestRIPLookup.VAFileStart);
|
||||
} else {
|
||||
Symbols.Register((void*)Subblock.HostCodeStart, GuestRIP, Subblock.HostCodeSize);
|
||||
}
|
||||
Symbols.Register(BlockBasePtr, GuestRIP, Subblock.HostCodeSize);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
if (GuestRIPLookup.Entry) {
|
||||
Symbols.Register(CodePtr, DebugData->HostCodeSize, GuestRIPLookup.Entry->Filename, GuestRIP - GuestRIPLookup.Offset);
|
||||
Symbols.Register(FragmentBasePtr, DebugData->HostCodeSize, GuestRIPLookup.Entry->Filename, GuestRIP - GuestRIPLookup.VAFileStart);
|
||||
} else {
|
||||
Symbols.Register(CodePtr, GuestRIP, DebugData->HostCodeSize);
|
||||
Symbols.Register(FragmentBasePtr, GuestRIP, DebugData->HostCodeSize);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1046,7 +1094,7 @@ namespace FEXCore::Context {
|
||||
GuestRIP,
|
||||
StartAddr,
|
||||
Length,
|
||||
RAData,
|
||||
std::move(RAData),
|
||||
IRList,
|
||||
DebugData,
|
||||
GeneratedIR)) {
|
||||
@@ -1055,7 +1103,8 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
// Insert to lookup cache
|
||||
AddBlockMapping(Thread, GuestRIP, CodePtr, StartAddr, Length);
|
||||
// Pages containing this block are added via AddBlockExecutableRange before each page gets accessed in the frontend
|
||||
AddBlockMapping(Thread, GuestRIP, CodePtr);
|
||||
|
||||
return (uintptr_t)CodePtr;
|
||||
}
|
||||
@@ -1071,7 +1120,7 @@ namespace FEXCore::Context {
|
||||
// Now notify the thread that we are initialized
|
||||
Thread->ThreadWaiting.NotifyAll();
|
||||
|
||||
if (Thread != Thread->CTX->ParentThread || StartPaused) {
|
||||
if (Thread != Thread->CTX->ParentThread || StartPaused || Thread->StartPaused) {
|
||||
// Parent thread doesn't need to wait to run
|
||||
Thread->StartRunning.Wait();
|
||||
}
|
||||
@@ -1083,7 +1132,7 @@ namespace FEXCore::Context {
|
||||
|
||||
Thread->RunningEvents.Running = true;
|
||||
|
||||
Thread->CPUBackend->ExecuteDispatch(Thread->CurrentFrame);
|
||||
Thread->CTX->Dispatcher->ExecuteDispatch(Thread->CurrentFrame);
|
||||
|
||||
Thread->RunningEvents.Running = false;
|
||||
}
|
||||
@@ -1125,35 +1174,107 @@ namespace FEXCore::Context {
|
||||
|
||||
for (auto it = lower; it != upper; it++) {
|
||||
for (auto Address: it->second) {
|
||||
Context::RemoveThreadCodeEntry(Thread, Address);
|
||||
Context::ThreadRemoveCodeEntry(Thread, Address);
|
||||
}
|
||||
it->second.clear();
|
||||
}
|
||||
}
|
||||
|
||||
void InvalidateGuestCodeRange(FEXCore::Context::Context *CTX, uint64_t Start, uint64_t Length) {
|
||||
std::lock_guard<std::mutex> lk(CTX->ThreadCreationMutex);
|
||||
static void InvalidateGuestCodeRangeInternal(FEXCore::Context::Context *CTX, uint64_t Start, uint64_t Length) {
|
||||
std::lock_guard lk(CTX->ThreadCreationMutex);
|
||||
|
||||
for (auto &Thread : CTX->Threads) {
|
||||
if (Thread->RunningEvents.Running.load()) {
|
||||
InvalidateGuestThreadCodeRange(Thread, Start, Length);
|
||||
InvalidateGuestThreadCodeRange(Thread, Start, Length);
|
||||
}
|
||||
}
|
||||
|
||||
void InvalidateGuestCodeRange(FEXCore::Context::Context *CTX, uint64_t Start, uint64_t Length) {
|
||||
FHU::ScopedSignalMaskWithUniqueLock CodeInvalidationLock(CTX->CodeInvalidationMutex);
|
||||
|
||||
InvalidateGuestCodeRangeInternal(CTX, Start, Length);
|
||||
}
|
||||
|
||||
void InvalidateGuestCodeRange(FEXCore::Context::Context *CTX, uint64_t Start, uint64_t Length, std::function<void(uint64_t start, uint64_t Length)> CallAfter) {
|
||||
FHU::ScopedSignalMaskWithUniqueLock CodeInvalidationLock(CTX->CodeInvalidationMutex);
|
||||
|
||||
InvalidateGuestCodeRangeInternal(CTX, Start, Length);
|
||||
CallAfter(Start, Length);
|
||||
}
|
||||
|
||||
void Context::MarkMemoryShared() {
|
||||
if (!IsMemoryShared) {
|
||||
IsMemoryShared = true;
|
||||
|
||||
if (Config.TSOAutoMigration) {
|
||||
LogMan::Msg::IFmt("Migrating to shared memory mode");
|
||||
|
||||
std::lock_guard<std::mutex> lkThreads(ThreadCreationMutex);
|
||||
LogMan::Throw::AFmt(Threads.size() == 1, "First MarkMemoryShared called must be before creating any threads");
|
||||
|
||||
auto Thread = Threads[0];
|
||||
|
||||
// Only the lookup cache is cleared here, so that old code can keep running until next compilation
|
||||
std::lock_guard<std::recursive_mutex> lkLookupCache(Thread->LookupCache->WriteLock);
|
||||
Thread->LookupCache->ClearCache();
|
||||
|
||||
// DebugStore also needs to be cleared
|
||||
Thread->DebugStore.clear();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void Context::RemoveThreadCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
void MarkMemoryShared(FEXCore::Context::Context *CTX) {
|
||||
CTX->MarkMemoryShared();
|
||||
}
|
||||
|
||||
void Context::ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker) {
|
||||
std::shared_lock lk(Thread->CTX->CodeInvalidationMutex);
|
||||
|
||||
Thread->LookupCache->AddBlockLink(GuestDestination, HostLink, delinker);
|
||||
}
|
||||
|
||||
void Context::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
LogMan::Throw::AFmt(Thread->CTX->CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to be unique_locked here");
|
||||
|
||||
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
|
||||
|
||||
Thread->LocalIRCache.erase(GuestRIP);
|
||||
Thread->DebugStore.erase(GuestRIP);
|
||||
Thread->LookupCache->Erase(GuestRIP);
|
||||
}
|
||||
|
||||
CustomIRResult Context::AddCustomIREntrypoint(uintptr_t Entrypoint, std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)> Handler, void *Creator, void *Data) {
|
||||
LOGMAN_THROW_A_FMT(Config.Is64BitMode || !(Entrypoint >> 32), "64-bit Entrypoint in 32-bit mode {:x}", Entrypoint);
|
||||
|
||||
std::unique_lock lk(CustomIRMutex);
|
||||
|
||||
auto InsertedIterator = CustomIRHandlers.emplace(Entrypoint, std::tuple(Handler, Creator, Data));
|
||||
|
||||
if (!InsertedIterator.second) {
|
||||
const auto &[fn, Creator, Data] = InsertedIterator.first->second;
|
||||
return CustomIRResult(std::move(lk), Creator, Data);
|
||||
} else {
|
||||
lk.unlock();
|
||||
return CustomIRResult(std::move(lk), 0, 0);
|
||||
}
|
||||
}
|
||||
|
||||
void Context::RemoveCustomIREntrypoint(uintptr_t Entrypoint) {
|
||||
LOGMAN_THROW_A_FMT(Config.Is64BitMode || !(Entrypoint >> 32), "64-bit Entrypoint in 32-bit mode {:x}", Entrypoint);
|
||||
|
||||
std::scoped_lock lk(CustomIRMutex);
|
||||
|
||||
InvalidateGuestCodeRange(this, Entrypoint, 1, [this](uint64_t Entrypoint, uint64_t) {
|
||||
CustomIRHandlers.erase(Entrypoint);
|
||||
});
|
||||
}
|
||||
|
||||
// Debug interface
|
||||
void Context::CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) {
|
||||
uint64_t RIPBackup = Thread->CurrentFrame->State.rip;
|
||||
Thread->CurrentFrame->State.rip = RIP;
|
||||
|
||||
// Erase the RIP from all the storage backings if it exists
|
||||
RemoveThreadCodeEntry(Thread, RIP);
|
||||
ThreadRemoveCodeEntry(Thread, RIP);
|
||||
|
||||
// We don't care if compilation passes or not
|
||||
CompileBlock(Thread->CurrentFrame, RIP);
|
||||
@@ -1171,8 +1292,8 @@ namespace FEXCore::Context {
|
||||
|
||||
bool Context::GetDebugDataForRIP(uint64_t RIP, FEXCore::Core::DebugData *Data) {
|
||||
std::lock_guard<std::recursive_mutex> lk(ParentThread->LookupCache->WriteLock);
|
||||
auto it = ParentThread->LocalIRCache.find(RIP);
|
||||
if (it == ParentThread->LocalIRCache.end()) {
|
||||
auto it = ParentThread->DebugStore.find(RIP);
|
||||
if (it == ParentThread->DebugStore.end()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
+164
-134
@@ -16,20 +16,23 @@
|
||||
#include <array>
|
||||
#include <bit>
|
||||
#include <cmath>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
#include <stddef.h>
|
||||
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
#include "aarch64/constants-aarch64.h"
|
||||
#include "aarch64/operands-aarch64.h"
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
#include "code-buffer-vixl.h"
|
||||
#include "platform-vixl.h"
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
#include <aarch64/constants-aarch64.h>
|
||||
#include <aarch64/cpu-aarch64.h>
|
||||
#include <aarch64/operands-aarch64.h>
|
||||
#include <code-buffer-vixl.h>
|
||||
#include <platform-vixl.h>
|
||||
|
||||
#include <sys/syscall.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#define STATE_PTR(STATE_TYPE, FIELD) \
|
||||
MemOperand(STATE, offsetof(FEXCore::Core::STATE_TYPE, FIELD))
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
using namespace vixl;
|
||||
@@ -38,12 +41,11 @@ using namespace vixl::aarch64;
|
||||
static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
|
||||
#define STATE x28
|
||||
|
||||
Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config)
|
||||
: FEXCore::CPU::Dispatcher(ctx, Thread), Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE) {
|
||||
SRAEnabled = config.StaticRegisterAssignment;
|
||||
Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config)
|
||||
: FEXCore::CPU::Dispatcher(ctx, config), Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE) {
|
||||
SetAllowAssembler(true);
|
||||
|
||||
DispatchPtr = GetCursorAddress<CPUBackend::AsmDispatch>();
|
||||
DispatchPtr = GetCursorAddress<AsmDispatch>();
|
||||
|
||||
// while (true) {
|
||||
// Ptr = FindBlock(RIP)
|
||||
@@ -53,12 +55,9 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
// Ptr();
|
||||
// }
|
||||
|
||||
Literal l_PagePtr {Thread->LookupCache->GetPagePointer()};
|
||||
Literal l_CTX {reinterpret_cast<uintptr_t>(CTX)};
|
||||
Literal l_Sleep {reinterpret_cast<uint64_t>(SleepThread)};
|
||||
Literal l_CompileBlock {GetCompileBlockPtr()};
|
||||
Literal l_ExitFunctionLink {config.ExitFunctionLink};
|
||||
Literal l_ExitFunctionLinkThis {config.ExitFunctionLinkThis};
|
||||
|
||||
// Push all the register we need to save
|
||||
PushCalleeSavedRegisters();
|
||||
@@ -71,11 +70,11 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
// Save this stack pointer so we can cleanly shutdown the emulation with a long jump
|
||||
// regardless of where we were in the stack
|
||||
add(x0, sp, 0);
|
||||
str(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)));
|
||||
str(x0, STATE_PTR(CpuStateFrame, ReturningStackLocation));
|
||||
|
||||
AbsoluteLoopTopAddressFillSRA = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (SRAEnabled) {
|
||||
if (config.StaticRegisterAllocation) {
|
||||
FillStaticRegs();
|
||||
}
|
||||
|
||||
@@ -92,11 +91,11 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
|
||||
// Load in our RIP
|
||||
// Don't modify x2 since it contains our RIP once the block doesn't exist
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
|
||||
ldr(x2, STATE_PTR(CpuStateFrame, State.rip));
|
||||
auto RipReg = x2;
|
||||
|
||||
// L1 Cache
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.L1Pointer)));
|
||||
ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
|
||||
and_(x3, RipReg, LookupCache::L1_ENTRIES_MASK);
|
||||
add(x0, x0, Operand(x3, Shift::LSL, 4));
|
||||
@@ -104,21 +103,17 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
cmp(x0, RipReg);
|
||||
b(&FullLookup, Condition::ne);
|
||||
|
||||
if (!config.ExecuteBlocksWithCall) {
|
||||
br(x3);
|
||||
} else {
|
||||
b(&CallBlock);
|
||||
}
|
||||
br(x3);
|
||||
|
||||
// L1C check failed, do a full lookup
|
||||
bind(&FullLookup);
|
||||
|
||||
// This is the block cache lookup routine
|
||||
// It matches what is going on it LookupCache.h::FindBlock
|
||||
ldr(x0, &l_PagePtr);
|
||||
ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
|
||||
|
||||
// Mask the address by the virtual address size so we can check for aliases
|
||||
uint64_t VirtualMemorySize = Thread->LookupCache->GetVirtualMemorySize();
|
||||
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
|
||||
if (std::popcount(VirtualMemorySize) == 1) {
|
||||
and_(x3, RipReg, VirtualMemorySize - 1);
|
||||
}
|
||||
@@ -157,44 +152,21 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
// If we've made it here then we have a real compiled block
|
||||
{
|
||||
// update L1 cache
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.L1Pointer)));
|
||||
ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
|
||||
and_(x1, RipReg, LookupCache::L1_ENTRIES_MASK);
|
||||
add(x0, x0, Operand(x1, Shift::LSL, 4));
|
||||
stp(x3, x2, MemOperand(x0));
|
||||
|
||||
// Jump to the block
|
||||
if (!config.ExecuteBlocksWithCall) {
|
||||
br(x3);
|
||||
} else {
|
||||
bind(&CallBlock);
|
||||
mov(x0, STATE);
|
||||
blr(x3);
|
||||
|
||||
if (CTX->GetGdbServerStatus()) {
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
// This happens when single stepping
|
||||
|
||||
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
ldr(x0, &l_CTX);
|
||||
ldr(w0, MemOperand(x0, offsetof(FEXCore::Context::Context, Config.RunningMode)));
|
||||
// If the value == 0 then branch to the top
|
||||
cbz(x0, &LoopTop);
|
||||
// Else we need to pause now
|
||||
b(&ThreadPauseHandler);
|
||||
} else {
|
||||
// Unconditionally loop to the top
|
||||
// We will only stop on error when compiling a block or signal
|
||||
b(&LoopTop);
|
||||
}
|
||||
}
|
||||
br(x3);
|
||||
}
|
||||
}
|
||||
|
||||
{
|
||||
bind(&ExitSpillSRA);
|
||||
ThreadStopHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
|
||||
if (SRAEnabled)
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
|
||||
ThreadStopHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
@@ -209,7 +181,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
constexpr bool SignalSafeCompile = true;
|
||||
{
|
||||
ExitFunctionLinkerAddress = GetCursorAddress<uint64_t>();
|
||||
if (SRAEnabled)
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
@@ -231,11 +203,10 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
svc(0);
|
||||
}
|
||||
|
||||
ldr(x0, &l_ExitFunctionLinkThis);
|
||||
mov(x1, STATE);
|
||||
mov(x2, lr);
|
||||
mov(x0, STATE);
|
||||
mov(x1, lr);
|
||||
|
||||
ldr(x3, &l_ExitFunctionLink);
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
|
||||
blr(x3);
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
@@ -256,7 +227,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
mov(x0, x4);
|
||||
}
|
||||
|
||||
if (SRAEnabled)
|
||||
if (config.StaticRegisterAllocation)
|
||||
FillStaticRegs();
|
||||
br(x0);
|
||||
}
|
||||
@@ -265,7 +236,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
{
|
||||
bind(&NoBlock);
|
||||
|
||||
if (SRAEnabled)
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
@@ -312,7 +283,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
add(sp, sp, 16);
|
||||
}
|
||||
|
||||
if (SRAEnabled)
|
||||
if (config.StaticRegisterAllocation)
|
||||
FillStaticRegs();
|
||||
|
||||
b(&LoopTop);
|
||||
@@ -329,42 +300,44 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
{
|
||||
// Guest SIGILL handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
UnimplementedInstructionAddress = GetCursorAddress<uint64_t>();
|
||||
GuestSignal_SIGILL = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (SRAEnabled)
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
|
||||
hlt(0);
|
||||
}
|
||||
|
||||
{
|
||||
// Guest Overflow handler
|
||||
// Guest SIGTRAP handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
OverflowExceptionInstructionAddress = GetCursorAddress<uint64_t>();
|
||||
GuestSignal_SIGTRAP = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (SRAEnabled)
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
|
||||
LoadConstant(x0, reinterpret_cast<uint64_t>(&SynchronousFaultData));
|
||||
LoadConstant(w1, 1);
|
||||
strb(w1, MemOperand(x0, offsetof(Dispatcher::SynchronousFaultDataStruct, FaultToTopAndGeneratedException)));
|
||||
LoadConstant(w1, X86State::X86_TRAPNO_OF);
|
||||
str(w1, MemOperand(x0, offsetof(Dispatcher::SynchronousFaultDataStruct, TrapNo)));
|
||||
LoadConstant(w1, 0x80);
|
||||
str(w1, MemOperand(x0, offsetof(Dispatcher::SynchronousFaultDataStruct, si_code)));
|
||||
LoadConstant(x1, 0);
|
||||
str(w1, MemOperand(x0, offsetof(Dispatcher::SynchronousFaultDataStruct, err_code)));
|
||||
brk(0);
|
||||
}
|
||||
|
||||
{
|
||||
// Guest Overflow handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
GuestSignal_SIGSEGV = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
|
||||
// hlt/udf = SIGILL
|
||||
// brk = SIGTRAP
|
||||
// ??? = SIGSEGV
|
||||
// Force a SIGSEGV by loading zero
|
||||
LoadConstant(x1, 0);
|
||||
ldr(x1, MemOperand(x1));
|
||||
}
|
||||
|
||||
{
|
||||
ThreadPauseHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
|
||||
if (SRAEnabled)
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
|
||||
bind(&ThreadPauseHandler);
|
||||
@@ -399,7 +372,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
// On return to the thunk, the thunk can get whatever its return value is from the thread context depending on ABI handling on its end
|
||||
// When the thunk itself returns, it'll do its regular return logic there
|
||||
// void ReentrantCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP);
|
||||
CallbackPtr = GetCursorAddress<CPUBackend::JITCallback>();
|
||||
CallbackPtr = GetCursorAddress<JITCallback>();
|
||||
|
||||
// We expect the thunk to have previously pushed the registers it was using
|
||||
PushCalleeSavedRegisters();
|
||||
@@ -408,46 +381,39 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
mov(STATE, x0);
|
||||
|
||||
// Make sure to adjust the refcounter so we don't clear the cache now
|
||||
LoadConstant(x0, reinterpret_cast<uint64_t>(&SignalHandlerRefCounter));
|
||||
ldr(w2, MemOperand(x0));
|
||||
ldr(w2, STATE_PTR(CpuStateFrame, SignalHandlerRefCounter));
|
||||
add(w2, w2, 1);
|
||||
str(w2, MemOperand(x0));
|
||||
str(w2, STATE_PTR(CpuStateFrame, SignalHandlerRefCounter));
|
||||
|
||||
// Now push the callback return trampoline to the guest stack
|
||||
// Guest will be misaligned because calling a thunk won't correct the guest's stack once we call the callback from the host
|
||||
LoadConstant(x0, CTX->X86CodeGen.CallbackReturn);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])));
|
||||
ldr(x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
sub(x2, x2, 16);
|
||||
str(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])));
|
||||
str(x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
|
||||
// Store the trampoline to the guest stack
|
||||
// Guest stack is now correctly misaligned after a regular call instruction
|
||||
str(x0, MemOperand(x2));
|
||||
|
||||
// Store RIP to the context state
|
||||
str(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
|
||||
str(x1, STATE_PTR(CpuStateFrame, State.rip));
|
||||
|
||||
// load static regs
|
||||
if (SRAEnabled)
|
||||
if (config.StaticRegisterAllocation)
|
||||
FillStaticRegs();
|
||||
|
||||
// Now go back to the regular dispatcher loop
|
||||
b(&LoopTop);
|
||||
}
|
||||
|
||||
// Long division helpers
|
||||
uint64_t LUDIVHandler{};
|
||||
uint64_t LDIVHandler{};
|
||||
uint64_t LUREMHandler{};
|
||||
uint64_t LREMHandler{};
|
||||
|
||||
{
|
||||
LUDIVHandler = GetCursorAddress<uint64_t>();
|
||||
LUDIVHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LUDIV)));
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUDIV));
|
||||
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
@@ -462,11 +428,11 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
}
|
||||
|
||||
{
|
||||
LDIVHandler = GetCursorAddress<uint64_t>();
|
||||
LDIVHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LDIV)));
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LDIV));
|
||||
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
@@ -481,11 +447,11 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
}
|
||||
|
||||
{
|
||||
LUREMHandler = GetCursorAddress<uint64_t>();
|
||||
LUREMHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LUREM)));
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUREM));
|
||||
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
@@ -500,11 +466,11 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
}
|
||||
|
||||
{
|
||||
LREMHandler = GetCursorAddress<uint64_t>();
|
||||
LREMHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LREM)));
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LREM));
|
||||
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
@@ -518,12 +484,9 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
ret();
|
||||
}
|
||||
|
||||
place(&l_PagePtr);
|
||||
place(&l_CTX);
|
||||
place(&l_Sleep);
|
||||
place(&l_CompileBlock);
|
||||
place(&l_ExitFunctionLink);
|
||||
place(&l_ExitFunctionLinkThis);
|
||||
|
||||
|
||||
FinalizeCode();
|
||||
@@ -539,55 +502,122 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
if (CTX->Config.GlobalJITNaming()) {
|
||||
CTX->Symbols.RegisterJITSpace(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr));
|
||||
}
|
||||
|
||||
// Setup dispatcher specific pointers that need to be accessed from JIT code
|
||||
{
|
||||
auto &Pointers = ThreadState->CurrentFrame->Pointers.AArch64;
|
||||
|
||||
Pointers.DispatcherLoopTop = AbsoluteLoopTopAddress;
|
||||
Pointers.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
|
||||
Pointers.ThreadStopHandlerSpillSRA = ThreadStopHandlerAddressSpillSRA;
|
||||
Pointers.ThreadPauseHandlerSpillSRA = ThreadPauseHandlerAddressSpillSRA;
|
||||
Pointers.UnimplementedInstructionHandler = UnimplementedInstructionAddress;
|
||||
Pointers.OverflowExceptionHandler = OverflowExceptionInstructionAddress;
|
||||
Pointers.SignalReturnHandler = SignalHandlerReturnAddress;
|
||||
Pointers.L1Pointer = Thread->LookupCache->GetL1Pointer();
|
||||
Pointers.LUDIVHandler = LUDIVHandler;
|
||||
Pointers.LDIVHandler = LDIVHandler;
|
||||
Pointers.LUREMHandler = LUREMHandler;
|
||||
Pointers.LREMHandler = LREMHandler;
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Dispatcher::SpillSRA(void *ucontext, uint32_t IgnoreMask) {
|
||||
for(int i = 0; i < SRA64.size(); i++) {
|
||||
// Used by GenerateGDBPauseCheck, GenerateInterpreterTrampoline, destination buffer is set before use
|
||||
static thread_local vixl::aarch64::Assembler emit((uint8_t*)&emit, 1);
|
||||
|
||||
size_t Arm64Dispatcher::GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) {
|
||||
|
||||
*emit.GetBuffer() = vixl::CodeBuffer(CodeBuffer, MaxGDBPauseCheckSize);
|
||||
|
||||
vixl::CodeBufferCheckScope scope(&emit, MaxGDBPauseCheckSize, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
|
||||
aarch64::Label RunBlock;
|
||||
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
// This happens when single stepping
|
||||
|
||||
static_assert(sizeof(FEXCore::Context::Context::Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
emit.ldr(x0, STATE_PTR(CpuStateFrame, Thread)); // Get thread
|
||||
emit.ldr(x0, MemOperand(x0, offsetof(FEXCore::Core::InternalThreadState, CTX))); // Get Context
|
||||
emit.ldr(w0, MemOperand(x0, offsetof(FEXCore::Context::Context, Config.RunningMode)));
|
||||
|
||||
// If the value == 0 then we don't need to stop
|
||||
emit.cbz(w0, &RunBlock);
|
||||
{
|
||||
Literal l_GuestRIP {GuestRIP};
|
||||
// Make sure RIP is syncronized to the context
|
||||
emit.ldr(x0, &l_GuestRIP);
|
||||
emit.str(x0, STATE_PTR(CpuStateFrame, State.rip));
|
||||
|
||||
// Stop the thread
|
||||
emit.ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.ThreadPauseHandlerSpillSRA));
|
||||
emit.br(x0);
|
||||
emit.place(&l_GuestRIP);
|
||||
}
|
||||
emit.bind(&RunBlock);
|
||||
emit.FinalizeCode();
|
||||
|
||||
auto UsedBytes = emit.GetBuffer()->GetCursorOffset();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency(CodeBuffer, UsedBytes);
|
||||
return UsedBytes;
|
||||
}
|
||||
|
||||
size_t Arm64Dispatcher::GenerateInterpreterTrampoline(uint8_t *CodeBuffer) {
|
||||
LOGMAN_THROW_AA_FMT(!config.StaticRegisterAllocation, "GenerateInterpreterTrampoline dispatcher does not support SRA");
|
||||
|
||||
*emit.GetBuffer() = vixl::CodeBuffer(CodeBuffer, MaxInterpreterTrampolineSize);
|
||||
|
||||
vixl::CodeBufferCheckScope scope(&emit, MaxInterpreterTrampolineSize, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
|
||||
aarch64::Label InlineIRData;
|
||||
|
||||
emit.mov(x0, STATE);
|
||||
emit.adr(x1, &InlineIRData);
|
||||
|
||||
emit.ldr(x3, STATE_PTR(CpuStateFrame, Pointers.Interpreter.FragmentExecuter));
|
||||
emit.blr(x3);
|
||||
|
||||
emit.ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.DispatcherLoopTop));
|
||||
emit.br(x0);
|
||||
|
||||
emit.bind(&InlineIRData);
|
||||
|
||||
emit.FinalizeCode();
|
||||
|
||||
auto UsedBytes = emit.GetBuffer()->GetCursorOffset();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency(CodeBuffer, UsedBytes);
|
||||
return UsedBytes;
|
||||
}
|
||||
|
||||
void Arm64Dispatcher::SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) {
|
||||
for (size_t i = 0; i < SRA64.size(); i++) {
|
||||
if (IgnoreMask & (1U << SRA64[i].GetCode())) {
|
||||
// Skip this one, it's already spilled
|
||||
continue;
|
||||
}
|
||||
ThreadState->CurrentFrame->State.gregs[i] = ArchHelpers::Context::GetArmReg(ucontext, SRA64[i].GetCode());
|
||||
Thread->CurrentFrame->State.gregs[i] = ArchHelpers::Context::GetArmReg(ucontext, SRA64[i].GetCode());
|
||||
}
|
||||
|
||||
for(int i = 0; i < SRAFPR.size(); i++) {
|
||||
auto FPR = ArchHelpers::Context::GetArmFPR(ucontext, SRAFPR[i].GetCode());
|
||||
memcpy(&ThreadState->CurrentFrame->State.xmm[i][0], &FPR, sizeof(__uint128_t));
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
auto FPR = ArchHelpers::Context::GetArmFPR(ucontext, SRAFPR[i].GetCode());
|
||||
memcpy(&Thread->CurrentFrame->State.xmm.avx.data[i][0], &FPR, sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
auto FPR = ArchHelpers::Context::GetArmFPR(ucontext, SRAFPR[i].GetCode());
|
||||
memcpy(&Thread->CurrentFrame->State.xmm.sse.data[i][0], &FPR, sizeof(__uint128_t));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
void Arm64Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// Setup dispatcher specific pointers that need to be accessed from JIT code
|
||||
{
|
||||
auto &Common = Thread->CurrentFrame->Pointers.Common;
|
||||
|
||||
void InterpreterCore::CreateAsmDispatch(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
DispatcherConfig config;
|
||||
config.ExecuteBlocksWithCall = true;
|
||||
Common.DispatcherLoopTop = AbsoluteLoopTopAddress;
|
||||
Common.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
|
||||
Common.ExitFunctionLinker = ExitFunctionLinkerAddress;
|
||||
Common.ThreadStopHandlerSpillSRA = ThreadStopHandlerAddressSpillSRA;
|
||||
Common.ThreadPauseHandlerSpillSRA = ThreadPauseHandlerAddressSpillSRA;
|
||||
Common.GuestSignal_SIGILL = GuestSignal_SIGILL;
|
||||
Common.GuestSignal_SIGTRAP = GuestSignal_SIGTRAP;
|
||||
Common.GuestSignal_SIGSEGV = GuestSignal_SIGSEGV;
|
||||
Common.SignalReturnHandler = SignalHandlerReturnAddress;
|
||||
|
||||
Dispatcher = std::make_unique<Arm64Dispatcher>(ctx, Thread, config);
|
||||
DispatchPtr = Dispatcher->DispatchPtr;
|
||||
CallbackPtr = Dispatcher->CallbackPtr;
|
||||
|
||||
// TODO: It feels wrong to initialize this way
|
||||
ctx->InterpreterCallbackReturn = Dispatcher->ReturnPtr;
|
||||
auto &AArch64 = Thread->CurrentFrame->Pointers.AArch64;
|
||||
AArch64.LUDIVHandler = LUDIVHandlerAddress;
|
||||
AArch64.LDIVHandler = LDIVHandlerAddress;
|
||||
AArch64.LUREMHandler = LUREMHandlerAddress;
|
||||
AArch64.LREMHandler = LREMHandlerAddress;
|
||||
}
|
||||
}
|
||||
|
||||
#endif
|
||||
std::unique_ptr<Dispatcher> Dispatcher::CreateArm64(FEXCore::Context::Context *CTX, const DispatcherConfig &Config) {
|
||||
return std::make_unique<Arm64Dispatcher>(CTX, Config);
|
||||
}
|
||||
|
||||
}
|
||||
@@ -15,10 +15,20 @@ namespace FEXCore::CPU {
|
||||
|
||||
class Arm64Dispatcher final : public Dispatcher, public Arm64Emitter {
|
||||
public:
|
||||
Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config);
|
||||
Arm64Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config);
|
||||
void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) override;
|
||||
size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) override;
|
||||
|
||||
protected:
|
||||
void SpillSRA(void *ucontext, uint32_t IgnoreMask) override;
|
||||
void SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) override;
|
||||
|
||||
private:
|
||||
// Long division helpers
|
||||
uint64_t LUDIVHandlerAddress{};
|
||||
uint64_t LDIVHandlerAddress{};
|
||||
uint64_t LUREMHandlerAddress{};
|
||||
uint64_t LREMHandlerAddress{};
|
||||
};
|
||||
|
||||
}
|
||||
+149
-91
@@ -1,4 +1,5 @@
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
@@ -39,7 +40,7 @@ void Dispatcher::SleepThread(FEXCore::Context::Context *ctx, FEXCore::Core::CpuS
|
||||
ctx->IdleWaitCV.notify_all();
|
||||
}
|
||||
|
||||
ArchHelpers::Context::ContextBackup* Dispatcher::StoreThreadState(int Signal, void *ucontext) {
|
||||
ArchHelpers::Context::ContextBackup* Dispatcher::StoreThreadState(FEXCore::Core::InternalThreadState *Thread, int Signal, void *ucontext) {
|
||||
// We can end up getting a signal at any point in our host state
|
||||
// Jump to a handler that saves all state so we can safely return
|
||||
uint64_t OldSP = ArchHelpers::Context::GetSp(ucontext);
|
||||
@@ -64,7 +65,7 @@ ArchHelpers::Context::ContextBackup* Dispatcher::StoreThreadState(int Signal, vo
|
||||
// Save guest state
|
||||
// We can't guarantee if registers are in context or host GPRs
|
||||
// So we need to save everything
|
||||
memcpy(&Context->GuestState, ThreadState->CurrentFrame, sizeof(FEXCore::Core::CPUState));
|
||||
memcpy(&Context->GuestState, Thread->CurrentFrame, sizeof(FEXCore::Core::CPUState));
|
||||
|
||||
// Set the new SP
|
||||
ArchHelpers::Context::SetSp(ucontext, NewSP);
|
||||
@@ -81,13 +82,13 @@ ArchHelpers::Context::ContextBackup* Dispatcher::StoreThreadState(int Signal, vo
|
||||
Context->SigInfoLocation = 0;
|
||||
|
||||
// Store fault to top status and then reset it
|
||||
Context->FaultToTopAndGeneratedException = SynchronousFaultData.FaultToTopAndGeneratedException;
|
||||
SynchronousFaultData.FaultToTopAndGeneratedException = false;
|
||||
Context->FaultToTopAndGeneratedException = Thread->CurrentFrame->SynchronousFaultData.FaultToTopAndGeneratedException;
|
||||
Thread->CurrentFrame->SynchronousFaultData.FaultToTopAndGeneratedException = false;
|
||||
|
||||
return Context;
|
||||
}
|
||||
|
||||
void Dispatcher::RestoreThreadState(void *ucontext) {
|
||||
void Dispatcher::RestoreThreadState(FEXCore::Core::InternalThreadState *Thread, void *ucontext) {
|
||||
uint64_t OldSP{};
|
||||
if (CTX->Config.Core() == FEXCore::Config::CONFIG_IRJIT) {
|
||||
OldSP = ArchHelpers::Context::GetSp(ucontext);
|
||||
@@ -98,17 +99,18 @@ void Dispatcher::RestoreThreadState(void *ucontext) {
|
||||
SignalFrames.pop();
|
||||
}
|
||||
|
||||
const bool IsAVXEnabled = CTX->Config.EnableAVX;
|
||||
uintptr_t NewSP = OldSP;
|
||||
auto Context = reinterpret_cast<ArchHelpers::Context::ContextBackup*>(NewSP);
|
||||
|
||||
// First thing, reset the guest state
|
||||
memcpy(ThreadState->CurrentFrame, &Context->GuestState, sizeof(FEXCore::Core::CPUState));
|
||||
memcpy(Thread->CurrentFrame, &Context->GuestState, sizeof(FEXCore::Core::CPUState));
|
||||
|
||||
// Now restore host state
|
||||
ArchHelpers::Context::RestoreContext(ucontext, Context);
|
||||
|
||||
if (Context->UContextLocation) {
|
||||
auto Frame = ThreadState->CurrentFrame;
|
||||
auto Frame = Thread->CurrentFrame;
|
||||
|
||||
if (Context->Flags &ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_INJIT) {
|
||||
// XXX: Unsupported since it needs state reconstruction
|
||||
@@ -132,7 +134,7 @@ void Dispatcher::RestoreThreadState(void *ucontext) {
|
||||
Frame->State.rip = guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_RIP];
|
||||
// XXX: Full context setting
|
||||
uint32_t eflags = guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_EFL];
|
||||
for (size_t i = 0; i < 32; ++i) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_EFLAG_BITS; ++i) {
|
||||
Frame->State.flags[i] = (eflags & (1U << i)) ? 1 : 0;
|
||||
}
|
||||
|
||||
@@ -158,10 +160,22 @@ void Dispatcher::RestoreThreadState(void *ucontext) {
|
||||
COPY_REG(RCX);
|
||||
COPY_REG(RSP);
|
||||
#undef COPY_REG
|
||||
FEXCore::x86_64::_libc_fpstate *fpstate = reinterpret_cast<FEXCore::x86_64::_libc_fpstate*>(guest_uctx->uc_mcontext.fpregs);
|
||||
auto *xstate = reinterpret_cast<x86_64::xstate*>(guest_uctx->uc_mcontext.fpregs);
|
||||
auto *fpstate = &xstate->fpstate;
|
||||
|
||||
// Copy float registers
|
||||
memcpy(Frame->State.mm, fpstate->_st, sizeof(Frame->State.mm));
|
||||
memcpy(Frame->State.xmm, fpstate->_xmm, sizeof(Frame->State.xmm));
|
||||
|
||||
if (IsAVXEnabled) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&Frame->State.xmm.avx.data[i][0], &fpstate->_xmm[i], sizeof(__uint128_t));
|
||||
}
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&Frame->State.xmm.avx.data[i][2], &xstate->ymmh.ymmh_space[i], sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
memcpy(Frame->State.xmm.sse.data, fpstate->_xmm, sizeof(Frame->State.xmm.sse.data));
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
Frame->State.FCW = fpstate->fcw;
|
||||
@@ -190,7 +204,7 @@ void Dispatcher::RestoreThreadState(void *ucontext) {
|
||||
// XXX: Full context setting
|
||||
// First 32-bytes of flags is EFLAGS broken out
|
||||
uint32_t eflags = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EFL];
|
||||
for (size_t i = 0; i < 32; ++i) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_EFLAG_BITS; ++i) {
|
||||
Frame->State.flags[i] = (eflags & (1U << i)) ? 1 : 0;
|
||||
}
|
||||
|
||||
@@ -215,16 +229,26 @@ void Dispatcher::RestoreThreadState(void *ucontext) {
|
||||
COPY_REG(RCX);
|
||||
COPY_REG(RSP);
|
||||
#undef COPY_REG
|
||||
FEXCore::x86::_libc_fpstate *fpstate = reinterpret_cast<FEXCore::x86::_libc_fpstate*>(guest_uctx->uc_mcontext.fpregs);
|
||||
auto *xstate = reinterpret_cast<x86::xstate*>(guest_uctx->uc_mcontext.fpregs);
|
||||
auto *fpstate = &xstate->fpstate;
|
||||
|
||||
// Copy float registers
|
||||
for (size_t i = 0; i < 8; ++i) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
// 32-bit st register size is only 10 bytes. Not padded to 16byte like x86-64
|
||||
memcpy(&Frame->State.mm[i], &fpstate->_st[i], 10);
|
||||
}
|
||||
|
||||
// Extended XMM state
|
||||
memcpy(fpstate->_xmm, Frame->State.xmm, sizeof(Frame->State.xmm));
|
||||
if (IsAVXEnabled) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&fpstate->_xmm[i], &Frame->State.xmm.avx.data[i][0], sizeof(__uint128_t));
|
||||
}
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&xstate->ymmh.ymmh_space[i], &Frame->State.xmm.avx.data[i][2], sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
memcpy(Frame->State.xmm.sse.data, fpstate->_xmm, sizeof(Frame->State.xmm.sse.data));
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
Frame->State.FCW = fpstate->fcw;
|
||||
@@ -273,14 +297,34 @@ static uint32_t ConvertSignalToError(int Signal, siginfo_t *HostSigInfo) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) {
|
||||
auto ContextBackup = StoreThreadState(Signal, ucontext);
|
||||
template <typename T>
|
||||
static void SetXStateInfo(T* xstate, bool is_avx_enabled) {
|
||||
auto* fpstate = &xstate->fpstate;
|
||||
|
||||
auto Frame = ThreadState->CurrentFrame;
|
||||
fpstate->sw_reserved.magic1 = x86_64::fpx_sw_bytes::FP_XSTATE_MAGIC;
|
||||
fpstate->sw_reserved.extended_size = is_avx_enabled ? sizeof(T) : 0;
|
||||
|
||||
fpstate->sw_reserved.xfeatures |= x86_64::fpx_sw_bytes::FEATURE_FP |
|
||||
x86_64::fpx_sw_bytes::FEATURE_SSE;
|
||||
if (is_avx_enabled) {
|
||||
fpstate->sw_reserved.xfeatures |= x86_64::fpx_sw_bytes::FEATURE_YMM;
|
||||
}
|
||||
|
||||
fpstate->sw_reserved.xstate_size = fpstate->sw_reserved.extended_size;
|
||||
|
||||
if (is_avx_enabled) {
|
||||
xstate->xstate_hdr.xfeatures = 0;
|
||||
}
|
||||
}
|
||||
|
||||
bool Dispatcher::HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) {
|
||||
auto ContextBackup = StoreThreadState(Thread, Signal, ucontext);
|
||||
|
||||
auto Frame = Thread->CurrentFrame;
|
||||
|
||||
// Ref count our faults
|
||||
// We use this to track if it is safe to clear cache
|
||||
++SignalHandlerRefCounter;
|
||||
++Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
|
||||
uint64_t OldPC = ArchHelpers::Context::GetPc(ucontext);
|
||||
// Set the new PC
|
||||
@@ -292,14 +336,15 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
uint64_t NewGuestSP = OldGuestSP;
|
||||
|
||||
// Pulling from context here
|
||||
bool Is64BitMode = CTX->Config.Is64BitMode;
|
||||
uint64_t SignalReturn = CTX->X86CodeGen.SignalReturn;
|
||||
const bool Is64BitMode = CTX->Config.Is64BitMode;
|
||||
const bool IsAVXEnabled = CTX->Config.EnableAVX;
|
||||
const uint64_t SignalReturn = CTX->X86CodeGen.SignalReturn;
|
||||
|
||||
// Spill the SRA regardless of signal handler type
|
||||
// We are going to be returning to the top of the dispatcher which will fill again
|
||||
// Otherwise we might load garbage
|
||||
if (SRAEnabled) {
|
||||
if (IsAddressInJITCode(OldPC, false)) {
|
||||
if (config.StaticRegisterAllocation) {
|
||||
if (Thread->CPUBackend->IsAddressInCodeBuffer(OldPC)) {
|
||||
uint32_t IgnoreMask{};
|
||||
#ifdef _M_ARM_64
|
||||
if (Frame->InSyscallInfo != 0) {
|
||||
@@ -323,11 +368,11 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
#endif
|
||||
|
||||
// We are in jit, SRA must be spilled
|
||||
SpillSRA(ucontext, IgnoreMask);
|
||||
SpillSRA(Thread, ucontext, IgnoreMask);
|
||||
|
||||
ContextBackup->Flags |= ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_INJIT;
|
||||
} else {
|
||||
if (!IsAddressInJITCode(OldPC, true)) {
|
||||
if (!IsAddressInDispatcher(OldPC)) {
|
||||
// This is likely to cause issues but in some cases it isn't fatal
|
||||
// This can also happen if we have put a signal on hold, then we just reenabled the signal
|
||||
// So we are in the syscall handler
|
||||
@@ -373,8 +418,13 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
if (GuestAction->sa_flags & SA_SIGINFO) {
|
||||
// Setup ucontext a bit
|
||||
if (Is64BitMode) {
|
||||
NewGuestSP -= sizeof(FEXCore::x86_64::_libc_fpstate);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(FEXCore::x86_64::_libc_fpstate));
|
||||
if (IsAVXEnabled) {
|
||||
NewGuestSP -= sizeof(x86_64::xstate);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(x86_64::xstate));
|
||||
} else {
|
||||
NewGuestSP -= sizeof(x86_64::_libc_fpstate);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(x86_64::_libc_fpstate));
|
||||
}
|
||||
uint64_t FPStateLocation = NewGuestSP;
|
||||
|
||||
NewGuestSP -= sizeof(FEXCore::x86_64::ucontext_t);
|
||||
@@ -396,8 +446,9 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
guest_uctx->uc_flags = FEXCore::x86_64::UC_FP_XSTATE;
|
||||
|
||||
// Pointer to where the fpreg memory is
|
||||
guest_uctx->uc_mcontext.fpregs = reinterpret_cast<FEXCore::x86_64::_libc_fpstate*>(FPStateLocation);
|
||||
FEXCore::x86_64::_libc_fpstate *fpstate = reinterpret_cast<FEXCore::x86_64::_libc_fpstate*>(FPStateLocation);
|
||||
guest_uctx->uc_mcontext.fpregs = reinterpret_cast<x86_64::_libc_fpstate*>(FPStateLocation);
|
||||
auto *xstate = reinterpret_cast<x86_64::xstate*>(FPStateLocation);
|
||||
SetXStateInfo(xstate, IsAVXEnabled);
|
||||
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_RIP] = Frame->State.rip;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_EFL] = 0;
|
||||
@@ -409,11 +460,12 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
*guest_siginfo = *HostSigInfo;
|
||||
|
||||
if (ContextBackup->FaultToTopAndGeneratedException) {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_TRAPNO] = SynchronousFaultData.TrapNo;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_ERR] = SynchronousFaultData.err_code;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_TRAPNO] = Frame->SynchronousFaultData.TrapNo;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_ERR] = Frame->SynchronousFaultData.err_code;
|
||||
|
||||
// Overwrite si_code
|
||||
guest_siginfo->si_code = SynchronousFaultData.si_code;
|
||||
guest_siginfo->si_code = Thread->CurrentFrame->SynchronousFaultData.si_code;
|
||||
Signal = Frame->SynchronousFaultData.Signal;
|
||||
}
|
||||
else {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_TRAPNO] = ConvertSignalToTrapNo(Signal, HostSigInfo);
|
||||
@@ -442,9 +494,21 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
COPY_REG(RSP);
|
||||
#undef COPY_REG
|
||||
|
||||
auto* fpstate = &xstate->fpstate;
|
||||
|
||||
// Copy float registers
|
||||
memcpy(fpstate->_st, Frame->State.mm, sizeof(Frame->State.mm));
|
||||
memcpy(fpstate->_xmm, Frame->State.xmm, sizeof(Frame->State.xmm));
|
||||
|
||||
if (IsAVXEnabled) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&fpstate->_xmm[i], &Frame->State.xmm.avx.data[i][0], sizeof(__uint128_t));
|
||||
}
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
memcpy(&xstate->ymmh.ymmh_space[i], &Frame->State.xmm.avx.data[i][2], sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
memcpy(fpstate->_xmm, Frame->State.xmm.sse.data, sizeof(Frame->State.xmm.sse.data));
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
fpstate->fcw = Frame->State.FCW;
|
||||
@@ -469,8 +533,13 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
else {
|
||||
ContextBackup->Flags |= ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_32BIT;
|
||||
|
||||
NewGuestSP -= sizeof(FEXCore::x86::_libc_fpstate);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(FEXCore::x86::_libc_fpstate));
|
||||
if (IsAVXEnabled) {
|
||||
NewGuestSP -= sizeof(x86::xstate);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(x86::xstate));
|
||||
} else {
|
||||
NewGuestSP -= sizeof(x86::_libc_fpstate);
|
||||
NewGuestSP = AlignDown(NewGuestSP, alignof(x86::_libc_fpstate));
|
||||
}
|
||||
uint64_t FPStateLocation = NewGuestSP;
|
||||
|
||||
NewGuestSP -= sizeof(FEXCore::x86::ucontext_t);
|
||||
@@ -493,21 +562,23 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
|
||||
// Pointer to where the fpreg memory is
|
||||
guest_uctx->uc_mcontext.fpregs = static_cast<uint32_t>(FPStateLocation);
|
||||
FEXCore::x86::_libc_fpstate *fpstate = reinterpret_cast<FEXCore::x86::_libc_fpstate*>(FPStateLocation);
|
||||
auto *xstate = reinterpret_cast<x86::xstate*>(FPStateLocation);
|
||||
SetXStateInfo(xstate, IsAVXEnabled);
|
||||
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS] = Frame->State.gs;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS] = Frame->State.fs;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES] = Frame->State.es;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS] = Frame->State.ds;
|
||||
if (ContextBackup->FaultToTopAndGeneratedException) {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = SynchronousFaultData.TrapNo;
|
||||
guest_siginfo->si_code = SynchronousFaultData.si_code;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = SynchronousFaultData.err_code;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = Frame->SynchronousFaultData.TrapNo;
|
||||
guest_siginfo->si_code = Frame->SynchronousFaultData.si_code;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = Frame->SynchronousFaultData.err_code;
|
||||
Signal = Frame->SynchronousFaultData.Signal;
|
||||
}
|
||||
else {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = ConvertSignalToTrapNo(Signal, HostSigInfo);
|
||||
guest_siginfo->si_code = HostSigInfo->si_code;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = ConvertSignalToError(Signal, HostSigInfo);
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = ConvertSignalToError(Signal, HostSigInfo);
|
||||
}
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP] = Frame->State.rip;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS] = Frame->State.cs;
|
||||
@@ -527,15 +598,26 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
COPY_REG(RSP);
|
||||
#undef COPY_REG
|
||||
|
||||
auto *fpstate = &xstate->fpstate;
|
||||
|
||||
// Copy float registers
|
||||
for (size_t i = 0; i < 8; ++i) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
// 32-bit st register size is only 10 bytes. Not padded to 16byte like x86-64
|
||||
memcpy(&fpstate->_st[i], &Frame->State.mm[i], 10);
|
||||
}
|
||||
|
||||
// Extended XMM state
|
||||
fpstate->status = FEXCore::x86::fpstate_magic::MAGIC_XFPSTATE;
|
||||
memcpy(fpstate->_xmm, Frame->State.xmm, sizeof(Frame->State.xmm));
|
||||
if (IsAVXEnabled) {
|
||||
for (size_t i = 0; i < std::size(Frame->State.xmm.avx.data); i++) {
|
||||
memcpy(&fpstate->_xmm[i], &Frame->State.xmm.avx.data[i][0], sizeof(__uint128_t));
|
||||
}
|
||||
for (size_t i = 0; i < std::size(Frame->State.xmm.avx.data); i++) {
|
||||
memcpy(&xstate->ymmh.ymmh_space[i], &Frame->State.xmm.avx.data[i][2], sizeof(__uint128_t));
|
||||
}
|
||||
} else {
|
||||
memcpy(fpstate->_xmm, Frame->State.xmm.sse.data, sizeof(Frame->State.xmm.sse.data));
|
||||
}
|
||||
|
||||
// FCW store default
|
||||
fpstate->fcw = Frame->State.FCW;
|
||||
@@ -618,14 +700,14 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
else {
|
||||
NewGuestSP -= 4;
|
||||
*(uint32_t*)NewGuestSP = SignalReturn;
|
||||
LOGMAN_THROW_A_FMT(SignalReturn < 0x1'0000'0000ULL, "This needs to be below 4GB");
|
||||
LOGMAN_THROW_AA_FMT(SignalReturn < 0x1'0000'0000ULL, "This needs to be below 4GB");
|
||||
Frame->State.gregs[FEXCore::X86State::REG_RSP] = NewGuestSP;
|
||||
}
|
||||
|
||||
// The guest starts its signal frame with a zero initialized FPU
|
||||
// Set that up now. Little bit costly but it's a requirement
|
||||
// This state will be restored on rt_sigreturn
|
||||
memset(Frame->State.xmm, 0, sizeof(Frame->State.xmm));
|
||||
memset(Frame->State.xmm.avx.data, 0, sizeof(Frame->State.xmm));
|
||||
memset(Frame->State.mm, 0, sizeof(Frame->State.mm));
|
||||
Frame->State.FCW = 0x37F;
|
||||
Frame->State.FTW = 0xFFFF;
|
||||
@@ -633,44 +715,44 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
return true;
|
||||
}
|
||||
|
||||
bool Dispatcher::HandleSIGILL(int Signal, void *info, void *ucontext) {
|
||||
bool Dispatcher::HandleSIGILL(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) {
|
||||
|
||||
if (ArchHelpers::Context::GetPc(ucontext) == SignalHandlerReturnAddress) {
|
||||
RestoreThreadState(ucontext);
|
||||
RestoreThreadState(Thread, ucontext);
|
||||
|
||||
// Ref count our faults
|
||||
// We use this to track if it is safe to clear cache
|
||||
--SignalHandlerRefCounter;
|
||||
--Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
return true;
|
||||
}
|
||||
|
||||
if (ArchHelpers::Context::GetPc(ucontext) == PauseReturnInstruction) {
|
||||
RestoreThreadState(ucontext);
|
||||
RestoreThreadState(Thread, ucontext);
|
||||
|
||||
// Ref count our faults
|
||||
// We use this to track if it is safe to clear cache
|
||||
--SignalHandlerRefCounter;
|
||||
--Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
|
||||
FEXCore::Core::SignalEvent SignalReason = ThreadState->SignalReason.load();
|
||||
auto Frame = ThreadState->CurrentFrame;
|
||||
bool Dispatcher::HandleSignalPause(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) {
|
||||
FEXCore::Core::SignalEvent SignalReason = Thread->SignalReason.load();
|
||||
auto Frame = Thread->CurrentFrame;
|
||||
|
||||
if (SignalReason == FEXCore::Core::SignalEvent::Pause) {
|
||||
// Store our thread state so we can come back to this
|
||||
StoreThreadState(Signal, ucontext);
|
||||
StoreThreadState(Thread, Signal, ucontext);
|
||||
|
||||
if (SRAEnabled && IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), false)) {
|
||||
if (config.StaticRegisterAllocation && Thread->CPUBackend->IsAddressInCodeBuffer(ArchHelpers::Context::GetPc(ucontext))) {
|
||||
// We are in jit, SRA must be spilled
|
||||
ArchHelpers::Context::SetPc(ucontext, ThreadPauseHandlerAddressSpillSRA);
|
||||
} else {
|
||||
if (SRAEnabled) {
|
||||
if (config.StaticRegisterAllocation) {
|
||||
// We are in non-jit, SRA is already spilled
|
||||
LOGMAN_THROW_A_FMT(!IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), true),
|
||||
LOGMAN_THROW_A_FMT(!IsAddressInDispatcher(ArchHelpers::Context::GetPc(ucontext)),
|
||||
"Signals in dispatcher have unsynchronized context");
|
||||
}
|
||||
ArchHelpers::Context::SetPc(ucontext, ThreadPauseHandlerAddress);
|
||||
@@ -681,9 +763,9 @@ bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
|
||||
|
||||
// Ref count our faults
|
||||
// We use this to track if it is safe to clear cache
|
||||
++SignalHandlerRefCounter;
|
||||
++Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
|
||||
ThreadState->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -694,16 +776,16 @@ bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
|
||||
ArchHelpers::Context::SetSp(ucontext, Frame->ReturningStackLocation);
|
||||
|
||||
// Our ref counting doesn't matter anymore
|
||||
SignalHandlerRefCounter = 0;
|
||||
Thread->CurrentFrame->SignalHandlerRefCounter = 0;
|
||||
|
||||
// Set the new PC
|
||||
if (SRAEnabled && IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), false)) {
|
||||
if (config.StaticRegisterAllocation && Thread->CPUBackend->IsAddressInCodeBuffer(ArchHelpers::Context::GetPc(ucontext))) {
|
||||
// We are in jit, SRA must be spilled
|
||||
ArchHelpers::Context::SetPc(ucontext, ThreadStopHandlerAddressSpillSRA);
|
||||
} else {
|
||||
if (SRAEnabled) {
|
||||
if (config.StaticRegisterAllocation) {
|
||||
// We are in non-jit, SRA is already spilled
|
||||
LOGMAN_THROW_A_FMT(!IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), true),
|
||||
LOGMAN_THROW_A_FMT(!IsAddressInDispatcher(ArchHelpers::Context::GetPc(ucontext)),
|
||||
"Signals in dispatcher have unsynchronized context");
|
||||
}
|
||||
ArchHelpers::Context::SetPc(ucontext, ThreadStopHandlerAddress);
|
||||
@@ -712,24 +794,24 @@ bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
|
||||
// We need to be a little bit careful here
|
||||
// If we were already paused (due to GDB) and we are immediately stopping (due to gdb kill)
|
||||
// Then we need to ensure we don't double decrement our idle thread counter
|
||||
if (ThreadState->RunningEvents.ThreadSleeping) {
|
||||
if (Thread->RunningEvents.ThreadSleeping) {
|
||||
// If the thread was sleeping then its idle counter was decremented
|
||||
// Reincrement it here to not break logic
|
||||
++ThreadState->CTX->IdleWaitRefCount;
|
||||
++Thread->CTX->IdleWaitRefCount;
|
||||
}
|
||||
|
||||
ThreadState->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
return true;
|
||||
}
|
||||
|
||||
if (SignalReason == FEXCore::Core::SignalEvent::Return) {
|
||||
RestoreThreadState(ucontext);
|
||||
RestoreThreadState(Thread, ucontext);
|
||||
|
||||
// Ref count our faults
|
||||
// We use this to track if it is safe to clear cache
|
||||
--SignalHandlerRefCounter;
|
||||
--Thread->CurrentFrame->SignalHandlerRefCounter;
|
||||
|
||||
ThreadState->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -748,28 +830,4 @@ uint64_t Dispatcher::GetCompileBlockPtr() {
|
||||
return CompileBlockPtr.Data;
|
||||
}
|
||||
|
||||
void Dispatcher::RemoveCodeBuffer(uint8_t* start_to_remove) {
|
||||
for (auto iter = CodeBuffers.begin(); iter != CodeBuffers.end(); ++iter) {
|
||||
auto [start, end] = *iter;
|
||||
if (start == reinterpret_cast<uint64_t>(start_to_remove)) {
|
||||
CodeBuffers.erase(iter);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
bool Dispatcher::IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher) const {
|
||||
for (auto [start, end] : CodeBuffers) {
|
||||
if (Address >= start && Address < end) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
if (IncludeDispatcher && IsAddressInDispatcher(Address)) {
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
}
|
||||
+47
-42
@@ -1,8 +1,6 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers/MContext.h"
|
||||
|
||||
#include <cstdint>
|
||||
@@ -21,22 +19,20 @@ struct CpuStateFrame;
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::Context {
|
||||
struct Context;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
struct DispatcherConfig {
|
||||
bool ExecuteBlocksWithCall = false;
|
||||
uintptr_t ExitFunctionLink = 0;
|
||||
uintptr_t ExitFunctionLinkThis = 0;
|
||||
bool StaticRegisterAssignment = false;
|
||||
bool StaticRegisterAllocation = false;
|
||||
};
|
||||
|
||||
class Dispatcher {
|
||||
public:
|
||||
virtual ~Dispatcher() = default;
|
||||
CPUBackend::AsmDispatch DispatchPtr;
|
||||
CPUBackend::JITCallback CallbackPtr;
|
||||
FEXCore::Context::Context::IntCallbackReturn ReturnPtr;
|
||||
|
||||
|
||||
/**
|
||||
* @name Dispatch Helper functions
|
||||
* @{ */
|
||||
@@ -48,61 +44,70 @@ public:
|
||||
uint64_t ThreadPauseHandlerAddressSpillSRA{};
|
||||
uint64_t ExitFunctionLinkerAddress{};
|
||||
uint64_t SignalHandlerReturnAddress{};
|
||||
uint64_t UnimplementedInstructionAddress{};
|
||||
uint64_t OverflowExceptionInstructionAddress{};
|
||||
uint64_t GuestSignal_SIGILL{};
|
||||
uint64_t GuestSignal_SIGTRAP{};
|
||||
uint64_t GuestSignal_SIGSEGV{};
|
||||
uint64_t IntCallbackReturnAddress{};
|
||||
|
||||
uint64_t PauseReturnInstruction{};
|
||||
|
||||
/** @} */
|
||||
|
||||
uint32_t SignalHandlerRefCounter{};
|
||||
struct SynchronousFaultDataStruct {
|
||||
bool FaultToTopAndGeneratedException{};
|
||||
uint32_t TrapNo;
|
||||
uint32_t err_code;
|
||||
uint32_t si_code;
|
||||
} SynchronousFaultData;
|
||||
|
||||
uint64_t Start{};
|
||||
uint64_t End{};
|
||||
|
||||
bool HandleGuestSignal(int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack);
|
||||
bool HandleSIGILL(int Signal, void *info, void *ucontext);
|
||||
bool HandleSignalPause(int Signal, void *info, void *ucontext);
|
||||
bool HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack);
|
||||
bool HandleSIGILL(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext);
|
||||
bool HandleSignalPause(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext);
|
||||
|
||||
void RegisterCodeBuffer(uint8_t* start, size_t size) {
|
||||
CodeBuffers.emplace_back(reinterpret_cast<uint64_t>(start),
|
||||
reinterpret_cast<uint64_t>(start + size));
|
||||
}
|
||||
|
||||
void RemoveCodeBuffer(uint8_t* start);
|
||||
|
||||
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true) const;
|
||||
bool IsAddressInDispatcher(uint64_t Address) const {
|
||||
return Address >= Start && Address < End;
|
||||
}
|
||||
|
||||
protected:
|
||||
Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
: CTX {ctx}
|
||||
, ThreadState {Thread} {}
|
||||
virtual void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
|
||||
ArchHelpers::Context::ContextBackup* StoreThreadState(int Signal, void *ucontext);
|
||||
void RestoreThreadState(void *ucontext);
|
||||
// These are across all arches for now
|
||||
static constexpr size_t MaxGDBPauseCheckSize = 128;
|
||||
static constexpr size_t MaxInterpreterTrampolineSize = 128;
|
||||
|
||||
virtual size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) = 0;
|
||||
virtual size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) = 0;
|
||||
|
||||
static std::unique_ptr<Dispatcher> CreateX86(FEXCore::Context::Context *CTX, const DispatcherConfig &Config);
|
||||
static std::unique_ptr<Dispatcher> CreateArm64(FEXCore::Context::Context *CTX, const DispatcherConfig &Config);
|
||||
|
||||
void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) {
|
||||
DispatchPtr(Frame);
|
||||
}
|
||||
|
||||
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) {
|
||||
CallbackPtr(Frame, RIP);
|
||||
}
|
||||
|
||||
protected:
|
||||
Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &Config)
|
||||
: CTX {ctx}
|
||||
, config {Config}
|
||||
{}
|
||||
|
||||
ArchHelpers::Context::ContextBackup* StoreThreadState(FEXCore::Core::InternalThreadState *Thread, int Signal, void *ucontext);
|
||||
void RestoreThreadState(FEXCore::Core::InternalThreadState *Thread, void *ucontext);
|
||||
std::stack<uint64_t, std::vector<uint64_t>> SignalFrames;
|
||||
|
||||
bool SRAEnabled = false;
|
||||
virtual void SpillSRA(void *ucontext, uint32_t IgnoreMask) {}
|
||||
virtual void SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) {}
|
||||
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::Core::InternalThreadState *ThreadState;
|
||||
DispatcherConfig config;
|
||||
|
||||
static void SleepThread(FEXCore::Context::Context *ctx, FEXCore::Core::CpuStateFrame *Frame);
|
||||
|
||||
static uint64_t GetCompileBlockPtr();
|
||||
|
||||
private:
|
||||
std::vector<std::tuple<uint64_t, uint64_t>> CodeBuffers; // Start, End
|
||||
using AsmDispatch = void(*)(FEXCore::Core::CpuStateFrame *Frame);
|
||||
using JITCallback = void(*)(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP);
|
||||
|
||||
AsmDispatch DispatchPtr;
|
||||
JITCallback CallbackPtr;
|
||||
};
|
||||
|
||||
}
|
||||
+115
-84
@@ -18,21 +18,26 @@
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <sys/mman.h>
|
||||
#include "xbyak/xbyak.h"
|
||||
#include <xbyak/xbyak.h>
|
||||
|
||||
#define STATE_PTR(STATE_TYPE, FIELD) \
|
||||
[STATE + offsetof(FEXCore::Core::STATE_TYPE, FIELD)]
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
|
||||
#define STATE r14
|
||||
|
||||
X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config)
|
||||
: Dispatcher(ctx, Thread)
|
||||
X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config)
|
||||
: Dispatcher(ctx, config)
|
||||
, Xbyak::CodeGenerator(MAX_DISPATCHER_CODE_SIZE,
|
||||
FEXCore::Allocator::mmap(nullptr, MAX_DISPATCHER_CODE_SIZE, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0),
|
||||
nullptr) {
|
||||
|
||||
LOGMAN_THROW_AA_FMT(!config.StaticRegisterAllocation, "X86 dispatcher does not support SRA");
|
||||
|
||||
using namespace Xbyak;
|
||||
using namespace Xbyak::util;
|
||||
DispatchPtr = getCurr<CPUBackend::AsmDispatch>();
|
||||
DispatchPtr = getCurr<AsmDispatch>();
|
||||
|
||||
// Temp registers
|
||||
// rax, rcx, rdx, rsi, r8, r9,
|
||||
@@ -78,11 +83,10 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
|
||||
// Save this stack pointer so we can cleanly shutdown the emulation with a long jump
|
||||
// regardless of where we were in the stack
|
||||
mov(qword [rdi + offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)], rsp);
|
||||
mov(qword STATE_PTR(CpuStateFrame, ReturningStackLocation), rsp);
|
||||
|
||||
Label LoopTop;
|
||||
Label FullLookup;
|
||||
Label CallBlock;
|
||||
Label NoBlock;
|
||||
Label ExitBlock;
|
||||
Label ThreadPauseHandler;
|
||||
@@ -92,29 +96,24 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
|
||||
{
|
||||
// Load our RIP
|
||||
mov(rdx, qword [STATE + offsetof(FEXCore::Core::CPUState, rip)]);
|
||||
mov(rdx, qword STATE_PTR(CPUState, rip));
|
||||
|
||||
// L1 Cache
|
||||
mov(r13, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.L1Pointer)]);
|
||||
mov(r13, qword STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
mov(rax, rdx);
|
||||
|
||||
and_(rax, LookupCache::L1_ENTRIES_MASK);
|
||||
shl(rax, 4);
|
||||
cmp(qword[r13 + rax + 8], rdx);
|
||||
cmp(qword[r13 + rax + offsetof(FEXCore::LookupCache::LookupCacheEntry, GuestCode)], rdx);
|
||||
jne(FullLookup);
|
||||
|
||||
if (!config.ExecuteBlocksWithCall) {
|
||||
jmp(qword[r13 + rax + 0]);
|
||||
} else {
|
||||
mov(rax, qword[r13 + rax + 0]);
|
||||
jmp(CallBlock);
|
||||
}
|
||||
jmp(qword[r13 + rax + offsetof(FEXCore::LookupCache::LookupCacheEntry, HostCode)]);
|
||||
|
||||
L(FullLookup);
|
||||
mov(r13, Thread->LookupCache->GetPagePointer());
|
||||
mov(r13, qword STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
|
||||
|
||||
// Full lookup
|
||||
uint64_t VirtualMemorySize = Thread->LookupCache->GetVirtualMemorySize();
|
||||
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
|
||||
mov(rax, rdx);
|
||||
mov(rbx, VirtualMemorySize - 1);
|
||||
and_(rax, rbx);
|
||||
@@ -143,7 +142,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
je(NoBlock);
|
||||
|
||||
// Update L1
|
||||
mov(r13, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.L1Pointer)]);
|
||||
mov(r13, qword STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
mov(rcx, rdx);
|
||||
and_(rcx, LookupCache::L1_ENTRIES_MASK);
|
||||
shl(rcx, 1);
|
||||
@@ -151,30 +150,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
mov(qword[r13 + rcx*8 + 0], rax);
|
||||
|
||||
// Real block if we made it here
|
||||
if (!config.ExecuteBlocksWithCall) {
|
||||
jmp(rax);
|
||||
} else {
|
||||
L(CallBlock);
|
||||
mov(rdi, STATE);
|
||||
call(rax);
|
||||
|
||||
if (CTX->GetGdbServerStatus()) {
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
// This happens when single stepping
|
||||
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
mov(rax, qword [STATE + (offsetof(FEXCore::Core::InternalThreadState, CTX))]);
|
||||
|
||||
// If the value == 0 then branch to the top
|
||||
cmp(dword [rax + (offsetof(FEXCore::Context::Context, Config.RunningMode))], 0);
|
||||
je(LoopTop);
|
||||
// Else we need to pause now
|
||||
jmp(ThreadPauseHandler);
|
||||
ud2();
|
||||
}
|
||||
else {
|
||||
jmp(LoopTop);
|
||||
}
|
||||
}
|
||||
jmp(rax);
|
||||
}
|
||||
|
||||
{
|
||||
@@ -282,13 +258,11 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
mov(rax, r9);
|
||||
}
|
||||
|
||||
// {rdi, rsi, rdx}
|
||||
mov(rdi, config.ExitFunctionLinkThis);
|
||||
mov(rsi, STATE);
|
||||
mov(rdx, rax); // rax is set at the block end
|
||||
// {rdi, rsi}
|
||||
mov(rdi, STATE);
|
||||
mov(rsi, rax); // rax is set at the block end
|
||||
|
||||
mov(rax, config.ExitFunctionLink);
|
||||
call(rax);
|
||||
call(qword STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
@@ -330,7 +304,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
}
|
||||
|
||||
{
|
||||
CallbackPtr = getCurr<CPUBackend::JITCallback>();
|
||||
CallbackPtr = getCurr<JITCallback>();
|
||||
|
||||
push(rbx);
|
||||
push(rbp);
|
||||
@@ -345,7 +319,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
// XXX: XMM?
|
||||
|
||||
// Make sure to adjust the refcounter so we don't clear the cache now
|
||||
add(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.SignalHandlerRefCountPointer)], 1);
|
||||
add(qword STATE_PTR(CpuStateFrame, SignalHandlerRefCounter), 1);
|
||||
|
||||
// Now push the callback return trampoline to the guest stack
|
||||
// Guest will be misaligned because calling a thunk won't correct the guest's stack once we call the callback from the host
|
||||
@@ -353,12 +327,12 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
|
||||
// Store the trampoline to the guest stack
|
||||
// Guest stack is now correctly misaligned after a regular call instruction
|
||||
sub(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])], 16);
|
||||
mov(rbx, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])]);
|
||||
sub(qword STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]), 16);
|
||||
mov(rbx, qword STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
mov(qword [rbx], rax);
|
||||
|
||||
// Store RIP to the context state
|
||||
mov(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.rip)], rsi);
|
||||
mov(qword STATE_PTR(CpuStateFrame, State.rip), rsi);
|
||||
|
||||
// Back to the loop top now
|
||||
jmp(LoopTop);
|
||||
@@ -373,30 +347,34 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
{
|
||||
// Guest SIGILL handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
UnimplementedInstructionAddress = getCurr<uint64_t>();
|
||||
GuestSignal_SIGILL = getCurr<uint64_t>();
|
||||
ud2();
|
||||
}
|
||||
|
||||
{
|
||||
// Guest Overflow handler
|
||||
// Guest SIGTRAP handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
OverflowExceptionInstructionAddress = getCurr<uint64_t>();
|
||||
GuestSignal_SIGTRAP = getCurr<uint64_t>();
|
||||
|
||||
// ud2 = SIGILL
|
||||
// int3 = SIGTRAP
|
||||
// hlt = SIGSEGV
|
||||
int3();
|
||||
}
|
||||
|
||||
mov(rax, reinterpret_cast<uint64_t>(&SynchronousFaultData));
|
||||
add(byte [rax + offsetof(Dispatcher::SynchronousFaultDataStruct, FaultToTopAndGeneratedException)], 1);
|
||||
mov(dword [rax + offsetof(Dispatcher::SynchronousFaultDataStruct, TrapNo)], X86State::X86_TRAPNO_OF);
|
||||
mov(dword [rax + offsetof(Dispatcher::SynchronousFaultDataStruct, err_code)], 0);
|
||||
mov(dword [rax + offsetof(Dispatcher::SynchronousFaultDataStruct, si_code)], 0x80);
|
||||
{
|
||||
// Guest SIGSEGV handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
GuestSignal_SIGSEGV = getCurr<uint64_t>();
|
||||
|
||||
// ud2 = SIGILL
|
||||
// int3 = SIGTRAP
|
||||
// hlt = SIGSEGV
|
||||
hlt();
|
||||
}
|
||||
|
||||
{
|
||||
ReturnPtr = getCurr<FEXCore::Context::Context::IntCallbackReturn>();
|
||||
IntCallbackReturnAddress = getCurr<uint64_t>();
|
||||
// using CallbackReturn = FEX_NAKED void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
|
||||
|
||||
// rdi = thread
|
||||
@@ -430,39 +408,92 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
CTX->Symbols.RegisterJITSpace(reinterpret_cast<void*>(Start), End-Start);
|
||||
}
|
||||
|
||||
// Setup dispatcher specific pointers that need to be accessed from JIT code
|
||||
{
|
||||
auto &Pointers = ThreadState->CurrentFrame->Pointers.X86;
|
||||
}
|
||||
|
||||
Pointers.DispatcherLoopTop = AbsoluteLoopTopAddress;
|
||||
Pointers.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
|
||||
Pointers.ThreadStopHandler = ThreadStopHandlerAddress;
|
||||
Pointers.ThreadPauseHandler = ThreadPauseHandlerAddress;
|
||||
Pointers.UnimplementedInstructionHandler = UnimplementedInstructionAddress;
|
||||
Pointers.OverflowExceptionHandler = OverflowExceptionInstructionAddress;
|
||||
Pointers.SignalReturnHandler = SignalHandlerReturnAddress;
|
||||
Pointers.L1Pointer = Thread->LookupCache->GetL1Pointer();
|
||||
// Used by GenerateGDBPauseCheck, GenerateInterpreterTrampoline
|
||||
static thread_local Xbyak::CodeGenerator emit(1, &emit); // actual emit target set with setNewBuffer
|
||||
|
||||
size_t X86Dispatcher::GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) {
|
||||
using namespace Xbyak;
|
||||
using namespace Xbyak::util;
|
||||
|
||||
emit.setNewBuffer(CodeBuffer, MaxGDBPauseCheckSize);
|
||||
|
||||
Label RunBlock;
|
||||
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
// This happens when single stepping
|
||||
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
emit.mov(rax, reinterpret_cast<uint64_t>(CTX));
|
||||
|
||||
// If the value == 0 then we don't need to stop
|
||||
emit.cmp(dword [rax + (offsetof(FEXCore::Context::Context, Config.RunningMode))], 0);
|
||||
emit.je(RunBlock);
|
||||
{
|
||||
// Make sure RIP is syncronized to the context
|
||||
emit.mov(rax, GuestRIP);
|
||||
emit.mov(qword STATE_PTR(CpuStateFrame, State.rip), rax);
|
||||
|
||||
// Stop the thread
|
||||
emit.mov(rax, qword STATE_PTR(CpuStateFrame, Pointers.Common.ThreadPauseHandlerSpillSRA));
|
||||
emit.jmp(rax);
|
||||
}
|
||||
|
||||
emit.L(RunBlock);
|
||||
|
||||
emit.ready();
|
||||
|
||||
return emit.getSize();
|
||||
}
|
||||
|
||||
|
||||
size_t X86Dispatcher::GenerateInterpreterTrampoline(uint8_t *CodeBuffer) {
|
||||
using namespace Xbyak;
|
||||
using namespace Xbyak::util;
|
||||
|
||||
emit.setNewBuffer(CodeBuffer, MaxInterpreterTrampolineSize);
|
||||
|
||||
Label InlineIRData;
|
||||
|
||||
emit.mov(rdi, STATE);
|
||||
emit.lea(rsi, ptr[rip + InlineIRData]);
|
||||
emit.call(qword STATE_PTR(CpuStateFrame, Pointers.Interpreter.FragmentExecuter));
|
||||
|
||||
emit.jmp(qword STATE_PTR(CpuStateFrame, Pointers.Common.DispatcherLoopTop));
|
||||
|
||||
emit.L(InlineIRData);
|
||||
|
||||
emit.ready();
|
||||
|
||||
return emit.getSize();
|
||||
}
|
||||
|
||||
X86Dispatcher::~X86Dispatcher() {
|
||||
FEXCore::Allocator::munmap(top_, MAX_DISPATCHER_CODE_SIZE);
|
||||
}
|
||||
|
||||
#ifdef _M_X86_64
|
||||
void X86Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// Setup dispatcher specific pointers that need to be accessed from JIT code
|
||||
{
|
||||
auto &Common = Thread->CurrentFrame->Pointers.Common;
|
||||
|
||||
void InterpreterCore::CreateAsmDispatch(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
DispatcherConfig config;
|
||||
config.ExecuteBlocksWithCall = true;
|
||||
Common.DispatcherLoopTop = AbsoluteLoopTopAddress;
|
||||
Common.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
|
||||
Common.ExitFunctionLinker = ExitFunctionLinkerAddress;
|
||||
Common.ThreadStopHandlerSpillSRA = ThreadStopHandlerAddress;
|
||||
Common.ThreadPauseHandlerSpillSRA = ThreadPauseHandlerAddress;
|
||||
Common.GuestSignal_SIGILL = GuestSignal_SIGILL;
|
||||
Common.GuestSignal_SIGTRAP = GuestSignal_SIGTRAP;
|
||||
Common.GuestSignal_SIGSEGV = GuestSignal_SIGSEGV;
|
||||
Common.SignalReturnHandler = SignalHandlerReturnAddress;
|
||||
|
||||
Dispatcher = std::make_unique<X86Dispatcher>(ctx, Thread, config);
|
||||
DispatchPtr = Dispatcher->DispatchPtr;
|
||||
CallbackPtr = Dispatcher->CallbackPtr;
|
||||
|
||||
// TODO: It feels wrong to initialize this way
|
||||
ctx->InterpreterCallbackReturn = Dispatcher->ReturnPtr;
|
||||
auto &Interpreter = Thread->CurrentFrame->Pointers.Interpreter;
|
||||
(uintptr_t&)Interpreter.CallbackReturn = IntCallbackReturnAddress;
|
||||
}
|
||||
}
|
||||
|
||||
#endif
|
||||
std::unique_ptr<Dispatcher> Dispatcher::CreateX86(FEXCore::Context::Context *CTX, const DispatcherConfig &Config) {
|
||||
return std::make_unique<X86Dispatcher>(CTX, Config);
|
||||
}
|
||||
|
||||
}
|
||||
@@ -17,7 +17,10 @@ namespace FEXCore::CPU {
|
||||
|
||||
class X86Dispatcher final : public Dispatcher, public Xbyak::CodeGenerator {
|
||||
public:
|
||||
X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config);
|
||||
X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config);
|
||||
void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) override;
|
||||
size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) override;
|
||||
|
||||
virtual ~X86Dispatcher() override;
|
||||
};
|
||||
|
||||
+48
-22
@@ -19,6 +19,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
#include <FEXHeaderUtils/TypeDefines.h>
|
||||
#include <set>
|
||||
#include <sys/mman.h>
|
||||
|
||||
@@ -187,7 +188,7 @@ Decoder::~Decoder() {
|
||||
|
||||
uint8_t Decoder::ReadByte() {
|
||||
uint8_t Byte = InstStream[InstructionSize];
|
||||
LOGMAN_THROW_A_FMT(InstructionSize < MAX_INST_SIZE, "Max instruction size exceeded!");
|
||||
LOGMAN_THROW_AA_FMT(InstructionSize < MAX_INST_SIZE, "Max instruction size exceeded!");
|
||||
Instruction[InstructionSize] = Byte;
|
||||
InstructionSize++;
|
||||
return Byte;
|
||||
@@ -199,14 +200,7 @@ uint8_t Decoder::PeekByte(uint8_t Offset) const {
|
||||
}
|
||||
|
||||
uint64_t Decoder::ReadData(uint8_t Size) {
|
||||
if (Size == 0) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
if (Size > sizeof(uint64_t)) {
|
||||
LOGMAN_MSG_A_FMT("Unknown data size to read");
|
||||
return 0;
|
||||
}
|
||||
LOGMAN_THROW_AA_FMT(Size != 0 && Size <= sizeof(uint64_t), "Unknown data size to read");
|
||||
|
||||
uint64_t Res = 0;
|
||||
std::memcpy(&Res, &InstStream[InstructionSize], Size);
|
||||
@@ -346,13 +340,15 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModR
|
||||
Operand->Data.SIB.Index = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_X ? 1 : 0, SIB.index, false, false, false, false, 0b100);
|
||||
Operand->Data.SIB.Base = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, SIB.base, false, false, false, false, ModRM.mod == 0 ? 0b101 : 16);
|
||||
|
||||
LOGMAN_THROW_A_FMT(Displacement <= 4, "Number of bytes should be <= 4 for literal src");
|
||||
LOGMAN_THROW_AA_FMT(Displacement <= 4, "Number of bytes should be <= 4 for literal src");
|
||||
|
||||
uint64_t Literal = ReadData(Displacement);
|
||||
if (Displacement == 1) {
|
||||
Literal = static_cast<int8_t>(Literal);
|
||||
if (Displacement) {
|
||||
uint64_t Literal = ReadData(Displacement);
|
||||
if (Displacement == 1) {
|
||||
Literal = static_cast<int8_t>(Literal);
|
||||
}
|
||||
Operand->Data.SIB.Offset = Literal;
|
||||
}
|
||||
Operand->Data.SIB.Offset = Literal;
|
||||
}
|
||||
else if (ModRM.mod == 0) {
|
||||
// Explained in Table 1-14. "Operand Addressing Using ModRM and SIB Bytes"
|
||||
@@ -403,7 +399,7 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
return false;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(!(Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_P),
|
||||
LOGMAN_THROW_AA_FMT(!(Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_P),
|
||||
"Group Ops should have been decoded before this!");
|
||||
|
||||
uint8_t DestSize{};
|
||||
@@ -527,7 +523,7 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
}
|
||||
|
||||
if (HAS_NON_XMM_SUBFLAG(Info->Flags, FEXCore::X86Tables::InstFlags::FLAGS_SF_REX_IN_BYTE)) {
|
||||
LOGMAN_THROW_A_FMT(!HasMODRM, "This instruction shouldn't have ModRM!");
|
||||
LOGMAN_THROW_AA_FMT(!HasMODRM, "This instruction shouldn't have ModRM!");
|
||||
|
||||
// If the REX is in the byte that means the lower nibble of the OP contains the destination GPR
|
||||
// This also means that the destination is always a GPR on these ones
|
||||
@@ -636,7 +632,7 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
}
|
||||
|
||||
if (Bytes != 0) {
|
||||
LOGMAN_THROW_A_FMT(Bytes <= 8, "Number of bytes should be <= 8 for literal src");
|
||||
LOGMAN_THROW_AA_FMT(Bytes <= 8, "Number of bytes should be <= 8 for literal src");
|
||||
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Size = Bytes;
|
||||
|
||||
@@ -661,7 +657,7 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Bytes == 0, "Inst at 0x{:x}: 0x{:04x} '{}' Had an instruction of size {} with {} remaining",
|
||||
LOGMAN_THROW_AA_FMT(Bytes == 0, "Inst at 0x{:x}: 0x{:04x} '{}' Had an instruction of size {} with {} remaining",
|
||||
DecodeInst->PC, DecodeInst->OP, DecodeInst->TableInfo->Name ?: "UND", InstructionSize, Bytes);
|
||||
DecodeInst->InstSize = InstructionSize;
|
||||
return true;
|
||||
@@ -687,7 +683,7 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
return false;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Info->Type != FEXCore::X86Tables::TYPE_REX_PREFIX,
|
||||
LOGMAN_THROW_AA_FMT(Info->Type != FEXCore::X86Tables::TYPE_REX_PREFIX,
|
||||
"REX PREFIX should have been decoded before this!");
|
||||
|
||||
if (Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 &&
|
||||
@@ -744,7 +740,7 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
3,
|
||||
};
|
||||
uint8_t Field = RegToField[ModRM.reg];
|
||||
LOGMAN_THROW_A_FMT(Field != 255, "Invalid field selected!");
|
||||
LOGMAN_THROW_AA_FMT(Field != 255, "Invalid field selected!");
|
||||
|
||||
LocalOp = (Field << 3) | ModRM.rm;
|
||||
return NormalOp(&SecondModRMTableOps[LocalOp], LocalOp);
|
||||
@@ -1135,7 +1131,7 @@ const uint8_t *Decoder::AdjustAddrForSpecialRegion(uint8_t const* _InstStream, u
|
||||
return _InstStream - EntryPoint + RIP;
|
||||
}
|
||||
|
||||
void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC) {
|
||||
void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage) {
|
||||
Blocks.clear();
|
||||
BlocksToDecode.clear();
|
||||
HasBlocks.clear();
|
||||
@@ -1165,6 +1161,13 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC)
|
||||
// Entry is a jump target
|
||||
BlocksToDecode.emplace(PC);
|
||||
|
||||
uint64_t CurrentCodePage = PC & FHU::FEX_PAGE_MASK;
|
||||
|
||||
std::set<uint64_t> CodePages = { CurrentCodePage };
|
||||
|
||||
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
|
||||
|
||||
|
||||
while (!BlocksToDecode.empty()) {
|
||||
auto BlockDecodeIt = BlocksToDecode.begin();
|
||||
uint64_t RIPToDecode = *BlockDecodeIt;
|
||||
@@ -1181,10 +1184,33 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC)
|
||||
InstStream = AdjustAddrForSpecialRegion(_InstStream, EntryPoint, RIPToDecode);
|
||||
|
||||
while (1) {
|
||||
|
||||
// MAX_INST_SIZE assumes worst case
|
||||
auto OpMinAddress = RIPToDecode + PCOffset;
|
||||
auto OpMaxAddress = OpMinAddress + MAX_INST_SIZE;
|
||||
|
||||
auto OpMinPage = OpMinAddress & FHU::FEX_PAGE_MASK;
|
||||
auto OpMaxPage = OpMaxAddress & FHU::FEX_PAGE_MASK;
|
||||
|
||||
|
||||
if (OpMinPage != CurrentCodePage) {
|
||||
CurrentCodePage = OpMinPage;
|
||||
if (CodePages.insert(CurrentCodePage).second) {
|
||||
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
|
||||
}
|
||||
}
|
||||
|
||||
if (OpMaxPage != CurrentCodePage) {
|
||||
CurrentCodePage = OpMaxPage;
|
||||
if (CodePages.insert(CurrentCodePage).second) {
|
||||
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
|
||||
}
|
||||
}
|
||||
|
||||
bool ErrorDuringDecoding = !DecodeInstruction(RIPToDecode + PCOffset);
|
||||
|
||||
if (ErrorDuringDecoding) {
|
||||
LogMan::Msg::DFmt("Couldn't Decode something at 0x{:x}, Started at 0x{:x}", PC + PCOffset, PC);
|
||||
LogMan::Msg::DFmt("Couldn't Decode something at 0x{:x}, Started at 0x{:x}", RIPToDecode + PCOffset, PC);
|
||||
// Put an invalid instruction in the stream so the core can raise SIGILL if hit
|
||||
CurrentBlockDecoding.HasInvalidInstruction = true;
|
||||
// Error while decoding instruction. We don't know the table or instruction size
|
||||
|
||||
+1
-1
@@ -27,7 +27,7 @@ public:
|
||||
|
||||
Decoder(FEXCore::Context::Context *ctx);
|
||||
~Decoder();
|
||||
void DecodeInstructionsAtEntry(uint8_t const* InstStream, uint64_t PC);
|
||||
void DecodeInstructionsAtEntry(uint8_t const* InstStream, uint64_t PC, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage);
|
||||
|
||||
std::vector<DecodedBlocks> const *GetDecodedBlocks() const {
|
||||
return &Blocks;
|
||||
|
||||
+37
-11
@@ -125,7 +125,7 @@ static std::string hexstring(std::istringstream &ss, int delm) {
|
||||
return ret;
|
||||
}
|
||||
|
||||
static std::string encodeHex(unsigned char *data, size_t length) {
|
||||
static std::string encodeHex(const unsigned char *data, size_t length) {
|
||||
std::ostringstream ss;
|
||||
|
||||
for (size_t i=0; i < length; i++) {
|
||||
@@ -251,15 +251,15 @@ void GdbServer::SendACK(std::ostream &stream, bool NACK) {
|
||||
}
|
||||
|
||||
struct FEX_PACKED GDBContextDefinition {
|
||||
uint64_t gregs[16];
|
||||
uint64_t gregs[Core::CPUState::NUM_GPRS];
|
||||
uint64_t rip;
|
||||
uint32_t eflags;
|
||||
uint32_t cs, ss, ds, es, fs, gs;
|
||||
X80SoftFloat mm[8];
|
||||
X80SoftFloat mm[Core::CPUState::NUM_MMS];
|
||||
uint32_t fctrl;
|
||||
uint32_t fstat;
|
||||
uint32_t dummies[6];
|
||||
uint64_t xmm[16][2];
|
||||
uint64_t xmm[Core::CPUState::NUM_XMMS][4];
|
||||
uint32_t mxcsr;
|
||||
};
|
||||
|
||||
@@ -288,12 +288,12 @@ std::string GdbServer::readRegs() {
|
||||
memcpy(&GDB.gregs[0], &state.gregs[0], sizeof(GDB.gregs));
|
||||
memcpy(&GDB.rip, &state.rip, sizeof(GDB.rip));
|
||||
|
||||
for (size_t i = 0; i < 32; ++i) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_EFLAG_BITS; ++i) {
|
||||
uint64_t Flag = state.flags[i];
|
||||
GDB.eflags |= (Flag << i);
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < 8; ++i) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
memcpy(&GDB.mm[i], &state.mm[i], sizeof(GDB.mm));
|
||||
}
|
||||
|
||||
@@ -306,7 +306,7 @@ std::string GdbServer::readRegs() {
|
||||
GDB.fstat |= static_cast<uint32_t>(state.flags[FEXCore::X86State::X87FLAG_C2_LOC]) << 10;
|
||||
GDB.fstat |= static_cast<uint32_t>(state.flags[FEXCore::X86State::X87FLAG_C3_LOC]) << 14;
|
||||
|
||||
memcpy(&GDB.xmm[0], &state.xmm[0], sizeof(GDB.xmm));
|
||||
memcpy(&GDB.xmm[0], &state.xmm.avx.data[0], sizeof(GDB.xmm));
|
||||
|
||||
return encodeHex((unsigned char *)&GDB, sizeof(GDBContextDefinition));
|
||||
}
|
||||
@@ -346,7 +346,7 @@ GdbServer::HandledPacketType GdbServer::readReg(const std::string& packet) {
|
||||
}
|
||||
else if (addr == offsetof(GDBContextDefinition, eflags)) {
|
||||
uint32_t eflags{};
|
||||
for (size_t i = 0; i < 32; ++i) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_EFLAG_BITS; ++i) {
|
||||
uint64_t Flag = state.flags[i];
|
||||
eflags |= (Flag << i);
|
||||
}
|
||||
@@ -382,7 +382,9 @@ GdbServer::HandledPacketType GdbServer::readReg(const std::string& packet) {
|
||||
}
|
||||
else if (addr >= offsetof(GDBContextDefinition, xmm[0][0]) &&
|
||||
addr < offsetof(GDBContextDefinition, xmm[16][0])) {
|
||||
return {encodeHex((unsigned char *)(&state.xmm[(addr - offsetof(GDBContextDefinition, xmm[0][0])) / 16][0]), 16), HandledPacketType::TYPE_ACK};
|
||||
const auto XmmIndex = (addr - offsetof(GDBContextDefinition, xmm[0][0])) / Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto *Data = (unsigned char *)&state.xmm.avx.data[XmmIndex][0];
|
||||
return {encodeHex(Data, Core::CPUState::XMM_AVX_REG_SIZE), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
else if (addr == offsetof(GDBContextDefinition, mxcsr)) {
|
||||
uint32_t Empty{};
|
||||
@@ -424,7 +426,7 @@ std::string buildTargetXML() {
|
||||
// We want to just memcpy our x86 state to gdb, so we tell it the ordering.
|
||||
|
||||
// GPRs
|
||||
for (int i=0; i < 16; i++) {
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_GPRS; i++) {
|
||||
reg(FEXCore::Core::GetGRegName(i), "int64", 64);
|
||||
}
|
||||
|
||||
@@ -481,13 +483,37 @@ std::string buildTargetXML() {
|
||||
)";
|
||||
|
||||
// SSE regs
|
||||
for (int i=0; i < 16; i++) {
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
reg("xmm" + std::to_string(i), "vec128", 128);
|
||||
}
|
||||
|
||||
reg("mxcsr", "int", 32);
|
||||
|
||||
xml << "</feature>\n";
|
||||
|
||||
xml << "<feature name='org.gnu.gdb.i386.avx'>";
|
||||
xml <<
|
||||
R"(<vector id="v4f" type="ieee_single" count="4"/>
|
||||
<vector id="v2d" type="ieee_double" count="2"/>
|
||||
<vector id="v16i8" type="int8" count="16"/>
|
||||
<vector id="v8i16" type="int16" count="8"/>
|
||||
<vector id="v4i32" type="int32" count="4"/>
|
||||
<vector id="v2i64" type="int64" count="2"/>
|
||||
<union id="vec128">
|
||||
<field name="v4_float" type="v4f"/>
|
||||
<field name="v2_double" type="v2d"/>
|
||||
<field name="v16_int8" type="v16i8"/>
|
||||
<field name="v8_int16" type="v8i16"/>
|
||||
<field name="v4_int32" type="v4i32"/>
|
||||
<field name="v2_int64" type="v2i64"/>
|
||||
<field name="uint128" type="uint128"/>
|
||||
</union>
|
||||
)";
|
||||
for (size_t i = 0; i < Core::CPUState::NUM_XMMS; i++) {
|
||||
reg(fmt::format("ymm{}h", i), "vec128", 128);
|
||||
}
|
||||
xml << "</feature>\n";
|
||||
|
||||
xml << "</target>";
|
||||
xml << std::flush;
|
||||
|
||||
|
||||
+5
-1
@@ -1,5 +1,5 @@
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/HostFeatures.h"
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
@@ -62,6 +62,9 @@ HostFeatures::HostFeatures() {
|
||||
SupportsRCPC = Features.Has(vixl::CPUFeatures::Feature::kRCpc);
|
||||
SupportsTSOImm9 = Features.Has(vixl::CPUFeatures::Feature::kRCpcImm);
|
||||
|
||||
SupportsAVX = Features.Has(vixl::CPUFeatures::Feature::kSVE2) &&
|
||||
vixl::aarch64::CPU::ReadSVEVectorLengthInBits() >= 256;
|
||||
|
||||
// We need to get the CPU's cache line size
|
||||
// We expect sane targets that have correct cacheline sizes across clusters
|
||||
uint64_t CTR;
|
||||
@@ -82,6 +85,7 @@ HostFeatures::HostFeatures() {
|
||||
SupportsRAND = Features.has(Xbyak::util::Cpu::tRDRAND) && Features.has(Xbyak::util::Cpu::tRDSEED);
|
||||
SupportsRCPC = true;
|
||||
SupportsTSOImm9 = true;
|
||||
SupportsAVX = true;
|
||||
|
||||
// xbyak doesn't know how to check for CLZero
|
||||
uint32_t eax, ebx, ecx, edx;
|
||||
|
||||
+213
-211
@@ -20,7 +20,7 @@ DEF_OP(TruncElementPair) {
|
||||
|
||||
switch (IROp->Size) {
|
||||
case 4: {
|
||||
uint64_t *Src = GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t *Src = GetSrc<uint64_t*>(Data->SSAData, Op->Pair);
|
||||
uint64_t Result{};
|
||||
Result = Src[0] & ~0U;
|
||||
Result |= Src[1] << 32;
|
||||
@@ -69,11 +69,11 @@ DEF_OP(CycleCounter) {
|
||||
|
||||
DEF_OP(Add) {
|
||||
auto Op = IROp->C<IR::IROp_Add>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Func = [](auto a, auto b) { return a + b; };
|
||||
auto *Src1 = GetSrc<void*>(Data->SSAData, Op->Src1);
|
||||
auto *Src2 = GetSrc<void*>(Data->SSAData, Op->Src2);
|
||||
const auto Func = [](auto a, auto b) { return a + b; };
|
||||
|
||||
switch (OpSize) {
|
||||
DO_OP(4, uint32_t, Func)
|
||||
@@ -84,11 +84,11 @@ DEF_OP(Add) {
|
||||
|
||||
DEF_OP(Sub) {
|
||||
auto Op = IROp->C<IR::IROp_Sub>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Func = [](auto a, auto b) { return a - b; };
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Src1);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Src2);
|
||||
const auto Func = [](auto a, auto b) { return a - b; };
|
||||
|
||||
switch (OpSize) {
|
||||
DO_OP(4, uint32_t, Func)
|
||||
@@ -99,9 +99,9 @@ DEF_OP(Sub) {
|
||||
|
||||
DEF_OP(Neg) {
|
||||
auto Op = IROp->C<IR::IROp_Neg>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src = *GetSrc<int64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const uint64_t Src = *GetSrc<int64_t*>(Data->SSAData, Op->Src);
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
GD = -static_cast<int32_t>(Src);
|
||||
@@ -115,10 +115,10 @@ DEF_OP(Neg) {
|
||||
|
||||
DEF_OP(Mul) {
|
||||
auto Op = IROp->C<IR::IROp_Mul>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
@@ -138,10 +138,10 @@ DEF_OP(Mul) {
|
||||
|
||||
DEF_OP(UMul) {
|
||||
auto Op = IROp->C<IR::IROp_UMul>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
@@ -161,9 +161,9 @@ DEF_OP(UMul) {
|
||||
|
||||
DEF_OP(Div) {
|
||||
auto Op = IROp->C<IR::IROp_Div>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
@@ -179,7 +179,7 @@ DEF_OP(Div) {
|
||||
GD = static_cast<int64_t>(Src1) / static_cast<int64_t>(Src2);
|
||||
break;
|
||||
case 16: {
|
||||
__int128_t Tmp = *GetSrc<__int128_t*>(Data->SSAData, Op->Header.Args[0]) / *GetSrc<__int128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
__int128_t Tmp = *GetSrc<__int128_t*>(Data->SSAData, Op->Src1) / *GetSrc<__int128_t*>(Data->SSAData, Op->Src2);
|
||||
memcpy(GDP, &Tmp, 16);
|
||||
break;
|
||||
}
|
||||
@@ -189,10 +189,10 @@ DEF_OP(Div) {
|
||||
|
||||
DEF_OP(UDiv) {
|
||||
auto Op = IROp->C<IR::IROp_UDiv>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
@@ -208,7 +208,7 @@ DEF_OP(UDiv) {
|
||||
GD = static_cast<uint64_t>(Src1) / static_cast<uint64_t>(Src2);
|
||||
break;
|
||||
case 16: {
|
||||
__uint128_t Tmp = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]) / *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
__uint128_t Tmp = *GetSrc<__uint128_t*>(Data->SSAData, Op->Src1) / *GetSrc<__uint128_t*>(Data->SSAData, Op->Src2);
|
||||
memcpy(GDP, &Tmp, 16);
|
||||
break;
|
||||
}
|
||||
@@ -218,10 +218,10 @@ DEF_OP(UDiv) {
|
||||
|
||||
DEF_OP(Rem) {
|
||||
auto Op = IROp->C<IR::IROp_Rem>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
@@ -237,7 +237,7 @@ DEF_OP(Rem) {
|
||||
GD = static_cast<int64_t>(Src1) % static_cast<int64_t>(Src2);
|
||||
break;
|
||||
case 16: {
|
||||
__int128_t Tmp = *GetSrc<__int128_t*>(Data->SSAData, Op->Header.Args[0]) % *GetSrc<__int128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
__int128_t Tmp = *GetSrc<__int128_t*>(Data->SSAData, Op->Src1) % *GetSrc<__int128_t*>(Data->SSAData, Op->Src2);
|
||||
memcpy(GDP, &Tmp, 16);
|
||||
break;
|
||||
}
|
||||
@@ -247,10 +247,10 @@ DEF_OP(Rem) {
|
||||
|
||||
DEF_OP(URem) {
|
||||
auto Op = IROp->C<IR::IROp_URem>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
@@ -266,7 +266,7 @@ DEF_OP(URem) {
|
||||
GD = static_cast<uint64_t>(Src1) % static_cast<uint64_t>(Src2);
|
||||
break;
|
||||
case 16: {
|
||||
__uint128_t Tmp = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]) % *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
__uint128_t Tmp = *GetSrc<__uint128_t*>(Data->SSAData, Op->Src1) % *GetSrc<__uint128_t*>(Data->SSAData, Op->Src2);
|
||||
memcpy(GDP, &Tmp, 16);
|
||||
break;
|
||||
}
|
||||
@@ -276,10 +276,10 @@ DEF_OP(URem) {
|
||||
|
||||
DEF_OP(MulH) {
|
||||
auto Op = IROp->C<IR::IROp_MulH>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
@@ -298,10 +298,10 @@ DEF_OP(MulH) {
|
||||
|
||||
DEF_OP(UMulH) {
|
||||
auto Op = IROp->C<IR::IROp_UMulH>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
GD = static_cast<uint64_t>(Src1) * static_cast<uint64_t>(Src2);
|
||||
@@ -324,11 +324,11 @@ DEF_OP(UMulH) {
|
||||
|
||||
DEF_OP(Or) {
|
||||
auto Op = IROp->C<IR::IROp_Or>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Func = [](auto a, auto b) { return a | b; };
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Src1);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Src2);
|
||||
const auto Func = [](auto a, auto b) { return a | b; };
|
||||
|
||||
switch (OpSize) {
|
||||
DO_OP(1, uint8_t, Func)
|
||||
@@ -342,11 +342,11 @@ DEF_OP(Or) {
|
||||
|
||||
DEF_OP(And) {
|
||||
auto Op = IROp->C<IR::IROp_And>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Func = [](auto a, auto b) { return a & b; };
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Src1);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Src2);
|
||||
const auto Func = [](auto a, auto b) { return a & b; };
|
||||
|
||||
switch (OpSize) {
|
||||
DO_OP(1, uint8_t, Func)
|
||||
@@ -361,8 +361,8 @@ DEF_OP(Andn) {
|
||||
auto Op = IROp->C<IR::IROp_Andn>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Src1);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Src2);
|
||||
constexpr auto Func = [](auto a, auto b) {
|
||||
using Type = decltype(a);
|
||||
return static_cast<Type>(a & static_cast<Type>(~b));
|
||||
@@ -379,11 +379,11 @@ DEF_OP(Andn) {
|
||||
|
||||
DEF_OP(Xor) {
|
||||
auto Op = IROp->C<IR::IROp_Xor>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Func = [](auto a, auto b) { return a ^ b; };
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Src1);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Src2);
|
||||
const auto Func = [](auto a, auto b) { return a ^ b; };
|
||||
|
||||
switch (OpSize) {
|
||||
DO_OP(1, uint8_t, Func)
|
||||
@@ -396,11 +396,11 @@ DEF_OP(Xor) {
|
||||
|
||||
DEF_OP(Lshl) {
|
||||
auto Op = IROp->C<IR::IROp_Lshl>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint8_t Mask = OpSize * 8 - 1;
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
const uint8_t Mask = OpSize * 8 - 1;
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
GD = static_cast<uint32_t>(Src1) << (Src2 & Mask);
|
||||
@@ -414,11 +414,11 @@ DEF_OP(Lshl) {
|
||||
|
||||
DEF_OP(Lshr) {
|
||||
auto Op = IROp->C<IR::IROp_Lshr>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint8_t Mask = OpSize * 8 - 1;
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
const uint8_t Mask = OpSize * 8 - 1;
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
GD = static_cast<uint32_t>(Src1) >> (Src2 & Mask);
|
||||
@@ -432,11 +432,11 @@ DEF_OP(Lshr) {
|
||||
|
||||
DEF_OP(Ashr) {
|
||||
auto Op = IROp->C<IR::IROp_Ashr>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint8_t Mask = OpSize * 8 - 1;
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
const uint8_t Mask = OpSize * 8 - 1;
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
GD = (uint32_t)(static_cast<int32_t>(Src1) >> (Src2 & Mask));
|
||||
@@ -450,12 +450,12 @@ DEF_OP(Ashr) {
|
||||
|
||||
DEF_OP(Ror) {
|
||||
auto Op = IROp->C<IR::IROp_Ror>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Ror = [] (auto In, auto R) {
|
||||
auto RotateMask = sizeof(In) * 8 - 1;
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
const auto Ror = [] (auto In, auto R) {
|
||||
const auto RotateMask = sizeof(In) * 8 - 1;
|
||||
R &= RotateMask;
|
||||
return (In >> R) | (In << (sizeof(In) * 8 - R));
|
||||
};
|
||||
@@ -474,11 +474,11 @@ DEF_OP(Ror) {
|
||||
|
||||
DEF_OP(Extr) {
|
||||
auto Op = IROp->C<IR::IROp_Extr>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Extr = [] (auto Src1, auto Src2, uint8_t lsb) -> decltype(Src1) {
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Upper);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Lower);
|
||||
const auto Extr = [] (auto Src1, auto Src2, uint8_t lsb) -> decltype(Src1) {
|
||||
__uint128_t Result{};
|
||||
Result = Src1;
|
||||
Result <<= sizeof(Src1) * 8;
|
||||
@@ -500,7 +500,7 @@ DEF_OP(Extr) {
|
||||
}
|
||||
|
||||
DEF_OP(PDep) {
|
||||
const auto Op = IROp->C<IR::IROp_PExt>();
|
||||
const auto Op = IROp->C<IR::IROp_PDep>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (OpSize != 4 && OpSize != 8) {
|
||||
@@ -508,10 +508,10 @@ DEF_OP(PDep) {
|
||||
return;
|
||||
}
|
||||
|
||||
const uint64_t Input = OpSize == 4 ? *GetSrc<uint32_t*>(Data->SSAData, Op->Args(0))
|
||||
: *GetSrc<uint64_t*>(Data->SSAData, Op->Args(0));
|
||||
uint64_t Mask = OpSize == 4 ? *GetSrc<uint32_t*>(Data->SSAData, Op->Args(1))
|
||||
: *GetSrc<uint64_t*>(Data->SSAData, Op->Args(1));
|
||||
const uint64_t Input = OpSize == 4 ? *GetSrc<uint32_t*>(Data->SSAData, Op->Input)
|
||||
: *GetSrc<uint64_t*>(Data->SSAData, Op->Input);
|
||||
uint64_t Mask = OpSize == 4 ? *GetSrc<uint32_t*>(Data->SSAData, Op->Mask)
|
||||
: *GetSrc<uint64_t*>(Data->SSAData, Op->Mask);
|
||||
|
||||
uint64_t Result = 0;
|
||||
for (uint64_t Index = 0; Mask > 0; Index++) {
|
||||
@@ -532,10 +532,10 @@ DEF_OP(PExt) {
|
||||
return;
|
||||
}
|
||||
|
||||
const uint64_t Input = OpSize == 4 ? *GetSrc<uint32_t*>(Data->SSAData, Op->Args(0))
|
||||
: *GetSrc<uint64_t*>(Data->SSAData, Op->Args(0));
|
||||
uint64_t Mask = OpSize == 4 ? *GetSrc<uint32_t*>(Data->SSAData, Op->Args(1))
|
||||
: *GetSrc<uint64_t*>(Data->SSAData, Op->Args(1));
|
||||
const uint64_t Input = OpSize == 4 ? *GetSrc<uint32_t*>(Data->SSAData, Op->Input)
|
||||
: *GetSrc<uint64_t*>(Data->SSAData, Op->Input);
|
||||
uint64_t Mask = OpSize == 4 ? *GetSrc<uint32_t*>(Data->SSAData, Op->Mask)
|
||||
: *GetSrc<uint64_t*>(Data->SSAData, Op->Mask);
|
||||
|
||||
uint64_t Result = 0;
|
||||
for (uint64_t Offset = 0; Mask > 0; Offset++) {
|
||||
@@ -549,39 +549,39 @@ DEF_OP(PExt) {
|
||||
|
||||
DEF_OP(LDiv) {
|
||||
auto Op = IROp->C<IR::IROp_LDiv>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
uint16_t SrcLow = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t SrcHigh = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
int16_t Divisor = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
int32_t Source = (static_cast<uint32_t>(SrcHigh) << 16) | SrcLow;
|
||||
int32_t Res = Source / Divisor;
|
||||
const uint16_t SrcLow = *GetSrc<uint16_t*>(Data->SSAData, Op->Lower);
|
||||
const uint16_t SrcHigh = *GetSrc<uint16_t*>(Data->SSAData, Op->Upper);
|
||||
const int16_t Divisor = *GetSrc<uint16_t*>(Data->SSAData, Op->Divisor);
|
||||
const int32_t Source = (static_cast<uint32_t>(SrcHigh) << 16) | SrcLow;
|
||||
const int32_t Res = Source / Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
GD = static_cast<int16_t>(Res);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
uint32_t SrcLow = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t SrcHigh = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
int32_t Divisor = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
int64_t Source = (static_cast<uint64_t>(SrcHigh) << 32) | SrcLow;
|
||||
int64_t Res = Source / Divisor;
|
||||
const uint32_t SrcLow = *GetSrc<uint32_t*>(Data->SSAData, Op->Lower);
|
||||
const uint32_t SrcHigh = *GetSrc<uint32_t*>(Data->SSAData, Op->Upper);
|
||||
const int32_t Divisor = *GetSrc<uint32_t*>(Data->SSAData, Op->Divisor);
|
||||
const int64_t Source = (static_cast<uint64_t>(SrcHigh) << 32) | SrcLow;
|
||||
const int64_t Res = Source / Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
GD = static_cast<int32_t>(Res);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
uint64_t SrcLow = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t SrcHigh = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
int64_t Divisor = *GetSrc<int64_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
__int128_t Source = (static_cast<__int128_t>(SrcHigh) << 64) | SrcLow;
|
||||
__int128_t Res = Source / Divisor;
|
||||
const uint64_t SrcLow = *GetSrc<uint64_t*>(Data->SSAData, Op->Lower);
|
||||
const uint64_t SrcHigh = *GetSrc<uint64_t*>(Data->SSAData, Op->Upper);
|
||||
const int64_t Divisor = *GetSrc<int64_t*>(Data->SSAData, Op->Divisor);
|
||||
const __int128_t Source = (static_cast<__int128_t>(SrcHigh) << 64) | SrcLow;
|
||||
const __int128_t Res = Source / Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
memcpy(GDP, &Res, OpSize);
|
||||
@@ -593,39 +593,39 @@ DEF_OP(LDiv) {
|
||||
|
||||
DEF_OP(LUDiv) {
|
||||
auto Op = IROp->C<IR::IROp_LUDiv>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit divide from x86-64
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
uint16_t SrcLow = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t SrcHigh = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint16_t Divisor = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
uint32_t Source = (static_cast<uint32_t>(SrcHigh) << 16) | SrcLow;
|
||||
uint32_t Res = Source / Divisor;
|
||||
const uint16_t SrcLow = *GetSrc<uint16_t*>(Data->SSAData, Op->Lower);
|
||||
const uint16_t SrcHigh = *GetSrc<uint16_t*>(Data->SSAData, Op->Upper);
|
||||
const uint16_t Divisor = *GetSrc<uint16_t*>(Data->SSAData, Op->Divisor);
|
||||
const uint32_t Source = (static_cast<uint32_t>(SrcHigh) << 16) | SrcLow;
|
||||
const uint32_t Res = Source / Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
GD = static_cast<uint16_t>(Res);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
uint32_t SrcLow = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t SrcHigh = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint32_t Divisor = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
uint64_t Source = (static_cast<uint64_t>(SrcHigh) << 32) | SrcLow;
|
||||
uint64_t Res = Source / Divisor;
|
||||
const uint32_t SrcLow = *GetSrc<uint32_t*>(Data->SSAData, Op->Lower);
|
||||
const uint32_t SrcHigh = *GetSrc<uint32_t*>(Data->SSAData, Op->Upper);
|
||||
const uint32_t Divisor = *GetSrc<uint32_t*>(Data->SSAData, Op->Divisor);
|
||||
const uint64_t Source = (static_cast<uint64_t>(SrcHigh) << 32) | SrcLow;
|
||||
const uint64_t Res = Source / Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
GD = static_cast<uint32_t>(Res);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
uint64_t SrcLow = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t SrcHigh = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint64_t Divisor = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
__uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
|
||||
__uint128_t Res = Source / Divisor;
|
||||
const uint64_t SrcLow = *GetSrc<uint64_t*>(Data->SSAData, Op->Lower);
|
||||
const uint64_t SrcHigh = *GetSrc<uint64_t*>(Data->SSAData, Op->Upper);
|
||||
const uint64_t Divisor = *GetSrc<uint64_t*>(Data->SSAData, Op->Divisor);
|
||||
const __uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
|
||||
const __uint128_t Res = Source / Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
memcpy(GDP, &Res, OpSize);
|
||||
@@ -637,39 +637,39 @@ DEF_OP(LUDiv) {
|
||||
|
||||
DEF_OP(LRem) {
|
||||
auto Op = IROp->C<IR::IROp_LRem>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit Remainder from x86-64
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
uint16_t SrcLow = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t SrcHigh = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
int16_t Divisor = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
int32_t Source = (static_cast<uint32_t>(SrcHigh) << 16) | SrcLow;
|
||||
int32_t Res = Source % Divisor;
|
||||
const uint16_t SrcLow = *GetSrc<uint16_t*>(Data->SSAData, Op->Lower);
|
||||
const uint16_t SrcHigh = *GetSrc<uint16_t*>(Data->SSAData, Op->Upper);
|
||||
const int16_t Divisor = *GetSrc<uint16_t*>(Data->SSAData, Op->Divisor);
|
||||
const int32_t Source = (static_cast<uint32_t>(SrcHigh) << 16) | SrcLow;
|
||||
const int32_t Res = Source % Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
GD = static_cast<int16_t>(Res);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
uint32_t SrcLow = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t SrcHigh = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
int32_t Divisor = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
int64_t Source = (static_cast<uint64_t>(SrcHigh) << 32) | SrcLow;
|
||||
int64_t Res = Source % Divisor;
|
||||
const uint32_t SrcLow = *GetSrc<uint32_t*>(Data->SSAData, Op->Lower);
|
||||
const uint32_t SrcHigh = *GetSrc<uint32_t*>(Data->SSAData, Op->Upper);
|
||||
const int32_t Divisor = *GetSrc<uint32_t*>(Data->SSAData, Op->Divisor);
|
||||
const int64_t Source = (static_cast<uint64_t>(SrcHigh) << 32) | SrcLow;
|
||||
const int64_t Res = Source % Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
GD = static_cast<int32_t>(Res);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
uint64_t SrcLow = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t SrcHigh = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
int64_t Divisor = *GetSrc<int64_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
__int128_t Source = (static_cast<__int128_t>(SrcHigh) << 64) | SrcLow;
|
||||
__int128_t Res = Source % Divisor;
|
||||
const uint64_t SrcLow = *GetSrc<uint64_t*>(Data->SSAData, Op->Lower);
|
||||
const uint64_t SrcHigh = *GetSrc<uint64_t*>(Data->SSAData, Op->Upper);
|
||||
const int64_t Divisor = *GetSrc<int64_t*>(Data->SSAData, Op->Divisor);
|
||||
const __int128_t Source = (static_cast<__int128_t>(SrcHigh) << 64) | SrcLow;
|
||||
const __int128_t Res = Source % Divisor;
|
||||
// We only store the lower bits of the result
|
||||
memcpy(GDP, &Res, OpSize);
|
||||
break;
|
||||
@@ -680,39 +680,39 @@ DEF_OP(LRem) {
|
||||
|
||||
DEF_OP(LURem) {
|
||||
auto Op = IROp->C<IR::IROp_LURem>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
// Each source is OpSize in size
|
||||
// So you can have up to a 128bit Remainder from x86-64
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
uint16_t SrcLow = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint16_t SrcHigh = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint16_t Divisor = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
uint32_t Source = (static_cast<uint32_t>(SrcHigh) << 16) | SrcLow;
|
||||
uint32_t Res = Source % Divisor;
|
||||
const uint16_t SrcLow = *GetSrc<uint16_t*>(Data->SSAData, Op->Lower);
|
||||
const uint16_t SrcHigh = *GetSrc<uint16_t*>(Data->SSAData, Op->Upper);
|
||||
const uint16_t Divisor = *GetSrc<uint16_t*>(Data->SSAData, Op->Divisor);
|
||||
const uint32_t Source = (static_cast<uint32_t>(SrcHigh) << 16) | SrcLow;
|
||||
const uint32_t Res = Source % Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
GD = static_cast<uint16_t>(Res);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
uint32_t SrcLow = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint32_t SrcHigh = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint32_t Divisor = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
uint64_t Source = (static_cast<uint64_t>(SrcHigh) << 32) | SrcLow;
|
||||
uint64_t Res = Source % Divisor;
|
||||
const uint32_t SrcLow = *GetSrc<uint32_t*>(Data->SSAData, Op->Lower);
|
||||
const uint32_t SrcHigh = *GetSrc<uint32_t*>(Data->SSAData, Op->Upper);
|
||||
const uint32_t Divisor = *GetSrc<uint32_t*>(Data->SSAData, Op->Divisor);
|
||||
const uint64_t Source = (static_cast<uint64_t>(SrcHigh) << 32) | SrcLow;
|
||||
const uint64_t Res = Source % Divisor;
|
||||
|
||||
// We only store the lower bits of the result
|
||||
GD = static_cast<uint32_t>(Res);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
uint64_t SrcLow = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t SrcHigh = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint64_t Divisor = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
__uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
|
||||
__uint128_t Res = Source % Divisor;
|
||||
const uint64_t SrcLow = *GetSrc<uint64_t*>(Data->SSAData, Op->Lower);
|
||||
const uint64_t SrcHigh = *GetSrc<uint64_t*>(Data->SSAData, Op->Upper);
|
||||
const uint64_t Divisor = *GetSrc<uint64_t*>(Data->SSAData, Op->Divisor);
|
||||
const __uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
|
||||
const __uint128_t Res = Source % Divisor;
|
||||
// We only store the lower bits of the result
|
||||
memcpy(GDP, &Res, OpSize);
|
||||
break;
|
||||
@@ -723,62 +723,62 @@ DEF_OP(LURem) {
|
||||
|
||||
DEF_OP(Not) {
|
||||
auto Op = IROp->C<IR::IROp_Not>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Src);
|
||||
const uint64_t mask[9]= { 0, 0xFF, 0xFFFF, 0, 0xFFFFFFFF, 0, 0, 0, 0xFFFFFFFFFFFFFFFFULL };
|
||||
uint64_t Mask = mask[OpSize];
|
||||
const uint64_t Mask = mask[OpSize];
|
||||
GD = (~Src) & Mask;
|
||||
}
|
||||
|
||||
DEF_OP(Popcount) {
|
||||
auto Op = IROp->C<IR::IROp_Popcount>();
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Src);
|
||||
GD = std::popcount(Src);
|
||||
}
|
||||
|
||||
DEF_OP(FindLSB) {
|
||||
auto Op = IROp->C<IR::IROp_FindLSB>();
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Result = FindFirstSetBit(Src);
|
||||
const uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Src);
|
||||
const uint64_t Result = FindFirstSetBit(Src);
|
||||
GD = Result - 1;
|
||||
}
|
||||
|
||||
DEF_OP(FindMSB) {
|
||||
auto Op = IROp->C<IR::IROp_FindMSB>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]))) - 1; break;
|
||||
case 2: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]))) - 1; break;
|
||||
case 4: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]))) - 1; break;
|
||||
case 8: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]))) - 1; break;
|
||||
case 1: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint8_t*>(Data->SSAData, Op->Src))) - 1; break;
|
||||
case 2: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint16_t*>(Data->SSAData, Op->Src))) - 1; break;
|
||||
case 4: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint32_t*>(Data->SSAData, Op->Src))) - 1; break;
|
||||
case 8: GD = (OpSize * 8 - std::countl_zero(*GetSrc<uint64_t*>(Data->SSAData, Op->Src))) - 1; break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown FindMSB size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(FindTrailingZeros) {
|
||||
auto Op = IROp->C<IR::IROp_FindTrailingZeros>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
auto Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Src);
|
||||
GD = std::countr_zero(Src);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
auto Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Src);
|
||||
GD = std::countr_zero(Src);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
auto Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Src);
|
||||
GD = std::countr_zero(Src);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
auto Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Src);
|
||||
GD = std::countr_zero(Src);
|
||||
break;
|
||||
}
|
||||
@@ -788,26 +788,26 @@ DEF_OP(FindTrailingZeros) {
|
||||
|
||||
DEF_OP(CountLeadingZeroes) {
|
||||
auto Op = IROp->C<IR::IROp_CountLeadingZeroes>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
auto Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<uint8_t*>(Data->SSAData, Op->Src);
|
||||
GD = std::countl_zero(Src);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
auto Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<uint16_t*>(Data->SSAData, Op->Src);
|
||||
GD = std::countl_zero(Src);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
auto Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<uint32_t*>(Data->SSAData, Op->Src);
|
||||
GD = std::countl_zero(Src);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
auto Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Src);
|
||||
GD = std::countl_zero(Src);
|
||||
break;
|
||||
}
|
||||
@@ -817,12 +817,12 @@ DEF_OP(CountLeadingZeroes) {
|
||||
|
||||
DEF_OP(Rev) {
|
||||
auto Op = IROp->C<IR::IROp_Rev>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
switch (OpSize) {
|
||||
case 2: GD = BSwap16(*GetSrc<uint16_t*>(Data->SSAData, Op->Header.Args[0])); break;
|
||||
case 4: GD = BSwap32(*GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[0])); break;
|
||||
case 8: GD = BSwap64(*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0])); break;
|
||||
case 2: GD = BSwap16(*GetSrc<uint16_t*>(Data->SSAData, Op->Src)); break;
|
||||
case 4: GD = BSwap32(*GetSrc<uint32_t*>(Data->SSAData, Op->Src)); break;
|
||||
case 8: GD = BSwap64(*GetSrc<uint64_t*>(Data->SSAData, Op->Src)); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown REV size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
@@ -830,34 +830,36 @@ DEF_OP(Rev) {
|
||||
DEF_OP(Bfi) {
|
||||
auto Op = IROp->C<IR::IROp_Bfi>();
|
||||
uint64_t SourceMask = (1ULL << Op->Width) - 1;
|
||||
if (Op->Width == 64)
|
||||
if (Op->Width == 64) {
|
||||
SourceMask = ~0ULL;
|
||||
uint64_t DestMask = ~(SourceMask << Op->lsb);
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint64_t Res = (Src1 & DestMask) | ((Src2 & SourceMask) << Op->lsb);
|
||||
}
|
||||
const uint64_t DestMask = ~(SourceMask << Op->lsb);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Dest);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Src);
|
||||
const uint64_t Res = (Src1 & DestMask) | ((Src2 & SourceMask) << Op->lsb);
|
||||
GD = Res;
|
||||
}
|
||||
|
||||
DEF_OP(Bfe) {
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
|
||||
LOGMAN_THROW_A_FMT(IROp->Size <= 8, "OpSize is too large for BFE: {}", IROp->Size);
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size <= 8, "OpSize is too large for BFE: {}", IROp->Size);
|
||||
uint64_t SourceMask = (1ULL << Op->Width) - 1;
|
||||
if (Op->Width == 64)
|
||||
if (Op->Width == 64) {
|
||||
SourceMask = ~0ULL;
|
||||
}
|
||||
SourceMask <<= Op->lsb;
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Src);
|
||||
GD = (Src & SourceMask) >> Op->lsb;
|
||||
}
|
||||
|
||||
DEF_OP(Sbfe) {
|
||||
auto Op = IROp->C<IR::IROp_Sbfe>();
|
||||
|
||||
LOGMAN_THROW_A_FMT(IROp->Size <= 8, "OpSize is too large for SBFE: {}", IROp->Size);
|
||||
int64_t Src = *GetSrc<int64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t ShiftLeftAmount = (64 - (Op->Width + Op->lsb));
|
||||
uint64_t ShiftRightAmount = ShiftLeftAmount + Op->lsb;
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size <= 8, "OpSize is too large for SBFE: {}", IROp->Size);
|
||||
int64_t Src = *GetSrc<int64_t*>(Data->SSAData, Op->Src);
|
||||
const uint64_t ShiftLeftAmount = (64 - (Op->Width + Op->lsb));
|
||||
const uint64_t ShiftRightAmount = ShiftLeftAmount + Op->lsb;
|
||||
Src <<= ShiftLeftAmount;
|
||||
Src >>= ShiftRightAmount;
|
||||
GD = Src;
|
||||
@@ -865,20 +867,20 @@ DEF_OP(Sbfe) {
|
||||
|
||||
DEF_OP(Select) {
|
||||
auto Op = IROp->C<IR::IROp_Select>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp2);
|
||||
|
||||
uint64_t ArgTrue;
|
||||
uint64_t ArgFalse;
|
||||
|
||||
if (OpSize == 4) {
|
||||
ArgTrue = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
ArgFalse = *GetSrc<uint32_t*>(Data->SSAData, Op->Header.Args[3]);
|
||||
ArgTrue = *GetSrc<uint32_t*>(Data->SSAData, Op->TrueVal);
|
||||
ArgFalse = *GetSrc<uint32_t*>(Data->SSAData, Op->FalseVal);
|
||||
} else {
|
||||
ArgTrue = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[2]);
|
||||
ArgFalse = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[3]);
|
||||
ArgTrue = *GetSrc<uint64_t*>(Data->SSAData, Op->TrueVal);
|
||||
ArgFalse = *GetSrc<uint64_t*>(Data->SSAData, Op->FalseVal);
|
||||
}
|
||||
|
||||
bool CompResult;
|
||||
@@ -894,9 +896,9 @@ DEF_OP(Select) {
|
||||
DEF_OP(VExtractToGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
|
||||
uint32_t SourceSize = GetOpSize(Data->CurrentIR, Op->Header.Args[0]);
|
||||
const uint32_t SourceSize = GetOpSize(Data->CurrentIR, Op->Vector);
|
||||
|
||||
LOGMAN_THROW_A_FMT(IROp->Size <= 16, "OpSize is too large for VExtractToGPR: {}", IROp->Size);
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size <= 16, "OpSize is too large for VExtractToGPR: {}", IROp->Size);
|
||||
|
||||
if (SourceSize == 16) {
|
||||
__uint128_t SourceMask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
|
||||
@@ -904,7 +906,7 @@ DEF_OP(VExtractToGPR) {
|
||||
if (Op->Header.ElementSize == 8)
|
||||
SourceMask = ~0ULL;
|
||||
|
||||
__uint128_t Src = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
__uint128_t Src = *GetSrc<__uint128_t*>(Data->SSAData, Op->Vector);
|
||||
Src >>= Shift;
|
||||
Src &= SourceMask;
|
||||
memcpy(GDP, &Src, Op->Header.ElementSize);
|
||||
@@ -915,7 +917,7 @@ DEF_OP(VExtractToGPR) {
|
||||
if (Op->Header.ElementSize == 8)
|
||||
SourceMask = ~0ULL;
|
||||
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Vector);
|
||||
Src >>= Shift;
|
||||
Src &= SourceMask;
|
||||
GD = Src;
|
||||
@@ -924,25 +926,25 @@ DEF_OP(VExtractToGPR) {
|
||||
|
||||
DEF_OP(Float_ToGPR_ZS) {
|
||||
auto Op = IROp->C<IR::IROp_Float_ToGPR_ZS>();
|
||||
uint16_t Conv = (IROp->Size << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (IROp->Size << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0804: { // int64_t <- float
|
||||
int64_t Dst = (int64_t)std::trunc(*GetSrc<float*>(Data->SSAData, Op->Header.Args[0]));
|
||||
const int64_t Dst = (int64_t)std::trunc(*GetSrc<float*>(Data->SSAData, Op->Scalar));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
case 0x0808: { // int64_t <- double
|
||||
int64_t Dst = (int64_t)std::trunc(*GetSrc<double*>(Data->SSAData, Op->Header.Args[0]));
|
||||
const int64_t Dst = (int64_t)std::trunc(*GetSrc<double*>(Data->SSAData, Op->Scalar));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
case 0x0404: { // int32_t <- float
|
||||
int32_t Dst = (int32_t)std::trunc(*GetSrc<float*>(Data->SSAData, Op->Header.Args[0]));
|
||||
const int32_t Dst = (int32_t)std::trunc(*GetSrc<float*>(Data->SSAData, Op->Scalar));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // int32_t <- double
|
||||
int32_t Dst = (int32_t)std::trunc(*GetSrc<double*>(Data->SSAData, Op->Header.Args[0]));
|
||||
const int32_t Dst = (int32_t)std::trunc(*GetSrc<double*>(Data->SSAData, Op->Scalar));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
@@ -951,25 +953,25 @@ DEF_OP(Float_ToGPR_ZS) {
|
||||
|
||||
DEF_OP(Float_ToGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_ToGPR_S>();
|
||||
uint16_t Conv = (IROp->Size << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (IROp->Size << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0804: { // int64_t <- float
|
||||
int64_t Dst = (int64_t)std::nearbyint(*GetSrc<float*>(Data->SSAData, Op->Header.Args[0]));
|
||||
const int64_t Dst = (int64_t)std::nearbyint(*GetSrc<float*>(Data->SSAData, Op->Scalar));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
case 0x0808: { // int64_t <- double
|
||||
int64_t Dst = (int64_t)std::nearbyint(*GetSrc<double*>(Data->SSAData, Op->Header.Args[0]));
|
||||
const int64_t Dst = (int64_t)std::nearbyint(*GetSrc<double*>(Data->SSAData, Op->Scalar));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
case 0x0404: { // int32_t <- float
|
||||
int32_t Dst = (int32_t)std::nearbyint(*GetSrc<float*>(Data->SSAData, Op->Header.Args[0]));
|
||||
const int32_t Dst = (int32_t)std::nearbyint(*GetSrc<float*>(Data->SSAData, Op->Scalar));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // int32_t <- double
|
||||
int32_t Dst = (int32_t)std::nearbyint(*GetSrc<double*>(Data->SSAData, Op->Header.Args[0]));
|
||||
const int32_t Dst = (int32_t)std::nearbyint(*GetSrc<double*>(Data->SSAData, Op->Scalar));
|
||||
memcpy(GDP, &Dst, IROp->Size);
|
||||
break;
|
||||
}
|
||||
@@ -980,9 +982,9 @@ DEF_OP(FCmp) {
|
||||
auto Op = IROp->C<IR::IROp_FCmp>();
|
||||
uint32_t ResultFlags{};
|
||||
if (Op->ElementSize == 4) {
|
||||
float Src1 = *GetSrc<float*>(Data->SSAData, Op->Header.Args[0]);
|
||||
float Src2 = *GetSrc<float*>(Data->SSAData, Op->Header.Args[1]);
|
||||
bool Unordered = std::isnan(Src1) || std::isnan(Src2);
|
||||
const float Src1 = *GetSrc<float*>(Data->SSAData, Op->Scalar1);
|
||||
const float Src2 = *GetSrc<float*>(Data->SSAData, Op->Scalar2);
|
||||
const bool Unordered = std::isnan(Src1) || std::isnan(Src2);
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_LT)) {
|
||||
if (Unordered || (Src1 < Src2)) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_LT);
|
||||
@@ -1000,9 +1002,9 @@ DEF_OP(FCmp) {
|
||||
}
|
||||
}
|
||||
else {
|
||||
double Src1 = *GetSrc<double*>(Data->SSAData, Op->Header.Args[0]);
|
||||
double Src2 = *GetSrc<double*>(Data->SSAData, Op->Header.Args[1]);
|
||||
bool Unordered = std::isnan(Src1) || std::isnan(Src2);
|
||||
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Scalar1);
|
||||
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Scalar2);
|
||||
const bool Unordered = std::isnan(Src1) || std::isnan(Src2);
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_LT)) {
|
||||
if (Unordered || (Src1 < Src2)) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_LT);
|
||||
|
||||
@@ -4,6 +4,7 @@ tags: backend|interpreter
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
@@ -25,20 +26,13 @@ static void SignalReturn(FEXCore::Core::InternalThreadState *Thread) {
|
||||
}
|
||||
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(GuestCallDirect) {
|
||||
LogMan::Msg::DFmt("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(GuestCallIndirect) {
|
||||
LogMan::Msg::DFmt("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(SignalReturn) {
|
||||
SignalReturn(Data->State);
|
||||
}
|
||||
|
||||
DEF_OP(CallbackReturn) {
|
||||
Data->State->CTX->InterpreterCallbackReturn(Data->State, Data->StackEntry);
|
||||
Data->State->CurrentFrame->Pointers.Interpreter.CallbackReturn(Data->State, Data->StackEntry);
|
||||
}
|
||||
|
||||
DEF_OP(ExitFunction) {
|
||||
@@ -48,7 +42,7 @@ DEF_OP(ExitFunction) {
|
||||
uintptr_t* ContextPtr = reinterpret_cast<uintptr_t*>(Data->State->CurrentFrame);
|
||||
|
||||
void *ContextData = reinterpret_cast<void*>(ContextPtr);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->NewRIP);
|
||||
|
||||
memcpy(ContextData, Src, OpSize);
|
||||
|
||||
@@ -57,22 +51,22 @@ DEF_OP(ExitFunction) {
|
||||
|
||||
DEF_OP(Jump) {
|
||||
auto Op = IROp->C<IR::IROp_Jump>();
|
||||
uintptr_t ListBegin = Data->CurrentIR->GetListData();
|
||||
uintptr_t DataBegin = Data->CurrentIR->GetData();
|
||||
const uintptr_t ListBegin = Data->CurrentIR->GetListData();
|
||||
const uintptr_t DataBegin = Data->CurrentIR->GetData();
|
||||
|
||||
Data->BlockIterator = IR::NodeIterator(ListBegin, DataBegin, Op->Header.Args[0]);
|
||||
Data->BlockIterator = IR::NodeIterator(ListBegin, DataBegin, Op->TargetBlock);
|
||||
Data->BlockResults.Redo = true;
|
||||
}
|
||||
|
||||
DEF_OP(CondJump) {
|
||||
auto Op = IROp->C<IR::IROp_CondJump>();
|
||||
uintptr_t ListBegin = Data->CurrentIR->GetListData();
|
||||
uintptr_t DataBegin = Data->CurrentIR->GetData();
|
||||
const uintptr_t ListBegin = Data->CurrentIR->GetListData();
|
||||
const uintptr_t DataBegin = Data->CurrentIR->GetData();
|
||||
|
||||
bool CompResult;
|
||||
|
||||
uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp1);
|
||||
uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp2);
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp2);
|
||||
|
||||
if (Op->CompareSize == 4)
|
||||
CompResult = IsConditionTrue<uint32_t, int32_t, float>(Op->Cond.Val, Src1, Src2);
|
||||
@@ -133,7 +127,7 @@ DEF_OP(Thunk) {
|
||||
auto Op = IROp->C<IR::IROp_Thunk>();
|
||||
|
||||
auto thunkFn = Data->State->CTX->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
thunkFn(*GetSrc<void**>(Data->SSAData, Op->Header.Args[0]));
|
||||
thunkFn(*GetSrc<void**>(Data->SSAData, Op->ArgPtr));
|
||||
}
|
||||
|
||||
DEF_OP(ValidateCode) {
|
||||
@@ -147,15 +141,15 @@ DEF_OP(ValidateCode) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(RemoveThreadCodeEntry) {
|
||||
Data->State->CTX->RemoveThreadCodeEntry(Data->State, Data->CurrentEntry);
|
||||
DEF_OP(ThreadRemoveCodeEntry) {
|
||||
Data->State->CTX->ThreadRemoveCodeEntryFromJit(Data->State->CurrentFrame, Data->CurrentEntry);
|
||||
}
|
||||
|
||||
DEF_OP(CPUID) {
|
||||
auto Op = IROp->C<IR::IROp_CPUID>();
|
||||
uint64_t *DstPtr = GetDest<uint64_t*>(Data->SSAData, Node);
|
||||
uint64_t Arg = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Leaf = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const uint64_t Arg = *GetSrc<uint64_t*>(Data->SSAData, Op->Function);
|
||||
const uint64_t Leaf = *GetSrc<uint64_t*>(Data->SSAData, Op->Leaf);
|
||||
|
||||
auto Results = Data->State->CTX->CPUID.RunFunction(Arg, Leaf);
|
||||
memcpy(DstPtr, &Results, sizeof(uint32_t) * 4);
|
||||
|
||||
@@ -14,10 +14,10 @@ namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(VInsGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->DestVector);
|
||||
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Src);
|
||||
|
||||
uint64_t Offset = Op->DestIdx * Op->Header.ElementSize * 8;
|
||||
__uint128_t Mask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
|
||||
@@ -35,31 +35,31 @@ DEF_OP(VInsGPR) {
|
||||
|
||||
DEF_OP(VCastFromGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VCastFromGPR>();
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Header.Args[0]), Op->Header.ElementSize);
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Src), Op->Header.ElementSize);
|
||||
}
|
||||
|
||||
DEF_OP(Float_FromGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0404: { // Float <- int32_t
|
||||
float Dst = (float)*GetSrc<int32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const float Dst = (float)*GetSrc<int32_t*>(Data->SSAData, Op->Src);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- int64_t
|
||||
float Dst = (float)*GetSrc<int64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const float Dst = (float)*GetSrc<int64_t*>(Data->SSAData, Op->Src);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- int32_t
|
||||
double Dst = (double)*GetSrc<int32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const double Dst = (double)*GetSrc<int32_t*>(Data->SSAData, Op->Src);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
break;
|
||||
}
|
||||
case 0x0808: { // Double <- int64_t
|
||||
double Dst = (double)*GetSrc<int64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const double Dst = (double)*GetSrc<int64_t*>(Data->SSAData, Op->Src);
|
||||
memcpy(GDP, &Dst, Op->Header.ElementSize);
|
||||
break;
|
||||
}
|
||||
@@ -68,15 +68,15 @@ DEF_OP(Float_FromGPR_S) {
|
||||
|
||||
DEF_OP(Float_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FToF>();
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- Float
|
||||
double Dst = (double)*GetSrc<float*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const double Dst = (double)*GetSrc<float*>(Data->SSAData, Op->Scalar);
|
||||
memcpy(GDP, &Dst, 8);
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
float Dst = (float)*GetSrc<double*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const float Dst = (float)*GetSrc<double*>(Data->SSAData, Op->Scalar);
|
||||
memcpy(GDP, &Dst, 4);
|
||||
break;
|
||||
}
|
||||
@@ -86,14 +86,14 @@ DEF_OP(Float_FToF) {
|
||||
|
||||
DEF_OP(Vector_SToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
|
||||
auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
const auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, float, int32_t, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, double, int64_t, Func, 0, 0)
|
||||
@@ -104,14 +104,14 @@ DEF_OP(Vector_SToF) {
|
||||
|
||||
DEF_OP(Vector_FToZS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
|
||||
auto Func = [](auto a, auto min, auto max) { return std::trunc(a); };
|
||||
const auto Func = [](auto a, auto min, auto max) { return std::trunc(a); };
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, float, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, double, Func, 0, 0)
|
||||
@@ -122,14 +122,14 @@ DEF_OP(Vector_FToZS) {
|
||||
|
||||
DEF_OP(Vector_FToS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
|
||||
auto Func = [](auto a, auto min, auto max) { return std::nearbyint(a); };
|
||||
const auto Func = [](auto a, auto min, auto max) { return std::nearbyint(a); };
|
||||
switch (Op->Header.ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, float, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, double, Func, 0, 0)
|
||||
@@ -140,14 +140,14 @@ DEF_OP(Vector_FToS) {
|
||||
|
||||
DEF_OP(Vector_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
const auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- float
|
||||
// Only the lower elements from the source
|
||||
@@ -172,17 +172,17 @@ DEF_OP(Vector_FToF) {
|
||||
|
||||
DEF_OP(Vector_FToI) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
|
||||
uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
auto Func_Nearest = [](auto a) { return std::rint(a); };
|
||||
auto Func_Neg = [](auto a) { return std::floor(a); };
|
||||
auto Func_Pos = [](auto a) { return std::ceil(a); };
|
||||
auto Func_Trunc = [](auto a) { return std::trunc(a); };
|
||||
auto Func_Host = [](auto a) { return std::rint(a); };
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const auto Func_Nearest = [](auto a) { return std::rint(a); };
|
||||
const auto Func_Neg = [](auto a) { return std::floor(a); };
|
||||
const auto Func_Pos = [](auto a) { return std::ceil(a); };
|
||||
const auto Func_Trunc = [](auto a) { return std::trunc(a); };
|
||||
const auto Func_Host = [](auto a) { return std::rint(a); };
|
||||
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
|
||||
@@ -360,7 +360,7 @@ namespace FEXCore::CPU {
|
||||
|
||||
DEF_OP(AESImc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESImc>();
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Vector);
|
||||
|
||||
// Pseudo-code
|
||||
// Dst = InvMixColumns(STATE)
|
||||
@@ -371,8 +371,8 @@ DEF_OP(AESImc) {
|
||||
|
||||
DEF_OP(AESEnc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->State);
|
||||
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Key);
|
||||
|
||||
// Pseudo-code
|
||||
// STATE = Src1
|
||||
@@ -391,8 +391,8 @@ DEF_OP(AESEnc) {
|
||||
|
||||
DEF_OP(AESEncLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->State);
|
||||
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Key);
|
||||
|
||||
// Pseudo-code
|
||||
// STATE = Src1
|
||||
@@ -409,8 +409,8 @@ DEF_OP(AESEncLast) {
|
||||
|
||||
DEF_OP(AESDec) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->State);
|
||||
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Key);
|
||||
|
||||
// Pseudo-code
|
||||
// STATE = Src1
|
||||
@@ -429,8 +429,8 @@ DEF_OP(AESDec) {
|
||||
|
||||
DEF_OP(AESDecLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
__uint128_t Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
__uint128_t Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->State);
|
||||
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Key);
|
||||
|
||||
// Pseudo-code
|
||||
// STATE = Src1
|
||||
@@ -447,7 +447,7 @@ DEF_OP(AESDecLast) {
|
||||
|
||||
DEF_OP(AESKeyGenAssist) {
|
||||
auto Op = IROp->C<IR::IROp_VAESKeyGenAssist>();
|
||||
uint8_t *Src1 = GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const uint8_t *Src1 = GetSrc<uint8_t*>(Data->SSAData, Op->Src);
|
||||
|
||||
// Pseudo-code
|
||||
// X3 = Src1[127:96]
|
||||
@@ -513,6 +513,44 @@ DEF_OP(CRC32) {
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
}
|
||||
|
||||
DEF_OP(PCLMUL) {
|
||||
auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
|
||||
const auto Selector = Op->Selector;
|
||||
auto* Dst = GetDest<uint64_t*>(Data->SSAData, Node);
|
||||
auto* Src1 = GetSrc<uint64_t*>(Data->SSAData, Op->Src1);
|
||||
auto* Src2 = GetSrc<uint64_t*>(Data->SSAData, Op->Src2);
|
||||
|
||||
const uint64_t TMP1 = (Selector & 0x01) == 0 ? Src1[0] : Src1[1];
|
||||
const uint64_t TMP2 = (Selector & 0x10) == 0 ? Src2[0] : Src2[1];
|
||||
|
||||
const auto make_lo = [](uint64_t lhs, uint64_t rhs) {
|
||||
uint64_t result = 0;
|
||||
|
||||
for (size_t i = 0; i < 64; i++) {
|
||||
if ((lhs & (1ULL << i)) != 0) {
|
||||
result ^= rhs << i;
|
||||
}
|
||||
}
|
||||
|
||||
return result;
|
||||
};
|
||||
const auto make_hi = [](uint64_t lhs, uint64_t rhs) {
|
||||
uint64_t result = 0;
|
||||
|
||||
for (size_t i = 1; i < 64; i++) {
|
||||
if ((lhs & (1ULL << i)) != 0) {
|
||||
result ^= rhs >> (64 - i);
|
||||
}
|
||||
}
|
||||
|
||||
return result;
|
||||
};
|
||||
|
||||
Dst[0] = make_lo(TMP1, TMP2);
|
||||
Dst[1] = make_hi(TMP1, TMP2);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
+83
-96
@@ -20,99 +20,90 @@ DEF_OP(F80LOADFCW) {
|
||||
|
||||
DEF_OP(F80ADD) {
|
||||
auto Op = IROp->C<IR::IROp_F80Add>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FADD(Src1, Src2);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FADD(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80SUB) {
|
||||
auto Op = IROp->C<IR::IROp_F80Sub>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FSUB(Src1, Src2);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FSUB(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80MUL) {
|
||||
auto Op = IROp->C<IR::IROp_F80Mul>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FMUL(Src1, Src2);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FMUL(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80DIV) {
|
||||
auto Op = IROp->C<IR::IROp_F80Div>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FDIV(Src1, Src2);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FDIV(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80FYL2X) {
|
||||
auto Op = IROp->C<IR::IROp_F80FYL2X>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FYL2X(Src1, Src2);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FYL2X(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80ATAN) {
|
||||
auto Op = IROp->C<IR::IROp_F80ATAN>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FATAN(Src1, Src2);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FATAN(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80FPREM1) {
|
||||
auto Op = IROp->C<IR::IROp_F80FPREM1>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FREM1(Src1, Src2);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FREM1(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80FPREM) {
|
||||
auto Op = IROp->C<IR::IROp_F80FPREM>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FREM(Src1, Src2);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FREM(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80SCALE) {
|
||||
auto Op = IROp->C<IR::IROp_F80SCALE>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FSCALE(Src1, Src2);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FSCALE(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80CVT) {
|
||||
auto Op = IROp->C<IR::IROp_F80CVT>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
@@ -131,9 +122,9 @@ DEF_OP(F80CVT) {
|
||||
|
||||
DEF_OP(F80CVTINT) {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTInt>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
@@ -160,13 +151,13 @@ DEF_OP(F80CVTTO) {
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 4: {
|
||||
float Src = *GetSrc<float *>(Data->SSAData, Op->Header.Args[0]);
|
||||
float Src = *GetSrc<float *>(Data->SSAData, Op->X80Src);
|
||||
X80SoftFloat Tmp = Src;
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
double Src = *GetSrc<double *>(Data->SSAData, Op->Header.Args[0]);
|
||||
double Src = *GetSrc<double *>(Data->SSAData, Op->X80Src);
|
||||
X80SoftFloat Tmp = Src;
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
break;
|
||||
@@ -180,13 +171,13 @@ DEF_OP(F80CVTTOINT) {
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 2: {
|
||||
int16_t Src = *GetSrc<int16_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
int16_t Src = *GetSrc<int16_t*>(Data->SSAData, Op->Src);
|
||||
X80SoftFloat Tmp = Src;
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
int32_t Src = *GetSrc<int32_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
int32_t Src = *GetSrc<int32_t*>(Data->SSAData, Op->Src);
|
||||
X80SoftFloat Tmp = Src;
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
break;
|
||||
@@ -197,77 +188,73 @@ DEF_OP(F80CVTTOINT) {
|
||||
|
||||
DEF_OP(F80ROUND) {
|
||||
auto Op = IROp->C<IR::IROp_F80Round>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FRNDINT(Src);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FRNDINT(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80F2XM1) {
|
||||
auto Op = IROp->C<IR::IROp_F80F2XM1>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::F2XM1(Src);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::F2XM1(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80TAN) {
|
||||
auto Op = IROp->C<IR::IROp_F80TAN>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FTAN(Src);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FTAN(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80SQRT) {
|
||||
auto Op = IROp->C<IR::IROp_F80SQRT>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FSQRT(Src);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FSQRT(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80SIN) {
|
||||
auto Op = IROp->C<IR::IROp_F80SIN>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FSIN(Src);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FSIN(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80COS) {
|
||||
auto Op = IROp->C<IR::IROp_F80COS>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FCOS(Src);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FCOS(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80XTRACT_EXP) {
|
||||
auto Op = IROp->C<IR::IROp_F80XTRACT_EXP>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FXTRACT_EXP(Src);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FXTRACT_EXP(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80XTRACT_SIG) {
|
||||
auto Op = IROp->C<IR::IROp_F80XTRACT_SIG>();
|
||||
X80SoftFloat Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Tmp;
|
||||
Tmp = X80SoftFloat::FXTRACT_SIG(Src);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FXTRACT_SIG(Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80CMP) {
|
||||
auto Op = IROp->C<IR::IROp_F80Cmp>();
|
||||
uint32_t ResultFlags{};
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
bool eq, lt, nan;
|
||||
X80SoftFloat::FCMP(Src1, Src2, &eq, <, &nan);
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_LT) &&
|
||||
@@ -288,7 +275,7 @@ DEF_OP(F80CMP) {
|
||||
|
||||
DEF_OP(F80BCDLOAD) {
|
||||
auto Op = IROp->C<IR::IROp_F80BCDLoad>();
|
||||
uint8_t *Src1 = GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const uint8_t *Src1 = GetSrc<uint8_t*>(Data->SSAData, Op->X80Src);
|
||||
uint64_t BCD{};
|
||||
// We walk through each uint8_t and pull out the BCD encoding
|
||||
// Each 4bit split is a digit
|
||||
@@ -323,7 +310,7 @@ DEF_OP(F80BCDLOAD) {
|
||||
|
||||
DEF_OP(F80BCDSTORE) {
|
||||
auto Op = IROp->C<IR::IROp_F80BCDStore>();
|
||||
X80SoftFloat Src1 = X80SoftFloat::FRNDINT(*GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]));
|
||||
X80SoftFloat Src1 = X80SoftFloat::FRNDINT(*GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src));
|
||||
bool Negative = Src1.Sign;
|
||||
|
||||
// Clear the Sign bit
|
||||
@@ -358,74 +345,74 @@ DEF_OP(F80BCDSTORE) {
|
||||
|
||||
DEF_OP(F64SIN) {
|
||||
auto Op = IROp->C<IR::IROp_F64SIN>();
|
||||
double Src = *GetSrc<double*>(Data->SSAData, Op->Header.Args[0]);
|
||||
double Tmp = sin(Src);
|
||||
const double Src = *GetSrc<double*>(Data->SSAData, Op->Src);
|
||||
const double Tmp = sin(Src);
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64COS) {
|
||||
auto Op = IROp->C<IR::IROp_F64COS>();
|
||||
double Src = *GetSrc<double*>(Data->SSAData, Op->Header.Args[0]);
|
||||
double Tmp = cos(Src);
|
||||
const double Src = *GetSrc<double*>(Data->SSAData, Op->Src);
|
||||
const double Tmp = cos(Src);
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64TAN) {
|
||||
auto Op = IROp->C<IR::IROp_F64TAN>();
|
||||
double Src = *GetSrc<double*>(Data->SSAData, Op->Header.Args[0]);
|
||||
double Tmp = tan(Src);
|
||||
const double Src = *GetSrc<double*>(Data->SSAData, Op->Src);
|
||||
const double Tmp = tan(Src);
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64F2XM1) {
|
||||
auto Op = IROp->C<IR::IROp_F64F2XM1>();
|
||||
double Src = *GetSrc<double*>(Data->SSAData, Op->Header.Args[0]);
|
||||
double Tmp = exp2(Src) - 1.0;
|
||||
const double Src = *GetSrc<double*>(Data->SSAData, Op->Src);
|
||||
const double Tmp = exp2(Src) - 1.0;
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64ATAN) {
|
||||
auto Op = IROp->C<IR::IROp_F64ATAN>();
|
||||
double Src1 = *GetSrc<double*>(Data->SSAData, Op->Header.Args[0]);
|
||||
double Src2 = *GetSrc<double*>(Data->SSAData, Op->Header.Args[1]);
|
||||
double Tmp = atan2(Src1, Src2);
|
||||
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src1);
|
||||
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
|
||||
const double Tmp = atan2(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64FPREM) {
|
||||
auto Op = IROp->C<IR::IROp_F64FPREM>();
|
||||
double Src1 = *GetSrc<double*>(Data->SSAData, Op->Header.Args[0]);
|
||||
double Src2 = *GetSrc<double*>(Data->SSAData, Op->Header.Args[1]);
|
||||
double Tmp = fmod(Src1, Src2);
|
||||
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src1);
|
||||
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
|
||||
const double Tmp = fmod(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64FPREM1) {
|
||||
auto Op = IROp->C<IR::IROp_F64FPREM1>();
|
||||
double Src1 = *GetSrc<double*>(Data->SSAData, Op->Header.Args[0]);
|
||||
double Src2 = *GetSrc<double*>(Data->SSAData, Op->Header.Args[1]);
|
||||
double Tmp = remainder(Src1, Src2);
|
||||
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src1);
|
||||
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
|
||||
const double Tmp = remainder(Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64FYL2X) {
|
||||
auto Op = IROp->C<IR::IROp_F64FYL2X>();
|
||||
double Src1 = *GetSrc<double*>(Data->SSAData, Op->Header.Args[0]);
|
||||
double Src2 = *GetSrc<double*>(Data->SSAData, Op->Header.Args[1]);
|
||||
double Tmp = Src2 * log2(Src1);
|
||||
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src);
|
||||
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
|
||||
const double Tmp = Src2 * log2(Src1);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
DEF_OP(F64SCALE) {
|
||||
auto Op = IROp->C<IR::IROp_F64SCALE>();
|
||||
double Src1 = *GetSrc<double*>(Data->SSAData, Op->Header.Args[0]);
|
||||
double Src2 = *GetSrc<double*>(Data->SSAData, Op->Header.Args[1]);
|
||||
double trunc = (double)(int64_t)(Src2); //truncate
|
||||
double Tmp = Src1 * exp2(trunc);
|
||||
const double Src1 = *GetSrc<double*>(Data->SSAData, Op->Src1);
|
||||
const double Src2 = *GetSrc<double*>(Data->SSAData, Op->Src2);
|
||||
const double trunc = (double)(int64_t)(Src2); //truncate
|
||||
const double Tmp = Src1 * exp2(trunc);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(double));
|
||||
}
|
||||
|
||||
@@ -14,7 +14,7 @@ namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(GetHostFlag) {
|
||||
auto Op = IROp->C<IR::IROp_GetHostFlag>();
|
||||
GD = (*GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]) >> Op->Flag) & 1;
|
||||
GD = (*GetSrc<uint64_t*>(Data->SSAData, Op->Value) >> Op->Flag) & 1;
|
||||
}
|
||||
#undef DEF_OP
|
||||
|
||||
|
||||
@@ -8,6 +8,7 @@
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
class Dispatcher;
|
||||
class X86DispatchGenerator;
|
||||
class Arm64DispatchGenerator;
|
||||
|
||||
@@ -20,7 +21,7 @@ using DestMapType = std::vector<uint32_t>;
|
||||
|
||||
class InterpreterCore final : public CPUBackend {
|
||||
public:
|
||||
explicit InterpreterCore(FEXCore::Context::Context *ctx,
|
||||
explicit InterpreterCore(Dispatcher *Dispatch,
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
[[nodiscard]] std::string GetName() override { return "Interpreter"; }
|
||||
@@ -28,23 +29,19 @@ public:
|
||||
[[nodiscard]] void *CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) override;
|
||||
|
||||
[[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
|
||||
[[nodiscard]] bool NeedsOpDispatch() override { return true; }
|
||||
|
||||
void CreateAsmDispatch(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
static void InitializeSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
|
||||
bool NeedsRetainedIRCopy() const override { return true; }
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
private:
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::Core::InternalThreadState *State;
|
||||
|
||||
std::unique_ptr<Dispatcher> Dispatcher{};
|
||||
size_t BufferUsed;
|
||||
Dispatcher *Dispatch;
|
||||
};
|
||||
|
||||
template<typename T>
|
||||
|
||||
@@ -9,70 +9,101 @@
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
#include <memory>
|
||||
#include <signal.h>
|
||||
#include <stdint.h>
|
||||
#include <unordered_map>
|
||||
#include <utility>
|
||||
|
||||
#include "InterpreterOps.h"
|
||||
|
||||
#if defined(_M_X86_64)
|
||||
#include "Interface/Core/Dispatcher/X86Dispatcher.h"
|
||||
#elif defined(_M_ARM_64)
|
||||
#include "Interface/Core/Dispatcher/Arm64Dispatcher.h"
|
||||
#else
|
||||
#error missing arch
|
||||
#endif
|
||||
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class IRListView;
|
||||
class RegisterAllocationData;
|
||||
}
|
||||
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
class CPUBackend;
|
||||
|
||||
static void InterpreterExecution(FEXCore::Core::CpuStateFrame *Frame) {
|
||||
auto Thread = Frame->Thread;
|
||||
InterpreterCore::InterpreterCore(Dispatcher *Dispatcher, FEXCore::Core::InternalThreadState *Thread)
|
||||
: CPUBackend(Thread, INITIAL_CODE_SIZE, MAX_CODE_SIZE)
|
||||
, Dispatch(Dispatcher)
|
||||
{
|
||||
|
||||
auto LocalEntry = Thread->LocalIRCache.find(Thread->CurrentFrame->State.rip);
|
||||
auto &Interpreter = Thread->CurrentFrame->Pointers.Interpreter;
|
||||
|
||||
InterpreterOps::InterpretIR(Thread, Thread->CurrentFrame->State.rip, LocalEntry->second.IR.get(), LocalEntry->second.DebugData.get());
|
||||
}
|
||||
Interpreter.FragmentExecuter = reinterpret_cast<uint64_t>(&InterpreterOps::InterpretIR);
|
||||
|
||||
InterpreterCore::InterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
: CTX {ctx}
|
||||
, State {Thread} {
|
||||
|
||||
CreateAsmDispatch(ctx, Thread);
|
||||
ClearCache();
|
||||
}
|
||||
|
||||
void InterpreterCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SignalDelegator::SIGNAL_FOR_PAUSE, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
InterpreterCore *Core = reinterpret_cast<InterpreterCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleSignalPause(Signal, info, ucontext);
|
||||
}, true);
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
return FEXCore::ArchHelpers::Arm64::HandleSIGBUS(true, Signal, info, ucontext);
|
||||
}, true);
|
||||
#endif
|
||||
|
||||
auto GuestSignalHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) -> bool {
|
||||
InterpreterCore *Core = reinterpret_cast<InterpreterCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleGuestSignal(Signal, info, ucontext, GuestAction, GuestStack);
|
||||
};
|
||||
|
||||
for (uint32_t Signal = 0; Signal <= SignalDelegator::MAX_SIGNALS; ++Signal) {
|
||||
CTX->SignalDelegation->RegisterHostSignalHandlerForGuest(Signal, GuestSignalHandler);
|
||||
}
|
||||
}
|
||||
|
||||
void *InterpreterCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
return reinterpret_cast<void*>(InterpreterExecution);
|
||||
void *InterpreterCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) {
|
||||
|
||||
const auto IRSize = AlignUp(IR->GetInlineSize(), 16);
|
||||
const auto MaxSize = IRSize + Dispatcher::MaxInterpreterTrampolineSize + GDBEnabled * Dispatcher::MaxGDBPauseCheckSize;
|
||||
|
||||
if ((BufferUsed + MaxSize) > CurrentCodeBuffer->Size) {
|
||||
ThreadState->CTX->ClearCodeCache(ThreadState);
|
||||
}
|
||||
|
||||
const auto BufferStart = CurrentCodeBuffer->Ptr + BufferUsed;
|
||||
|
||||
auto DestBuffer = BufferStart;
|
||||
|
||||
if (GDBEnabled) {
|
||||
const auto GDBSize = Dispatch->GenerateGDBPauseCheck(DestBuffer, Entry);
|
||||
DestBuffer += GDBSize;
|
||||
BufferUsed += GDBSize;
|
||||
}
|
||||
|
||||
const auto TrampolineSize = Dispatch->GenerateInterpreterTrampoline(DestBuffer);
|
||||
DestBuffer += TrampolineSize;
|
||||
BufferUsed += TrampolineSize;
|
||||
|
||||
|
||||
IR->Serialize(DestBuffer);
|
||||
DestBuffer += IRSize;
|
||||
BufferUsed += IRSize;
|
||||
|
||||
return BufferStart;
|
||||
}
|
||||
|
||||
void InterpreterCore::ClearCache() {
|
||||
// Calling this one is needed to setup the initial CurrentCodeBuffer
|
||||
[[maybe_unused]] auto CodeBuffer = GetEmptyCodeBuffer();
|
||||
BufferUsed = 0;
|
||||
}
|
||||
|
||||
std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return std::make_unique<InterpreterCore>(ctx, Thread);
|
||||
return std::make_unique<InterpreterCore>(ctx->Dispatcher.get(), Thread);
|
||||
}
|
||||
|
||||
void InitializeInterpreterSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
InterpreterCore::InitializeSignalHandlers(CTX);
|
||||
}
|
||||
|
||||
CPUBackendFeatures GetInterpreterBackendFeatures() {
|
||||
return CPUBackendFeatures { };
|
||||
}
|
||||
}
|
||||
@@ -12,10 +12,11 @@ namespace FEXCore::Core {
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
class CPUBackend;
|
||||
struct DispatcherConfig;
|
||||
|
||||
[[nodiscard]] std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
void InitializeInterpreterSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
CPUBackendFeatures GetInterpreterBackendFeatures();
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -112,8 +112,6 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(ATOMICFETCHNEG, AtomicFetchNeg);
|
||||
|
||||
// Branch ops
|
||||
REGISTER_OP(GUESTCALLDIRECT, GuestCallDirect);
|
||||
REGISTER_OP(GUESTCALLINDIRECT, GuestCallIndirect);
|
||||
REGISTER_OP(SIGNALRETURN, SignalReturn);
|
||||
REGISTER_OP(CALLBACKRETURN, CallbackReturn);
|
||||
REGISTER_OP(EXITFUNCTION, ExitFunction);
|
||||
@@ -123,7 +121,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(INLINESYSCALL, InlineSyscall);
|
||||
REGISTER_OP(THUNK, Thunk);
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(REMOVETHREADCODEENTRY, RemoveThreadCodeEntry);
|
||||
REGISTER_OP(THREADREMOVECODEENTRY, ThreadRemoveCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
|
||||
// Conversion ops
|
||||
@@ -166,6 +164,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(CODEBLOCK, NoOp);
|
||||
REGISTER_OP(BEGINBLOCK, NoOp);
|
||||
REGISTER_OP(ENDBLOCK, NoOp);
|
||||
REGISTER_OP(GUESTOPCODE, NoOp);
|
||||
REGISTER_OP(FENCE, Fence);
|
||||
REGISTER_OP(BREAK, Break);
|
||||
REGISTER_OP(PHI, NoOp);
|
||||
@@ -284,6 +283,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(VAESDECLAST, AESDecLast);
|
||||
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
|
||||
REGISTER_OP(CRC32, CRC32);
|
||||
REGISTER_OP(PCLMUL, PCLMUL);
|
||||
|
||||
// F80 ops
|
||||
REGISTER_OP(F80LOADFCW, F80LOADFCW);
|
||||
@@ -333,7 +333,7 @@ void InterpreterOps::Op_Unhandled(FEXCore::IR::IROp_Header *IROp, IROpData *Data
|
||||
void InterpreterOps::Op_NoOp(FEXCore::IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node) {
|
||||
}
|
||||
|
||||
void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, uint64_t Entry, FEXCore::IR::IRListView *CurrentIR, FEXCore::Core::DebugData *DebugData) {
|
||||
void InterpreterOps::InterpretIR(FEXCore::Core::CpuStateFrame *Frame, FEXCore::IR::IRListView const *CurrentIR) {
|
||||
volatile void *StackEntry = alloca(0);
|
||||
|
||||
uintptr_t ListSize = CurrentIR->GetSSACount();
|
||||
@@ -344,9 +344,9 @@ void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, uin
|
||||
auto BlockEnd = CurrentIR->GetBlocks().end();
|
||||
|
||||
InterpreterOps::IROpData OpData{};
|
||||
OpData.State = Thread;
|
||||
OpData.State = Frame->Thread;
|
||||
OpData.SSAData = alloca(ListSize * 16);
|
||||
OpData.CurrentEntry = Entry;
|
||||
OpData.CurrentEntry = Frame->State.rip;
|
||||
OpData.CurrentIR = CurrentIR;
|
||||
OpData.StackEntry = StackEntry;
|
||||
OpData.BlockIterator = CurrentIR->GetBlocks().begin();
|
||||
@@ -358,7 +358,7 @@ void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, uin
|
||||
using namespace FEXCore::IR;
|
||||
auto [BlockNode, BlockHeader] = OpData.BlockIterator();
|
||||
auto BlockIROp = BlockHeader->CW<IROp_CodeBlock>();
|
||||
LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
LOGMAN_THROW_AA_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
|
||||
// Reset the block results per block
|
||||
memset(&OpData.BlockResults, 0, sizeof(OpData.BlockResults));
|
||||
|
||||
@@ -47,14 +47,14 @@ namespace FEXCore::CPU {
|
||||
class InterpreterOps {
|
||||
|
||||
public:
|
||||
static void InterpretIR(FEXCore::Core::InternalThreadState *Thread, uint64_t Entry, FEXCore::IR::IRListView *CurrentIR, FEXCore::Core::DebugData *DebugData);
|
||||
static void InterpretIR(FEXCore::Core::CpuStateFrame *Frame, FEXCore::IR::IRListView const *IR);
|
||||
static void FillFallbackIndexPointers(uint64_t *Info);
|
||||
static bool GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Info);
|
||||
|
||||
struct IROpData {
|
||||
FEXCore::Core::InternalThreadState *State{};
|
||||
uint64_t CurrentEntry{};
|
||||
FEXCore::IR::IRListView *CurrentIR{};
|
||||
FEXCore::IR::IRListView const *CurrentIR{};
|
||||
volatile void *StackEntry{};
|
||||
void *SSAData{};
|
||||
struct {
|
||||
@@ -142,8 +142,6 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(AtomicFetchNeg);
|
||||
|
||||
///< Branch ops
|
||||
DEF_OP(GuestCallDirect);
|
||||
DEF_OP(GuestCallIndirect);
|
||||
DEF_OP(SignalReturn);
|
||||
DEF_OP(CallbackReturn);
|
||||
DEF_OP(ExitFunction);
|
||||
@@ -153,7 +151,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(InlineSyscall);
|
||||
DEF_OP(Thunk);
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(RemoveThreadCodeEntry);
|
||||
DEF_OP(ThreadRemoveCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
|
||||
///< Conversion ops
|
||||
@@ -304,6 +302,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(AESDecLast);
|
||||
DEF_OP(AESKeyGenAssist);
|
||||
DEF_OP(CRC32);
|
||||
DEF_OP(PCLMUL);
|
||||
|
||||
///< F80 ops
|
||||
DEF_OP(F80LOADFCW);
|
||||
@@ -408,7 +407,7 @@ namespace FEXCore::CPU {
|
||||
return CompResult;
|
||||
}
|
||||
|
||||
static uint8_t GetOpSize(FEXCore::IR::IRListView *CurrentIR, IR::OrderedNodeWrapper Node) {
|
||||
static uint8_t GetOpSize(FEXCore::IR::IRListView const *CurrentIR, IR::OrderedNodeWrapper Node) {
|
||||
auto IROp = CurrentIR->GetOp<FEXCore::IR::IROp_Header>(Node);
|
||||
return IROp->Size;
|
||||
}
|
||||
|
||||
@@ -4,6 +4,7 @@ tags: backend|interpreter
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
@@ -58,7 +59,7 @@ DEF_OP(StoreContext) {
|
||||
ContextPtr += Op->Offset;
|
||||
|
||||
void *MemData = reinterpret_cast<void*>(ContextPtr);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Value);
|
||||
memcpy(MemData, Src, OpSize);
|
||||
}
|
||||
|
||||
@@ -72,7 +73,7 @@ DEF_OP(StoreRegister) {
|
||||
|
||||
DEF_OP(LoadContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
uint64_t Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
|
||||
@@ -102,14 +103,14 @@ DEF_OP(LoadContextIndexed) {
|
||||
|
||||
DEF_OP(StoreContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
uint64_t Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[1]);
|
||||
uint64_t Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += Op->BaseOffset;
|
||||
ContextPtr += Index * Op->Stride;
|
||||
|
||||
void *MemData = reinterpret_cast<void*>(ContextPtr);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Value);
|
||||
memcpy(MemData, Src, IROp->Size);
|
||||
}
|
||||
|
||||
@@ -133,7 +134,7 @@ DEF_OP(LoadFlag) {
|
||||
|
||||
DEF_OP(StoreFlag) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
uint8_t Arg = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint8_t Arg = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += offsetof(FEXCore::Core::CPUState, flags[0]);
|
||||
@@ -227,9 +228,9 @@ DEF_OP(StoreMem) {
|
||||
|
||||
DEF_OP(VLoadMemElement) {
|
||||
auto Op = IROp->C<IR::IROp_VLoadMemElement>();
|
||||
void const *MemData = *GetSrc<void const**>(Data->SSAData, Op->Header.Args[0]);
|
||||
void const *MemData = *GetSrc<void const**>(Data->SSAData, Op->Value);
|
||||
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Header.Args[1]), 16);
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Addr), 16);
|
||||
memcpy(reinterpret_cast<void*>(reinterpret_cast<uintptr_t>(GDP) + (Op->Header.ElementSize * Op->Index)),
|
||||
MemData, Op->Header.ElementSize);
|
||||
}
|
||||
@@ -237,8 +238,8 @@ DEF_OP(VLoadMemElement) {
|
||||
DEF_OP(VStoreMemElement) {
|
||||
#define STORE_DATA(x, y) \
|
||||
case x: { \
|
||||
y *MemData = *GetSrc<y**>(Data->SSAData, Op->Header.Args[0]); \
|
||||
memcpy(MemData, &GetSrc<y*>(Data->SSAData, Op->Header.Args[1])[Op->Index], sizeof(y)); \
|
||||
y *MemData = *GetSrc<y**>(Data->SSAData, Op->Value); \
|
||||
memcpy(MemData, &GetSrc<y*>(Data->SSAData, Op->Addr)[Op->Index], sizeof(y)); \
|
||||
break; \
|
||||
}
|
||||
|
||||
|
||||
+26
-12
@@ -4,6 +4,8 @@ tags: backend|interpreter
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterClass.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
@@ -44,14 +46,26 @@ DEF_OP(Fence) {
|
||||
|
||||
DEF_OP(Break) {
|
||||
auto Op = IROp->C<IR::IROp_Break>();
|
||||
switch (Op->Reason) {
|
||||
case FEXCore::IR::Break_Halt: // HLT
|
||||
StopThread(Data->State);
|
||||
|
||||
Data->State->CurrentFrame->SynchronousFaultData.FaultToTopAndGeneratedException = 1;
|
||||
Data->State->CurrentFrame->SynchronousFaultData.Signal = Op->Reason.Signal;
|
||||
Data->State->CurrentFrame->SynchronousFaultData.TrapNo = Op->Reason.TrapNumber;
|
||||
Data->State->CurrentFrame->SynchronousFaultData.err_code = Op->Reason.ErrorRegister;
|
||||
Data->State->CurrentFrame->SynchronousFaultData.si_code = Op->Reason.si_code;
|
||||
|
||||
switch (Op->Reason.Signal) {
|
||||
case SIGILL:
|
||||
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGILL);
|
||||
break;
|
||||
case FEXCore::IR::Break_InvalidInstruction:
|
||||
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGILL);
|
||||
case SIGTRAP:
|
||||
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGTRAP);
|
||||
break;
|
||||
case SIGSEGV:
|
||||
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGSEGV);
|
||||
break;
|
||||
default:
|
||||
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGTRAP);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Break Reason: {}", Op->Reason); break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -86,7 +100,7 @@ DEF_OP(GetRoundingMode) {
|
||||
|
||||
DEF_OP(SetRoundingMode) {
|
||||
auto Op = IROp->C<IR::IROp_SetRoundingMode>();
|
||||
uint8_t GuestRounding = *GetSrc<uint8_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto GuestRounding = *GetSrc<uint8_t*>(Data->SSAData, Op->RoundMode);
|
||||
#ifdef _M_ARM_64
|
||||
uint64_t HostRounding{};
|
||||
__asm volatile(R"(
|
||||
@@ -126,16 +140,16 @@ DEF_OP(SetRoundingMode) {
|
||||
|
||||
DEF_OP(Print) {
|
||||
auto Op = IROp->C<IR::IROp_Print>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
if (OpSize <= 8) {
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Value);
|
||||
LogMan::Msg::IFmt(">>>> Value in Arg: 0x{:x}, {}", Src, Src);
|
||||
}
|
||||
else if (OpSize == 16) {
|
||||
__uint128_t Src = *GetSrc<__uint128_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t Src0 = Src;
|
||||
uint64_t Src1 = Src >> 64;
|
||||
const auto Src = *GetSrc<__uint128_t*>(Data->SSAData, Op->Value);
|
||||
const uint64_t Src0 = Src;
|
||||
const uint64_t Src1 = Src >> 64;
|
||||
LogMan::Msg::IFmt(">>>> Value[0] in Arg: 0x{:x}, {}", Src0, Src0);
|
||||
LogMan::Msg::IFmt(" Value[1] in Arg: 0x{:x}, {}", Src1, Src1);
|
||||
}
|
||||
|
||||
@@ -14,15 +14,15 @@ namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(ExtractElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_ExtractElementPair>();
|
||||
uintptr_t Src = GetSrc<uintptr_t>(Data->SSAData, Op->Header.Args[0]);
|
||||
const auto Src = GetSrc<uintptr_t>(Data->SSAData, Op->Pair);
|
||||
memcpy(GDP,
|
||||
reinterpret_cast<void*>(Src + Op->Header.Size * Op->Element), Op->Header.Size);
|
||||
}
|
||||
|
||||
DEF_OP(CreateElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_CreateElementPair>();
|
||||
void *Src_Lower = GetSrc<void*>(Data->SSAData, Op->Header.Args[0]);
|
||||
void *Src_Upper = GetSrc<void*>(Data->SSAData, Op->Header.Args[1]);
|
||||
const void *Src_Lower = GetSrc<void*>(Data->SSAData, Op->Lower);
|
||||
const void *Src_Upper = GetSrc<void*>(Data->SSAData, Op->Upper);
|
||||
|
||||
uint8_t *Dst = GetDest<uint8_t*>(Data->SSAData, Node);
|
||||
|
||||
@@ -32,9 +32,9 @@ DEF_OP(CreateElementPair) {
|
||||
|
||||
DEF_OP(Mov) {
|
||||
auto Op = IROp->C<IR::IROp_Mov>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Header.Args[0]), OpSize);
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Value), OpSize);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
+395
-374
File diff suppressed because it is too large.
Load diff
+245
-207
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,131 @@
|
||||
/*
|
||||
$info$
|
||||
tags: backend|arm64
|
||||
desc: relocation logic of the arm64 splatter backend
|
||||
$end_info$
|
||||
*/
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
uint64_t Arm64JITCore::GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
switch (Op) {
|
||||
case FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER:
|
||||
return ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker;
|
||||
break;
|
||||
default:
|
||||
ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op));
|
||||
break;
|
||||
}
|
||||
return ~0ULL;
|
||||
}
|
||||
|
||||
void Arm64JITCore::InsertNamedThunkRelocation(vixl::aarch64::Register Reg, const IR::SHA256Sum &Sum) {
|
||||
Relocation MoveABI{};
|
||||
MoveABI.NamedThunkMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE;
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t *>();
|
||||
MoveABI.NamedThunkMove.Offset = CurrentCursor - GuestEntry;
|
||||
MoveABI.NamedThunkMove.Symbol = Sum;
|
||||
MoveABI.NamedThunkMove.RegisterIndex = Reg.GetCode();
|
||||
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Sum));
|
||||
|
||||
LoadConstant(Reg, Pointer, EmitterCTX->Config.CacheObjectCodeCompilation());
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
Arm64JITCore::NamedSymbolLiteralPair Arm64JITCore::InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Op);
|
||||
|
||||
Arm64JITCore::NamedSymbolLiteralPair Lit {
|
||||
.Lit = Literal(Pointer),
|
||||
.MoveABI = {
|
||||
.NamedSymbolLiteral = {
|
||||
.Header = {
|
||||
.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL,
|
||||
},
|
||||
.Symbol = Op,
|
||||
.Offset = 0,
|
||||
},
|
||||
},
|
||||
};
|
||||
return Lit;
|
||||
}
|
||||
|
||||
void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit) {
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t *>();
|
||||
Lit.MoveABI.NamedSymbolLiteral.Offset = CurrentCursor - GuestEntry;
|
||||
|
||||
place(&Lit.Lit);
|
||||
Relocations.emplace_back(Lit.MoveABI);
|
||||
}
|
||||
|
||||
void Arm64JITCore::InsertGuestRIPMove(vixl::aarch64::Register Reg, uint64_t Constant) {
|
||||
Relocation MoveABI{};
|
||||
MoveABI.GuestRIPMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE;
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t *>();
|
||||
MoveABI.GuestRIPMove.Offset = CurrentCursor - GuestEntry;
|
||||
MoveABI.GuestRIPMove.GuestRIP = Constant;
|
||||
MoveABI.GuestRIPMove.RegisterIndex = Reg.GetCode();
|
||||
|
||||
LoadConstant(Reg, Constant, EmitterCTX->Config.CacheObjectCodeCompilation());
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations) {
|
||||
size_t DataIndex{};
|
||||
for (size_t j = 0; j < NumRelocations; ++j) {
|
||||
const FEXCore::CPU::Relocation *Reloc = reinterpret_cast<const FEXCore::CPU::Relocation *>(&EntryRelocations[DataIndex]);
|
||||
LOGMAN_THROW_AA_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
|
||||
|
||||
switch (Reloc->Header.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Reloc->NamedSymbolLiteral.Symbol);
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor
|
||||
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->NamedSymbolLiteral.Offset);
|
||||
|
||||
// Generate a literal so we can place it
|
||||
Literal<uint64_t> Lit(Pointer);
|
||||
place(&Lit);
|
||||
|
||||
DataIndex += sizeof(Reloc->NamedSymbolLiteral);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc->NamedThunkMove.Symbol));
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->NamedThunkMove.Offset);
|
||||
LoadConstant(vixl::aarch64::XRegister(Reloc->NamedThunkMove.RegisterIndex), Pointer, true);
|
||||
DataIndex += sizeof(Reloc->NamedThunkMove);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
|
||||
// XXX: Reenable once the JIT Object Cache is upstream
|
||||
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
|
||||
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->GuestRIPMove.Offset);
|
||||
LoadConstant(vixl::aarch64::XRegister(Reloc->GuestRIPMove.RegisterIndex), Pointer, true);
|
||||
DataIndex += sizeof(Reloc->GuestRIPMove);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4,6 +4,7 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
+21
-29
@@ -4,6 +4,7 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
@@ -19,13 +20,6 @@ namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(GuestCallDirect) {
|
||||
LogMan::Msg::DFmt("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(GuestCallIndirect) {
|
||||
LogMan::Msg::DFmt("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(SignalReturn) {
|
||||
// First we must reset the stack
|
||||
@@ -33,7 +27,7 @@ DEF_OP(SignalReturn) {
|
||||
|
||||
// Now branch to our signal return helper
|
||||
// This can't be a direct branch since the code needs to live at a constant location
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.SignalReturnHandler)));
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SignalReturnHandler)));
|
||||
br(x0);
|
||||
}
|
||||
|
||||
@@ -46,10 +40,10 @@ DEF_OP(CallbackReturn) {
|
||||
ResetStack();
|
||||
|
||||
// We can now lower the ref counter again
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.SignalHandlerRefCountPointer)));
|
||||
ldr(w2, MemOperand(x0));
|
||||
|
||||
ldr(w2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter)));
|
||||
sub(w2, w2, 1);
|
||||
str(w2, MemOperand(x0));
|
||||
str(w2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter)));
|
||||
|
||||
// We need to adjust an additional 8 bytes to get back to the original "misaligned" RSP state
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])));
|
||||
@@ -73,7 +67,7 @@ DEF_OP(ExitFunction) {
|
||||
uint64_t NewRIP;
|
||||
|
||||
if (IsInlineConstant(Op->NewRIP, &NewRIP) || IsInlineEntrypointOffset(Op->NewRIP, &NewRIP)) {
|
||||
Literal l_BranchHost{Dispatcher->ExitFunctionLinkerAddress};
|
||||
Literal l_BranchHost{ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker};
|
||||
Literal l_BranchGuest{NewRIP};
|
||||
|
||||
ldr(x0, &l_BranchHost);
|
||||
@@ -82,10 +76,10 @@ DEF_OP(ExitFunction) {
|
||||
place(&l_BranchHost);
|
||||
place(&l_BranchGuest);
|
||||
} else {
|
||||
RipReg = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
RipReg = GetReg<RA_64>(Op->NewRIP.ID());
|
||||
|
||||
// L1 Cache
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.L1Pointer)));
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer)));
|
||||
|
||||
and_(x3, RipReg, LookupCache::L1_ENTRIES_MASK);
|
||||
add(x0, x0, Operand(x3, Shift::LSL, 4));
|
||||
@@ -96,7 +90,7 @@ DEF_OP(ExitFunction) {
|
||||
br(x1);
|
||||
|
||||
bind(&FullLookup);
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.DispatcherLoopTop)));
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop)));
|
||||
str(RipReg, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
|
||||
br(TMP1);
|
||||
}
|
||||
@@ -104,9 +98,9 @@ DEF_OP(ExitFunction) {
|
||||
|
||||
DEF_OP(Jump) {
|
||||
const auto Op = IROp->C<IR::IROp_Jump>();
|
||||
const auto ArgID = Op->Args(0).ID();
|
||||
const auto Target = Op->TargetBlock.ID();
|
||||
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(ArgID).first->second;
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(Target).first->second;
|
||||
}
|
||||
|
||||
#define GRCMP(Node) (Op->CompareSize == 4 ? GetReg<RA_32>(Node) : GetReg<RA_64>(Node))
|
||||
@@ -199,8 +193,8 @@ DEF_OP(Syscall) {
|
||||
str(GetReg<RA_64>(Op->Header.Args[i].ID()), MemOperand(sp, i * 8));
|
||||
}
|
||||
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.SyscallHandlerObj)));
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.SyscallHandlerFunc)));
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SyscallHandlerObj)));
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SyscallHandlerFunc)));
|
||||
mov(x1, STATE);
|
||||
mov(x2, sp);
|
||||
blr(x3);
|
||||
@@ -383,7 +377,7 @@ DEF_OP(Thunk) {
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
mov(x0, GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
mov(x0, GetReg<RA_64>(Op->ArgPtr.ID()));
|
||||
|
||||
auto thunkFn = ThreadState->CTX->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
LoadConstant(x2, (uintptr_t)thunkFn);
|
||||
@@ -442,7 +436,7 @@ DEF_OP(ValidateCode) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(RemoveThreadCodeEntry) {
|
||||
DEF_OP(ThreadRemoveCodeEntry) {
|
||||
// Arguments are passed as follows:
|
||||
// X0: Thread
|
||||
// X1: RIP
|
||||
@@ -452,7 +446,7 @@ DEF_OP(RemoveThreadCodeEntry) {
|
||||
mov(x0, STATE);
|
||||
LoadConstant(x1, Entry);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.RemoveThreadCodeEntryFromJIT)));
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.ThreadRemoveCodeEntryFromJIT)));
|
||||
SpillStaticRegs();
|
||||
blr(x2);
|
||||
FillStaticRegs();
|
||||
@@ -469,10 +463,10 @@ DEF_OP(CPUID) {
|
||||
// x0 = CPUID Handler
|
||||
// x1 = CPUID Function
|
||||
// x2 = CPUID Leaf
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.CPUIDObj)));
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.CPUIDFunction)));
|
||||
mov(x1, GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
mov(x2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj)));
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDFunction)));
|
||||
mov(x1, GetReg<RA_64>(Op->Function.ID()));
|
||||
mov(x2, GetReg<RA_64>(Op->Leaf.ID()));
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
FillStaticRegs();
|
||||
@@ -489,8 +483,6 @@ DEF_OP(CPUID) {
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterBranchHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
REGISTER_OP(GUESTCALLDIRECT, GuestCallDirect);
|
||||
REGISTER_OP(GUESTCALLINDIRECT, GuestCallIndirect);
|
||||
REGISTER_OP(SIGNALRETURN, SignalReturn);
|
||||
REGISTER_OP(CALLBACKRETURN, CallbackReturn);
|
||||
REGISTER_OP(EXITFUNCTION, ExitFunction);
|
||||
@@ -500,7 +492,7 @@ void Arm64JITCore::RegisterBranchHandlers() {
|
||||
REGISTER_OP(INLINESYSCALL, InlineSyscall);
|
||||
REGISTER_OP(THUNK, Thunk);
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(REMOVETHREADCODEENTRY, RemoveThreadCodeEntry);
|
||||
REGISTER_OP(THREADREMOVECODEENTRY, ThreadRemoveCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
|
||||
@@ -13,22 +13,22 @@ using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(VInsGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
mov(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
mov(GetDst(Node), GetSrc(Op->DestVector.ID()));
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1: {
|
||||
ins(GetDst(Node).V16B(), Op->DestIdx, GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
ins(GetDst(Node).V16B(), Op->DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
ins(GetDst(Node).V8H(), Op->DestIdx, GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
ins(GetDst(Node).V8H(), Op->DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
ins(GetDst(Node).V4S(), Op->DestIdx, GetReg<RA_32>(Op->Header.Args[1].ID()));
|
||||
ins(GetDst(Node).V4S(), Op->DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
ins(GetDst(Node).V2D(), Op->DestIdx, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
ins(GetDst(Node).V2D(), Op->DestIdx, GetReg<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
@@ -39,18 +39,18 @@ DEF_OP(VCastFromGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VCastFromGPR>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1:
|
||||
uxtb(TMP1.W(), GetReg<RA_32>(Op->Header.Args[0].ID()));
|
||||
uxtb(TMP1.W(), GetReg<RA_32>(Op->Src.ID()));
|
||||
fmov(GetDst(Node).S(), TMP1.W());
|
||||
break;
|
||||
case 2:
|
||||
uxth(TMP1.W(), GetReg<RA_32>(Op->Header.Args[0].ID()));
|
||||
uxth(TMP1.W(), GetReg<RA_32>(Op->Src.ID()));
|
||||
fmov(GetDst(Node).S(), TMP1.W());
|
||||
break;
|
||||
case 4:
|
||||
fmov(GetDst(Node).S(), GetReg<RA_32>(Op->Header.Args[0].ID()).W());
|
||||
fmov(GetDst(Node).S(), GetReg<RA_32>(Op->Src.ID()).W());
|
||||
break;
|
||||
case 8:
|
||||
fmov(GetDst(Node).D(), GetReg<RA_64>(Op->Header.Args[0].ID()).X());
|
||||
fmov(GetDst(Node).D(), GetReg<RA_64>(Op->Src.ID()).X());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown castGPR element size: {}", Op->Header.ElementSize);
|
||||
}
|
||||
@@ -58,22 +58,22 @@ DEF_OP(VCastFromGPR) {
|
||||
|
||||
DEF_OP(Float_FromGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0404: { // Float <- int32_t
|
||||
scvtf(GetDst(Node).S(), GetReg<RA_32>(Op->Header.Args[0].ID()));
|
||||
scvtf(GetDst(Node).S(), GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- int64_t
|
||||
scvtf(GetDst(Node).S(), GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
scvtf(GetDst(Node).S(), GetReg<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- int32_t
|
||||
scvtf(GetDst(Node).D(), GetReg<RA_32>(Op->Header.Args[0].ID()));
|
||||
scvtf(GetDst(Node).D(), GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 0x0808: { // Double <- int64_t
|
||||
scvtf(GetDst(Node).D(), GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
scvtf(GetDst(Node).D(), GetReg<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
}
|
||||
@@ -81,14 +81,14 @@ DEF_OP(Float_FromGPR_S) {
|
||||
|
||||
DEF_OP(Float_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FToF>();
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- Float
|
||||
fcvt(GetDst(Node).D(), GetSrc(Op->Header.Args[0].ID()).S());
|
||||
fcvt(GetDst(Node).D(), GetSrc(Op->Scalar.ID()).S());
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvt(GetDst(Node).S(), GetSrc(Op->Header.Args[0].ID()).D());
|
||||
fcvt(GetDst(Node).S(), GetSrc(Op->Scalar.ID()).D());
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown FCVT sizes: 0x{:x}", Conv);
|
||||
@@ -99,10 +99,10 @@ DEF_OP(Vector_SToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
scvtf(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
scvtf(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
scvtf(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
scvtf(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", Op->Header.ElementSize);
|
||||
}
|
||||
@@ -112,10 +112,10 @@ DEF_OP(Vector_FToZS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
fcvtzs(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
fcvtzs(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
fcvtzs(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
fcvtzs(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", Op->Header.ElementSize);
|
||||
}
|
||||
@@ -125,11 +125,11 @@ DEF_OP(Vector_FToS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frinti(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
frinti(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
fcvtzs(GetDst(Node).V4S(), GetDst(Node).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frinti(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
frinti(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
fcvtzs(GetDst(Node).V2D(), GetDst(Node).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", Op->Header.ElementSize);
|
||||
@@ -142,11 +142,11 @@ DEF_OP(Vector_FToF) {
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- Float
|
||||
fcvtl(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2S());
|
||||
fcvtl(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2S());
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvtn(GetDst(Node).V2S(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
fcvtn(GetDst(Node).V2S(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv); break;
|
||||
@@ -159,50 +159,50 @@ DEF_OP(Vector_FToI) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintn(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
frintn(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintn(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
frintn(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintm(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
frintm(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintm(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
frintm(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintp(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
frintp(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintp(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
frintp(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintz(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
frintz(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintz(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
frintz(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frinti(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S());
|
||||
frinti(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frinti(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D());
|
||||
frinti(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
|
||||
@@ -14,41 +14,41 @@ using namespace vixl::aarch64;
|
||||
|
||||
DEF_OP(AESImc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESImc>();
|
||||
aesimc(GetDst(Node).V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
|
||||
aesimc(GetDst(Node).V16B(), GetSrc(Op->Vector.ID()).V16B());
|
||||
}
|
||||
|
||||
DEF_OP(AESEnc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
|
||||
aese(VTMP1.V16B(), VTMP2.V16B());
|
||||
aesmc(VTMP1.V16B(), VTMP1.V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Header.Args[1].ID()).V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Key.ID()).V16B());
|
||||
}
|
||||
|
||||
DEF_OP(AESEncLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
|
||||
aese(VTMP1.V16B(), VTMP2.V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Header.Args[1].ID()).V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Key.ID()).V16B());
|
||||
}
|
||||
|
||||
DEF_OP(AESDec) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
|
||||
aesd(VTMP1.V16B(), VTMP2.V16B());
|
||||
aesimc(VTMP1.V16B(), VTMP1.V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Header.Args[1].ID()).V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Key.ID()).V16B());
|
||||
}
|
||||
|
||||
DEF_OP(AESDecLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->State.ID()).V16B());
|
||||
aesd(VTMP1.V16B(), VTMP2.V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Header.Args[1].ID()).V16B());
|
||||
eor(GetDst(Node).V16B(), VTMP1.V16B(), GetSrc(Op->Key.ID()).V16B());
|
||||
}
|
||||
|
||||
DEF_OP(AESKeyGenAssist) {
|
||||
@@ -59,7 +59,7 @@ DEF_OP(AESKeyGenAssist) {
|
||||
|
||||
// Do a "regular" AESE step
|
||||
eor(VTMP2.V16B(), VTMP2.V16B(), VTMP2.V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->Header.Args[0].ID()).V16B());
|
||||
mov(VTMP1.V16B(), GetSrc(Op->Src.ID()).V16B());
|
||||
aese(VTMP1.V16B(), VTMP2.V16B());
|
||||
|
||||
// Do a table shuffle to undo ShiftRows
|
||||
@@ -102,16 +102,45 @@ DEF_OP(CRC32) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(PCLMUL) {
|
||||
auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
|
||||
auto Dst = GetDst(Node).Q();
|
||||
auto Src1 = GetSrc(Op->Src1.ID()).V2D();
|
||||
auto Src2 = GetSrc(Op->Src2.ID()).V2D();
|
||||
|
||||
switch (Op->Selector) {
|
||||
case 0b00000000:
|
||||
pmull(Dst, Src1, Src2);
|
||||
break;
|
||||
case 0b00000001:
|
||||
mov(VTMP1.V1D(), Src1, 1);
|
||||
pmull(Dst, VTMP1.V2D(), Src2);
|
||||
break;
|
||||
case 0b00010000:
|
||||
mov(VTMP1.V1D(), Src2, 1);
|
||||
pmull(Dst, VTMP1.V2D(), Src1);
|
||||
break;
|
||||
case 0b00010001:
|
||||
pmull2(Dst, Src1, Src2);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown PCLMUL selector: {}", Op->Selector);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterEncryptionHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
REGISTER_OP(VAESIMC, AESImc);
|
||||
REGISTER_OP(VAESENC, AESEnc);
|
||||
REGISTER_OP(VAESENCLAST, AESEncLast);
|
||||
REGISTER_OP(VAESDEC, AESDec);
|
||||
REGISTER_OP(VAESDECLAST, AESDecLast);
|
||||
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
|
||||
REGISTER_OP(VAESIMC, AESImc);
|
||||
REGISTER_OP(VAESENC, AESEnc);
|
||||
REGISTER_OP(VAESENCLAST, AESEncLast);
|
||||
REGISTER_OP(VAESDEC, AESDec);
|
||||
REGISTER_OP(VAESDECLAST, AESDecLast);
|
||||
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
|
||||
REGISTER_OP(CRC32, CRC32);
|
||||
REGISTER_OP(PCLMUL, PCLMUL);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
@@ -13,7 +13,7 @@ using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(GetHostFlag) {
|
||||
auto Op = IROp->C<IR::IROp_GetHostFlag>();
|
||||
ubfx(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Header.Args[0].ID()), Op->Flag, 1);
|
||||
ubfx(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Value.ID()), Op->Flag, 1);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
+149
-222
@@ -27,6 +27,7 @@ $end_info$
|
||||
#include <FEXCore/Core/UContext.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
|
||||
@@ -35,6 +36,10 @@ $end_info$
|
||||
#include <unistd.h>
|
||||
#include <string.h>
|
||||
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
// We don't want to move above 128MB atm because that means we will have to encode longer jumps
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
|
||||
|
||||
namespace {
|
||||
static uint64_t LUDIV(uint64_t SrcHigh, uint64_t SrcLow, uint64_t Divisor) {
|
||||
__uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
|
||||
@@ -88,7 +93,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
uxth(w0, GetReg<RA_32>(IROp->Args[0].ID()));
|
||||
ldr(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
ldr(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x1);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -103,7 +108,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
fmov(v0.S(), GetSrc(IROp->Args[0].ID()).S()) ;
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x0);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -122,7 +127,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
mov(v0.D(), GetSrc(IROp->Args[0].ID()).D());
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x0);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -147,7 +152,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
else {
|
||||
mov(w0, GetReg<RA_32>(IROp->Args[0].ID()));
|
||||
}
|
||||
ldr(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
ldr(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x1);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -168,7 +173,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -187,7 +192,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -204,7 +209,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
mov(v0.D(), GetSrc(IROp->Args[0].ID()).D());
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x0);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -223,7 +228,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
|
||||
mov(v0.D(), GetSrc(IROp->Args[0].ID()).D());
|
||||
mov(v1.D(), GetSrc(IROp->Args[1].ID()).D());
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x0);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -243,7 +248,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -261,7 +266,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -279,7 +284,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -300,7 +305,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(x2, GetSrc(IROp->Args[1].ID()).V2D(), 0);
|
||||
umov(w3, GetSrc(IROp->Args[1].ID()).V8H(), 4);
|
||||
|
||||
ldr(x4, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
ldr(x4, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x4);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -318,7 +323,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(x0, GetSrc(IROp->Args[0].ID()).V2D(), 0);
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x2);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -341,7 +346,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(x2, GetSrc(IROp->Args[1].ID()).V2D(), 0);
|
||||
umov(w3, GetSrc(IROp->Args[1].ID()).V8H(), 4);
|
||||
|
||||
ldr(x4, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
ldr(x4, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
blr(x4);
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -357,42 +362,72 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
case FABI_UNKNOWN:
|
||||
default:
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
LOGMAN_MSG_A_FMT("Unhandled IR Fallback ABI: {} {}", FEXCore::IR::GetName(IROp->Op), Info.ABI);
|
||||
LOGMAN_MSG_A_FMT("Unhandled IR Fallback ABI: {} {}",
|
||||
FEXCore::IR::GetName(IROp->Op), ToUnderlying(Info.ABI));
|
||||
#endif
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
|
||||
auto Thread = Frame->Thread;
|
||||
auto GuestRip = record[1];
|
||||
|
||||
auto HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
|
||||
if (!HostCode) {
|
||||
//fmt::print("ExitFunctionLink: Aborting, {:X} not in cache\n", GuestRip);
|
||||
Frame->State.rip = GuestRip;
|
||||
return Frame->Pointers.Common.DispatcherLoopTop;
|
||||
}
|
||||
|
||||
uintptr_t branch = (uintptr_t)(record) - 8;
|
||||
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
|
||||
|
||||
auto offset = HostCode/4 - branch/4;
|
||||
if (IsInt26(offset)) {
|
||||
// optimal case - can branch directly
|
||||
// patch the code
|
||||
vixl::aarch64::Assembler emit((uint8_t*)(branch), 24);
|
||||
vixl::CodeBufferCheckScope scope(&emit, 24, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
emit.b(offset);
|
||||
emit.FinalizeCode();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency((void*)branch, 24);
|
||||
|
||||
// Add de-linking handler
|
||||
Context::Context::ThreadAddBlockLink(Thread, GuestRip, (uintptr_t)record, [branch, LinkerAddress]{
|
||||
vixl::aarch64::Assembler emit((uint8_t*)(branch), 24);
|
||||
vixl::CodeBufferCheckScope scope(&emit, 24, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
Literal l_BranchHost{LinkerAddress};
|
||||
emit.ldr(x0, &l_BranchHost);
|
||||
emit.blr(x0);
|
||||
emit.place(&l_BranchHost);
|
||||
emit.FinalizeCode();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency((void*)branch, 24);
|
||||
});
|
||||
} else {
|
||||
// fallback case - do a soft-er link by patching the pointer
|
||||
record[0] = HostCode;
|
||||
|
||||
// Add de-linking handler
|
||||
Context::Context::ThreadAddBlockLink(Thread, GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
record[0] = LinkerAddress;
|
||||
});
|
||||
}
|
||||
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
void Arm64JITCore::Op_NoOp(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
}
|
||||
|
||||
Arm64JITCore::CodeBuffer Arm64JITCore::AllocateNewCodeBuffer(size_t Size) {
|
||||
CodeBuffer Buffer;
|
||||
Buffer.Size = Size;
|
||||
Buffer.Ptr = static_cast<uint8_t*>(
|
||||
FEXCore::Allocator::mmap(nullptr,
|
||||
Buffer.Size,
|
||||
PROT_READ | PROT_WRITE | PROT_EXEC,
|
||||
MAP_PRIVATE | MAP_ANONYMOUS,
|
||||
-1, 0));
|
||||
LOGMAN_THROW_A_FMT(!!Buffer.Ptr, "Couldn't allocate code buffer");
|
||||
Dispatcher->RegisterCodeBuffer(Buffer.Ptr, Buffer.Size);
|
||||
if (CTX->Config.GlobalJITNaming()) {
|
||||
CTX->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
return Buffer;
|
||||
}
|
||||
|
||||
void Arm64JITCore::FreeCodeBuffer(CodeBuffer Buffer) {
|
||||
FEXCore::Allocator::munmap(Buffer.Ptr, Buffer.Size);
|
||||
Dispatcher->RemoveCodeBuffer(Buffer.Ptr);
|
||||
}
|
||||
|
||||
Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
: Arm64Emitter(ctx, 0)
|
||||
, CTX {ctx}
|
||||
, ThreadState {Thread} {
|
||||
: CPUBackend(Thread, INITIAL_CODE_SIZE, MAX_CODE_SIZE)
|
||||
, Arm64Emitter(ctx, 0)
|
||||
, CTX {ctx} {
|
||||
|
||||
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
|
||||
#if DEBUG
|
||||
@@ -432,84 +467,57 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::Intern
|
||||
RegisterVectorHandlers();
|
||||
RegisterEncryptionHandlers();
|
||||
|
||||
{
|
||||
DispatcherConfig config;
|
||||
config.ExitFunctionLink = reinterpret_cast<uintptr_t>(&ExitFunctionLink);
|
||||
config.ExitFunctionLinkThis = reinterpret_cast<uintptr_t>(this);
|
||||
config.StaticRegisterAssignment = ctx->Config.StaticRegisterAllocation;
|
||||
|
||||
Dispatcher = std::make_unique<Arm64Dispatcher>(CTX, ThreadState, config);
|
||||
DispatchPtr = Dispatcher->DispatchPtr;
|
||||
CallbackPtr = Dispatcher->CallbackPtr;
|
||||
}
|
||||
|
||||
{
|
||||
// Set up pointers that the JIT needs to load
|
||||
auto &Pointers = ThreadState->CurrentFrame->Pointers.AArch64;
|
||||
// Process specific
|
||||
Pointers.LUDIV = reinterpret_cast<uint64_t>(LUDIV);
|
||||
Pointers.LDIV = reinterpret_cast<uint64_t>(LDIV);
|
||||
Pointers.LUREM = reinterpret_cast<uint64_t>(LUREM);
|
||||
Pointers.LREM = reinterpret_cast<uint64_t>(LREM);
|
||||
Pointers.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
|
||||
Pointers.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
|
||||
Pointers.RemoveThreadCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::Context::RemoveThreadCodeEntryFromJit);
|
||||
Pointers.CPUIDObj = reinterpret_cast<uint64_t>(&CTX->CPUID);
|
||||
|
||||
// Common
|
||||
auto &Common = ThreadState->CurrentFrame->Pointers.Common;
|
||||
|
||||
Common.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
|
||||
Common.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
|
||||
Common.ThreadRemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::Context::ThreadRemoveCodeEntryFromJit);
|
||||
Common.CPUIDObj = reinterpret_cast<uint64_t>(&CTX->CPUID);
|
||||
|
||||
{
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::CPUIDEmu::RunFunction);
|
||||
Pointers.CPUIDFunction = PMF.GetConvertedPointer();
|
||||
Common.CPUIDFunction = PMF.GetConvertedPointer();
|
||||
}
|
||||
|
||||
Pointers.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
|
||||
Pointers.SyscallHandlerFunc = reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall);
|
||||
Common.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
|
||||
Common.SyscallHandlerFunc = reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall);
|
||||
Common.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Context::Context::ThreadExitFunctionLink<Arm64JITCore_ExitFunctionLink>);
|
||||
|
||||
|
||||
// Fill in the fallback handlers
|
||||
InterpreterOps::FillFallbackIndexPointers(Pointers.FallbackHandlerPointers);
|
||||
InterpreterOps::FillFallbackIndexPointers(Common.FallbackHandlerPointers);
|
||||
|
||||
// Thread Specific
|
||||
Pointers.SignalHandlerRefCountPointer = reinterpret_cast<uint64_t>(&Dispatcher->SignalHandlerRefCounter);
|
||||
// Platform Specific
|
||||
auto &AArch64 = ThreadState->CurrentFrame->Pointers.AArch64;
|
||||
|
||||
AArch64.LUDIV = reinterpret_cast<uint64_t>(LUDIV);
|
||||
AArch64.LDIV = reinterpret_cast<uint64_t>(LDIV);
|
||||
AArch64.LUREM = reinterpret_cast<uint64_t>(LUREM);
|
||||
AArch64.LREM = reinterpret_cast<uint64_t>(LREM);
|
||||
}
|
||||
|
||||
// Can't allocate a code buffer until after dispatcher is created
|
||||
InitialCodeBuffer = AllocateNewCodeBuffer(Arm64JITCore::INITIAL_CODE_SIZE);
|
||||
*GetBuffer() = vixl::CodeBuffer(InitialCodeBuffer.Ptr, InitialCodeBuffer.Size);
|
||||
// Must be done after Dispatcher init
|
||||
SetAllowAssembler(true);
|
||||
EmitDetectionString();
|
||||
|
||||
CurrentCodeBuffer = &InitialCodeBuffer;
|
||||
ClearCache();
|
||||
}
|
||||
|
||||
void Arm64JITCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGILL, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleSIGILL(Signal, info, ucontext);
|
||||
return Thread->CTX->Dispatcher->HandleSIGILL(Thread, Signal, info, ucontext);
|
||||
}, true);
|
||||
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
|
||||
|
||||
if (!Core->Dispatcher->IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext))) {
|
||||
if (!Thread->CPUBackend->IsAddressInCodeBuffer(ArchHelpers::Context::GetPc(ucontext))) {
|
||||
// Wasn't a sigbus in JIT code
|
||||
return false;
|
||||
}
|
||||
|
||||
return FEXCore::ArchHelpers::Arm64::HandleSIGBUS(Core->CTX->Config.ParanoidTSO(), Signal, info, ucontext);
|
||||
return FEXCore::ArchHelpers::Arm64::HandleSIGBUS(Thread->CTX->Config.ParanoidTSO(), Signal, info, ucontext);
|
||||
}, true);
|
||||
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SignalDelegator::SIGNAL_FOR_PAUSE, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleSignalPause(Signal, info, ucontext);
|
||||
}, true);
|
||||
|
||||
auto GuestSignalHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) -> bool {
|
||||
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleGuestSignal(Signal, info, ucontext, GuestAction, GuestStack);
|
||||
};
|
||||
|
||||
for (uint32_t Signal = 0; Signal <= SignalDelegator::MAX_SIGNALS; ++Signal) {
|
||||
CTX->SignalDelegation->RegisterHostSignalHandlerForGuest(Signal, GuestSignalHandler);
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64JITCore::EmitDetectionString() {
|
||||
@@ -521,54 +529,14 @@ void Arm64JITCore::EmitDetectionString() {
|
||||
|
||||
void Arm64JITCore::ClearCache() {
|
||||
// Get the backing code buffer
|
||||
auto Buffer = GetBuffer();
|
||||
if (Dispatcher->SignalHandlerRefCounter == 0) {
|
||||
if (!CodeBuffers.empty()) {
|
||||
// If we have more than one code buffer we are tracking then walk them and delete
|
||||
// This is a cleanup step
|
||||
for (auto CodeBuffer : CodeBuffers) {
|
||||
FreeCodeBuffer(CodeBuffer);
|
||||
}
|
||||
CodeBuffers.clear();
|
||||
|
||||
// Set the current code buffer to the initial
|
||||
*Buffer = vixl::CodeBuffer(InitialCodeBuffer.Ptr, InitialCodeBuffer.Size);
|
||||
CurrentCodeBuffer = &InitialCodeBuffer;
|
||||
}
|
||||
|
||||
if (CurrentCodeBuffer->Size == MAX_CODE_SIZE) {
|
||||
// Rewind to the start of the code cache start
|
||||
Buffer->Reset();
|
||||
}
|
||||
else {
|
||||
FreeCodeBuffer(InitialCodeBuffer);
|
||||
|
||||
// Resize the code buffer and reallocate our code size
|
||||
InitialCodeBuffer.Size *= 1.5;
|
||||
InitialCodeBuffer.Size = std::min(InitialCodeBuffer.Size, MAX_CODE_SIZE);
|
||||
|
||||
InitialCodeBuffer = AllocateNewCodeBuffer(InitialCodeBuffer.Size);
|
||||
*Buffer = vixl::CodeBuffer(InitialCodeBuffer.Ptr, InitialCodeBuffer.Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// We have signal handlers that have generated code
|
||||
// This means that we can not safely clear the code at this point in time
|
||||
// Allocate some new code buffers that we can switch over to instead
|
||||
auto NewCodeBuffer = Arm64JITCore::AllocateNewCodeBuffer(Arm64JITCore::INITIAL_CODE_SIZE);
|
||||
EmplaceNewCodeBuffer(NewCodeBuffer);
|
||||
*Buffer = vixl::CodeBuffer(NewCodeBuffer.Ptr, NewCodeBuffer.Size);
|
||||
}
|
||||
|
||||
auto CodeBuffer = GetEmptyCodeBuffer();
|
||||
*GetBuffer() = vixl::CodeBuffer(CodeBuffer->Ptr, CodeBuffer->Size);
|
||||
EmitDetectionString();
|
||||
}
|
||||
|
||||
Arm64JITCore::~Arm64JITCore() {
|
||||
for (auto CodeBuffer : CodeBuffers) {
|
||||
FreeCodeBuffer(CodeBuffer);
|
||||
}
|
||||
CodeBuffers.clear();
|
||||
|
||||
FreeCodeBuffer(InitialCodeBuffer);
|
||||
}
|
||||
|
||||
IR::PhysicalRegister Arm64JITCore::GetPhys(IR::NodeID Node) const {
|
||||
@@ -583,12 +551,12 @@ template<>
|
||||
aarch64::Register Arm64JITCore::GetReg<Arm64JITCore::RA_32>(IR::NodeID Node) const {
|
||||
auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
return SRA64[Reg.Reg].W();
|
||||
} else if (Reg.Class == IR::GPRClass.Val) {
|
||||
return RA64[Reg.Reg].W();
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "Unexpected Class: {}", Reg.Class);
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
@@ -598,12 +566,12 @@ template<>
|
||||
aarch64::Register Arm64JITCore::GetReg<Arm64JITCore::RA_64>(IR::NodeID Node) const {
|
||||
auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
return SRA64[Reg.Reg];
|
||||
} else if (Reg.Class == IR::GPRClass.Val) {
|
||||
return RA64[Reg.Reg];
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "Unexpected Class: {}", Reg.Class);
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
@@ -624,12 +592,12 @@ std::pair<aarch64::Register, aarch64::Register> Arm64JITCore::GetSrcPair<Arm64JI
|
||||
aarch64::VRegister Arm64JITCore::GetSrc(IR::NodeID Node) const {
|
||||
auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
return SRAFPR[Reg.Reg];
|
||||
} else if (Reg.Class == IR::FPRClass.Val) {
|
||||
return RAFPR[Reg.Reg];
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "Unexpected Class: {}", Reg.Class);
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
@@ -638,12 +606,12 @@ aarch64::VRegister Arm64JITCore::GetSrc(IR::NodeID Node) const {
|
||||
aarch64::VRegister Arm64JITCore::GetDst(IR::NodeID Node) const {
|
||||
auto Reg = GetPhys(Node);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
return SRAFPR[Reg.Reg];
|
||||
} else if (Reg.Class == IR::FPRClass.Val) {
|
||||
return RAFPR[Reg.Reg];
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "Unexpected Class: {}", Reg.Class);
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
@@ -699,13 +667,18 @@ bool Arm64JITCore::IsGPR(IR::NodeID Node) const {
|
||||
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
|
||||
}
|
||||
|
||||
void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
void *Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData,
|
||||
bool GDBEnabled) {
|
||||
using namespace aarch64;
|
||||
JumpTargets.clear();
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
|
||||
this->Entry = Entry;
|
||||
this->RAData = RAData;
|
||||
this->DebugData = DebugData;
|
||||
|
||||
#ifndef NDEBUG
|
||||
LoadConstant(x0, Entry);
|
||||
@@ -714,9 +687,9 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
|
||||
this->IR = IR;
|
||||
|
||||
// Fairly excessive buffer range to make sure we don't overflow
|
||||
uint32_t BufferRange = SSACount * 16;
|
||||
uint32_t BufferRange = SSACount * 16 + GDBEnabled * Dispatcher::MaxGDBPauseCheckSize;
|
||||
if ((GetCursorOffset() + BufferRange) > CurrentCodeBuffer->Size) {
|
||||
ThreadState->CTX->ClearCodeCache(ThreadState, false);
|
||||
CTX->ClearCodeCache(ThreadState);
|
||||
}
|
||||
|
||||
// AAPCS64
|
||||
@@ -739,31 +712,11 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
|
||||
// X1-X3 = Temp
|
||||
// X4-r18 = RA
|
||||
|
||||
GuestEntry = GetCursorAddress<uint64_t>();
|
||||
GuestEntry = GetCursorAddress<uint8_t *>();
|
||||
|
||||
if (CTX->GetGdbServerStatus()) {
|
||||
aarch64::Label RunBlock;
|
||||
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
// This happens when single stepping
|
||||
|
||||
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Thread))); // Get thread
|
||||
ldr(x0, MemOperand(x0, offsetof(FEXCore::Core::InternalThreadState, CTX))); // Get Context
|
||||
ldr(w0, MemOperand(x0, offsetof(FEXCore::Context::Context, Config.RunningMode)));
|
||||
|
||||
// If the value == 0 then we don't need to stop
|
||||
cbz(w0, &RunBlock);
|
||||
{
|
||||
// Make sure RIP is syncronized to the context
|
||||
LoadConstant(x0, Entry);
|
||||
str(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
|
||||
|
||||
// Stop the thread
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.ThreadPauseHandlerSpillSRA)));
|
||||
br(x0);
|
||||
}
|
||||
bind(&RunBlock);
|
||||
if (GDBEnabled) {
|
||||
auto GDBSize = CTX->Dispatcher->GenerateGDBPauseCheck(GuestEntry, Entry);
|
||||
GetBuffer()->CursorForward(GDBSize);
|
||||
}
|
||||
|
||||
//LOGMAN_THROW_A_FMT(RAData->HasFullRA(), "Arm64 JIT only works with RA");
|
||||
@@ -785,10 +738,10 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
|
||||
using namespace FEXCore::IR;
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
LOGMAN_THROW_AA_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
#endif
|
||||
|
||||
uintptr_t BlockStartHostCode = GetCursorAddress<uintptr_t>();
|
||||
auto BlockStartHostCode = GetCursorAddress<uint8_t *>();
|
||||
{
|
||||
const auto Node = IR->GetID(BlockNode);
|
||||
const auto IsTarget = JumpTargets.try_emplace(Node).first;
|
||||
@@ -812,7 +765,10 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
|
||||
}
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->Subblocks.push_back({BlockStartHostCode, static_cast<uint32_t>(GetCursorAddress<uintptr_t>() - BlockStartHostCode)});
|
||||
DebugData->Subblocks.push_back({
|
||||
static_cast<uint32_t>(BlockStartHostCode - GuestEntry),
|
||||
static_cast<uint32_t>(GetCursorAddress<uint8_t *>() - BlockStartHostCode)
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
@@ -825,66 +781,30 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
|
||||
|
||||
FinalizeCode();
|
||||
|
||||
auto CodeEnd = GetCursorAddress<uint64_t>();
|
||||
CPU.EnsureIAndDCacheCoherency(reinterpret_cast<void*>(GuestEntry), CodeEnd - reinterpret_cast<uint64_t>(GuestEntry));
|
||||
auto CodeEnd = GetCursorAddress<uint8_t *>();
|
||||
CPU.EnsureIAndDCacheCoherency(GuestEntry, CodeEnd - GuestEntry);
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->HostCodeSize = reinterpret_cast<uintptr_t>(CodeEnd) - reinterpret_cast<uintptr_t>(GuestEntry);
|
||||
DebugData->HostCodeSize = CodeEnd - GuestEntry;
|
||||
DebugData->Relocations = &Relocations;
|
||||
}
|
||||
|
||||
this->IR = nullptr;
|
||||
|
||||
return reinterpret_cast<void*>(GuestEntry);
|
||||
return GuestEntry;
|
||||
}
|
||||
|
||||
uint64_t Arm64JITCore::ExitFunctionLink(Arm64JITCore *core, FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
|
||||
auto Thread = Frame->Thread;
|
||||
auto GuestRip = record[1];
|
||||
void Arm64JITCore::ResetStack() {
|
||||
if (SpillSlots == 0)
|
||||
return;
|
||||
|
||||
auto HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
|
||||
if (!HostCode) {
|
||||
//fmt::print("ExitFunctionLink: Aborting, {:X} not in cache\n", GuestRip);
|
||||
Frame->State.rip = GuestRip;
|
||||
return core->Dispatcher->AbsoluteLoopTopAddress;
|
||||
}
|
||||
|
||||
uintptr_t branch = (uintptr_t)(record) - 8;
|
||||
auto LinkerAddress = core->Dispatcher->ExitFunctionLinkerAddress;
|
||||
|
||||
auto offset = HostCode/4 - branch/4;
|
||||
if (IsInt26(offset)) {
|
||||
// optimal case - can branch directly
|
||||
// patch the code
|
||||
vixl::aarch64::Assembler emit((uint8_t*)(branch), 24);
|
||||
vixl::CodeBufferCheckScope scope(&emit, 24, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
emit.b(offset);
|
||||
emit.FinalizeCode();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency((void*)branch, 24);
|
||||
|
||||
// Add de-linking handler
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [branch, LinkerAddress]{
|
||||
vixl::aarch64::Assembler emit((uint8_t*)(branch), 24);
|
||||
vixl::CodeBufferCheckScope scope(&emit, 24, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
Literal l_BranchHost{LinkerAddress};
|
||||
emit.ldr(x0, &l_BranchHost);
|
||||
emit.blr(x0);
|
||||
emit.place(&l_BranchHost);
|
||||
emit.FinalizeCode();
|
||||
vixl::aarch64::CPU::EnsureIAndDCacheCoherency((void*)branch, 24);
|
||||
});
|
||||
if (IsImmAddSub(SpillSlots * 16)) {
|
||||
add(sp, sp, SpillSlots * 16);
|
||||
} else {
|
||||
// fallback case - do a soft-er link by patching the pointer
|
||||
record[0] = HostCode;
|
||||
|
||||
// Add de-linking handler
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
record[0] = LinkerAddress;
|
||||
});
|
||||
// Too big to fit in a 12bit immediate
|
||||
LoadConstant(x0, SpillSlots * 16);
|
||||
add(sp, sp, x0);
|
||||
}
|
||||
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
std::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
@@ -894,4 +814,11 @@ std::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::Context *ctx, F
|
||||
void InitializeArm64JITSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
Arm64JITCore::InitializeSignalHandlers(CTX);
|
||||
}
|
||||
|
||||
CPUBackendFeatures GetArm64JITBackendFeatures() {
|
||||
return CPUBackendFeatures {
|
||||
.SupportsStaticRegisterAllocation = true
|
||||
};
|
||||
}
|
||||
|
||||
}
|
||||
+77
-49
@@ -6,17 +6,23 @@ $end_info$
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
#include "aarch64/disasm-aarch64.h"
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
#include <aarch64/disasm-aarch64.h>
|
||||
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <map>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#define STATE x28
|
||||
#define TMP1 x0
|
||||
#define TMP2 x1
|
||||
@@ -37,11 +43,6 @@ using namespace vixl::aarch64;
|
||||
|
||||
class Arm64JITCore final : public CPUBackend, public Arm64Emitter {
|
||||
public:
|
||||
struct CodeBuffer {
|
||||
uint8_t *Ptr;
|
||||
size_t Size;
|
||||
};
|
||||
|
||||
explicit Arm64JITCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
~Arm64JITCore() override;
|
||||
@@ -51,7 +52,7 @@ public:
|
||||
[[nodiscard]] void *CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) override;
|
||||
|
||||
[[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
|
||||
@@ -59,13 +60,6 @@ public:
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
[[nodiscard]] CodeBuffer AllocateNewCodeBuffer(size_t Size);
|
||||
|
||||
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true) const override {
|
||||
return Dispatcher->IsAddressInJITCode(Address, IncludeDispatcher);
|
||||
}
|
||||
|
||||
static void InitializeSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
|
||||
void ClearRelocations() override { Relocations.clear(); }
|
||||
@@ -73,10 +67,8 @@ public:
|
||||
private:
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
|
||||
std::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
Label *PendingTargetLabel;
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::Core::InternalThreadState *ThreadState;
|
||||
FEXCore::IR::IRListView const *IR;
|
||||
uint64_t Entry;
|
||||
|
||||
@@ -147,38 +139,77 @@ private:
|
||||
vixl::aarch64::Decoder Decoder;
|
||||
#endif
|
||||
|
||||
void EmplaceNewCodeBuffer(CodeBuffer Buffer) {
|
||||
CurrentCodeBuffer = &CodeBuffers.emplace_back(Buffer);
|
||||
}
|
||||
|
||||
void FreeCodeBuffer(CodeBuffer Buffer);
|
||||
|
||||
// This is the initial code buffer that we will fall back to
|
||||
// In a program without signals and code clearing, we will typically
|
||||
// only have this code buffer
|
||||
CodeBuffer InitialCodeBuffer{};
|
||||
// This is the array of /additional/ code buffers that we may need to allocate
|
||||
// Allocation only occurs when we've hit signals and need to clear code cache
|
||||
// For code safety we can't delete code buffers until outside of all signals
|
||||
std::vector<CodeBuffer> CodeBuffers{};
|
||||
|
||||
// This is the current code buffer that we are tracking
|
||||
CodeBuffer *CurrentCodeBuffer{};
|
||||
|
||||
// We don't want to mvoe above 128MB atm because that means we will have to encode longer jumps
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
|
||||
static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096 * 2;
|
||||
|
||||
#if DEBUG
|
||||
vixl::aarch64::Disassembler Disasm;
|
||||
#endif
|
||||
|
||||
static uint64_t ExitFunctionLink(Arm64JITCore *core, FEXCore::Core::CpuStateFrame *Frame, uint64_t *record);
|
||||
|
||||
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
|
||||
void EmitDetectionString();
|
||||
IR::RegisterAllocationPass *RAPass;
|
||||
IR::RegisterAllocationData *RAData;
|
||||
FEXCore::Core::DebugData *DebugData;
|
||||
|
||||
void ResetStack();
|
||||
/**
|
||||
* @name Relocations
|
||||
* @{ */
|
||||
|
||||
uint64_t GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
|
||||
|
||||
/**
|
||||
* @brief A literal pair relocation object for named symbol literals
|
||||
*/
|
||||
struct NamedSymbolLiteralPair {
|
||||
Literal<uint64_t> Lit;
|
||||
Relocation MoveABI{};
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Inserts a thunk relocation
|
||||
*
|
||||
* @param Reg - The GPR to move the thunk handler in to
|
||||
* @param Sum - The hash of the thunk
|
||||
*/
|
||||
void InsertNamedThunkRelocation(vixl::aarch64::Register Reg, const IR::SHA256Sum &Sum);
|
||||
|
||||
/**
|
||||
* @brief Inserts a guest GPR move relocation
|
||||
*
|
||||
* @param Reg - The GPR to move the guest RIP in to
|
||||
* @param Constant - The guest RIP that will be relocated
|
||||
*/
|
||||
void InsertGuestRIPMove(vixl::aarch64::Register Reg, uint64_t Constant);
|
||||
|
||||
/**
|
||||
* @brief Inserts a named symbol as a literal in memory
|
||||
*
|
||||
* Need to use `PlaceNamedSymbolLiteral` with the return value to place the literal in the desired location
|
||||
*
|
||||
* @param Op The named symbol to place
|
||||
*
|
||||
* @return A temporary `NamedSymbolLiteralPair`
|
||||
*/
|
||||
NamedSymbolLiteralPair InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
|
||||
|
||||
/**
|
||||
* @brief Place the named symbol literal relocation in memory
|
||||
*
|
||||
* @param Lit - Which literal to place
|
||||
*/
|
||||
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit);
|
||||
|
||||
std::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
|
||||
///< Relocation code loading
|
||||
bool ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations);
|
||||
|
||||
/** @} */
|
||||
|
||||
uint32_t SpillSlots{};
|
||||
/**
|
||||
* @brief Current guest RIP entrypoint
|
||||
*/
|
||||
uint8_t *GuestEntry{};
|
||||
|
||||
using OpHandler = void (Arm64JITCore::*)(IR::IROp_Header *IROp, IR::NodeID Node);
|
||||
std::array<OpHandler, IR::IROps::OP_LAST + 1> OpHandlers {};
|
||||
@@ -269,8 +300,6 @@ private:
|
||||
DEF_OP(AtomicFetchNeg);
|
||||
|
||||
///< Branch ops
|
||||
DEF_OP(GuestCallDirect);
|
||||
DEF_OP(GuestCallIndirect);
|
||||
DEF_OP(SignalReturn);
|
||||
DEF_OP(CallbackReturn);
|
||||
DEF_OP(ExitFunction);
|
||||
@@ -280,7 +309,7 @@ private:
|
||||
DEF_OP(InlineSyscall);
|
||||
DEF_OP(Thunk);
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(RemoveThreadCodeEntry);
|
||||
DEF_OP(ThreadRemoveCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
|
||||
///< Conversion ops
|
||||
@@ -320,7 +349,7 @@ private:
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
///< Misc ops
|
||||
DEF_OP(EndBlock);
|
||||
DEF_OP(GuestOpcode);
|
||||
DEF_OP(Fence);
|
||||
DEF_OP(Break);
|
||||
DEF_OP(Phi);
|
||||
@@ -437,9 +466,8 @@ private:
|
||||
DEF_OP(AESDecLast);
|
||||
DEF_OP(AESKeyGenAssist);
|
||||
DEF_OP(CRC32);
|
||||
DEF_OP(PCLMUL);
|
||||
#undef DEF_OP
|
||||
};
|
||||
|
||||
|
||||
}
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
+56
-50
@@ -4,6 +4,8 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
@@ -58,26 +60,26 @@ DEF_OP(LoadContext) {
|
||||
|
||||
DEF_OP(StoreContext) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
strb(GetReg<RA_32>(Op->Header.Args[0].ID()), MemOperand(STATE, Op->Offset));
|
||||
strb(GetReg<RA_32>(Op->Value.ID()), MemOperand(STATE, Op->Offset));
|
||||
break;
|
||||
case 2:
|
||||
strh(GetReg<RA_32>(Op->Header.Args[0].ID()), MemOperand(STATE, Op->Offset));
|
||||
strh(GetReg<RA_32>(Op->Value.ID()), MemOperand(STATE, Op->Offset));
|
||||
break;
|
||||
case 4:
|
||||
str(GetReg<RA_32>(Op->Header.Args[0].ID()), MemOperand(STATE, Op->Offset));
|
||||
str(GetReg<RA_32>(Op->Value.ID()), MemOperand(STATE, Op->Offset));
|
||||
break;
|
||||
case 8:
|
||||
str(GetReg<RA_64>(Op->Header.Args[0].ID()), MemOperand(STATE, Op->Offset));
|
||||
str(GetReg<RA_64>(Op->Value.ID()), MemOperand(STATE, Op->Offset));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto Src = GetSrc(Op->Header.Args[0].ID());
|
||||
auto Src = GetSrc(Op->Value.ID());
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
str(Src.B(), MemOperand(STATE, Op->Offset));
|
||||
@@ -104,7 +106,7 @@ DEF_OP(LoadRegister) {
|
||||
auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
auto regId = (Op->Offset - offsetof(FEXCore::Core::CpuStateFrame, State.gregs[0])) / 8;
|
||||
auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.gregs[0])) / Core::CPUState::GPR_REG_SIZE;
|
||||
auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRA64.size(), "out of range regId");
|
||||
@@ -113,30 +115,32 @@ DEF_OP(LoadRegister) {
|
||||
|
||||
switch(Op->Header.Size) {
|
||||
case 1:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0 || regOffs == 1, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0 || regOffs == 1, "unexpected regOffs");
|
||||
ubfx(GetReg<RA_64>(Node), reg, regOffs * 8, 8);
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
ubfx(GetReg<RA_64>(Node), reg, 0, 16);
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
if (GetReg<RA_64>(Node).GetCode() != reg.GetCode())
|
||||
mov(GetReg<RA_32>(Node), reg.W());
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
if (GetReg<RA_64>(Node).GetCode() != reg.GetCode())
|
||||
mov(GetReg<RA_64>(Node), reg);
|
||||
break;
|
||||
}
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
auto regId = (Op->Offset - offsetof(FEXCore::Core::CpuStateFrame, State.xmm[0][0])) / 16;
|
||||
auto regOffs = Op->Offset & 15;
|
||||
const auto regSize = CTX->HostFeatures.SupportsAVX ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
const auto regOffs = Op->Offset & 15;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRAFPR.size(), "out of range regId");
|
||||
|
||||
@@ -145,17 +149,17 @@ DEF_OP(LoadRegister) {
|
||||
|
||||
switch(Op->Header.Size) {
|
||||
case 1:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
mov(host.B(), guest.B());
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
fmov(host.H(), guest.H());
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_A_FMT((regOffs & 3) == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 3) == 0, "unexpected regOffs");
|
||||
if (regOffs == 0) {
|
||||
if (host.GetCode() != guest.GetCode())
|
||||
fmov(host.S(), guest.S());
|
||||
@@ -165,7 +169,7 @@ DEF_OP(LoadRegister) {
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_A_FMT((regOffs & 7) == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 7) == 0, "unexpected regOffs");
|
||||
if (regOffs == 0) {
|
||||
if (host.GetCode() != guest.GetCode())
|
||||
mov(host.D(), guest.D());
|
||||
@@ -175,13 +179,13 @@ DEF_OP(LoadRegister) {
|
||||
break;
|
||||
|
||||
case 16:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
if (host.GetCode() != guest.GetCode())
|
||||
mov(host.Q(), guest.Q());
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
LOGMAN_THROW_AA_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -189,7 +193,7 @@ DEF_OP(StoreRegister) {
|
||||
auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
auto regId = Op->Offset / 8 - 1;
|
||||
auto regId = (Op->Offset / Core::CPUState::GPR_REG_SIZE) - 1;
|
||||
auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRA64.size(), "out of range regId");
|
||||
@@ -198,29 +202,31 @@ DEF_OP(StoreRegister) {
|
||||
|
||||
switch(Op->Header.Size) {
|
||||
case 1:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0 || regOffs == 1, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0 || regOffs == 1, "unexpected regOffs");
|
||||
bfi(reg, GetReg<RA_64>(Op->Value.ID()), regOffs * 8, 8);
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
bfi(reg, GetReg<RA_64>(Op->Value.ID()), 0, 16);
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
bfi(reg, GetReg<RA_64>(Op->Value.ID()), 0, 32);
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
if (GetReg<RA_64>(Op->Value.ID()).GetCode() != reg.GetCode())
|
||||
mov(reg, GetReg<RA_64>(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
auto regId = (Op->Offset - offsetof(FEXCore::Core::CpuStateFrame, State.xmm[0][0])) / 16;
|
||||
auto regOffs = Op->Offset & 15;
|
||||
const auto regSize = CTX->HostFeatures.SupportsAVX ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
const auto regOffs = Op->Offset & 15;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRAFPR.size(), "regId out of range");
|
||||
|
||||
@@ -233,36 +239,36 @@ DEF_OP(StoreRegister) {
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_A_FMT((regOffs & 1) == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 1) == 0, "unexpected regOffs");
|
||||
ins(guest.V8H(), regOffs/2, host.V8H(), 0);
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_A_FMT((regOffs & 3) == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 3) == 0, "unexpected regOffs");
|
||||
ins(guest.V4S(), regOffs/4, host.V4S(), 0);
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_A_FMT((regOffs & 7) == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 7) == 0, "unexpected regOffs");
|
||||
ins(guest.V2D(), regOffs / 8, host.V2D(), 0);
|
||||
break;
|
||||
|
||||
case 16:
|
||||
LOGMAN_THROW_A_FMT(regOffs == 0, "unexpected regOffs");
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
if (guest.GetCode() != host.GetCode())
|
||||
mov(guest.Q(), host.Q());
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
LOGMAN_THROW_AA_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
DEF_OP(LoadContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
size_t size = IROp->Size;
|
||||
auto index = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
const size_t size = IROp->Size;
|
||||
auto index = GetReg<RA_64>(Op->Index.ID());
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
switch (Op->Stride) {
|
||||
@@ -349,11 +355,11 @@ DEF_OP(LoadContextIndexed) {
|
||||
|
||||
DEF_OP(StoreContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
size_t size = IROp->Size;
|
||||
auto index = GetReg<RA_64>(Op->Header.Args[1].ID());
|
||||
const size_t size = IROp->Size;
|
||||
auto index = GetReg<RA_64>(Op->Index.ID());
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
auto value = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto value = GetReg<RA_64>(Op->Value.ID());
|
||||
|
||||
switch (Op->Stride) {
|
||||
case 1:
|
||||
@@ -392,7 +398,7 @@ DEF_OP(StoreContextIndexed) {
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto value = GetSrc(Op->Header.Args[0].ID());
|
||||
auto value = GetSrc(Op->Value.ID());
|
||||
|
||||
switch (Op->Stride) {
|
||||
case 1:
|
||||
@@ -441,25 +447,25 @@ DEF_OP(StoreContextIndexed) {
|
||||
|
||||
DEF_OP(SpillRegister) {
|
||||
auto Op = IROp->C<IR::IROp_SpillRegister>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
uint32_t SlotOffset = Op->Slot * 16;
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const uint32_t SlotOffset = Op->Slot * 16;
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
strb(GetReg<RA_64>(Op->Header.Args[0].ID()), MemOperand(sp, SlotOffset));
|
||||
strb(GetReg<RA_64>(Op->Value.ID()), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
strh(GetReg<RA_64>(Op->Header.Args[0].ID()), MemOperand(sp, SlotOffset));
|
||||
strh(GetReg<RA_64>(Op->Value.ID()), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
str(GetReg<RA_32>(Op->Header.Args[0].ID()), MemOperand(sp, SlotOffset));
|
||||
str(GetReg<RA_32>(Op->Value.ID()), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
str(GetReg<RA_64>(Op->Header.Args[0].ID()), MemOperand(sp, SlotOffset));
|
||||
str(GetReg<RA_64>(Op->Value.ID()), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize);
|
||||
@@ -467,15 +473,15 @@ DEF_OP(SpillRegister) {
|
||||
} else if (Op->Class == FEXCore::IR::FPRClass) {
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
str(GetSrc(Op->Header.Args[0].ID()).S(), MemOperand(sp, SlotOffset));
|
||||
str(GetSrc(Op->Value.ID()).S(), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
str(GetSrc(Op->Header.Args[0].ID()).D(), MemOperand(sp, SlotOffset));
|
||||
str(GetSrc(Op->Value.ID()).D(), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
case 16: {
|
||||
str(GetSrc(Op->Header.Args[0].ID()), MemOperand(sp, SlotOffset));
|
||||
str(GetSrc(Op->Value.ID()), MemOperand(sp, SlotOffset));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize);
|
||||
@@ -539,7 +545,7 @@ DEF_OP(LoadFlag) {
|
||||
|
||||
DEF_OP(StoreFlag) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
strb(GetReg<RA_64>(Op->Header.Args[0].ID()), MemOperand(STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag));
|
||||
strb(GetReg<RA_64>(Op->Value.ID()), MemOperand(STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag));
|
||||
}
|
||||
|
||||
MemOperand Arm64JITCore::GenerateMemOperand(uint8_t AccessSize, aarch64::Register Base, IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale) {
|
||||
@@ -570,7 +576,7 @@ MemOperand Arm64JITCore::GenerateMemOperand(uint8_t AccessSize, aarch64::Registe
|
||||
DEF_OP(LoadMem) {
|
||||
auto Op = IROp->C<IR::IROp_LoadMem>();
|
||||
|
||||
auto MemReg = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto MemReg = GetReg<RA_64>(Op->Addr.ID());
|
||||
auto MemSrc = GenerateMemOperand(IROp->Size, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
@@ -992,7 +998,7 @@ DEF_OP(VStoreMemElement) {
|
||||
DEF_OP(CacheLineClear) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineClear>();
|
||||
|
||||
auto MemReg = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto MemReg = GetReg<RA_64>(Op->Addr.ID());
|
||||
|
||||
// Clear dcache only
|
||||
// icache doesn't matter here since the guest application shouldn't be calling clflush on JIT code.
|
||||
@@ -1007,7 +1013,7 @@ DEF_OP(CacheLineClear) {
|
||||
DEF_OP(CacheLineZero) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineZero>();
|
||||
|
||||
auto MemReg = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto MemReg = GetReg<RA_64>(Op->Addr.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsCLZERO) {
|
||||
// We can use this instruction directly
|
||||
|
||||
+45
-41
@@ -4,13 +4,21 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include <syscall.h>
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(GuestOpcode) {
|
||||
auto Op = IROp->C<IR::IROp_GuestOpcode>();
|
||||
// metadata
|
||||
DebugData->GuestOpcodes.push_back({Op->GuestEntryOffset, GetCursorAddress<uint8_t*>() - GuestEntry});
|
||||
}
|
||||
|
||||
DEF_OP(Fence) {
|
||||
auto Op = IROp->C<IR::IROp_Fence>();
|
||||
switch (Op->Fence) {
|
||||
@@ -29,43 +37,38 @@ DEF_OP(Fence) {
|
||||
|
||||
DEF_OP(Break) {
|
||||
auto Op = IROp->C<IR::IROp_Break>();
|
||||
switch (Op->Reason) {
|
||||
case FEXCore::IR::Break_Unimplemented: // Hard fault
|
||||
case FEXCore::IR::Break_Interrupt: // Guest ud2
|
||||
hlt(4);
|
||||
break;
|
||||
case FEXCore::IR::Break_Overflow: // overflow
|
||||
ResetStack();
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.OverflowExceptionHandler)));
|
||||
br(TMP1);
|
||||
break;
|
||||
case FEXCore::IR::Break_Halt: { // HLT
|
||||
// Time to quit
|
||||
// Set our stack to the starting stack location
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)));
|
||||
add(sp, TMP1, 0);
|
||||
|
||||
// Now we need to jump to the thread stop handler
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.ThreadStopHandlerSpillSRA)));
|
||||
br(TMP1);
|
||||
break;
|
||||
}
|
||||
case FEXCore::IR::Break_Interrupt3: { // INT3
|
||||
ResetStack();
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.ThreadPauseHandlerSpillSRA)));
|
||||
br(TMP1);
|
||||
break;
|
||||
}
|
||||
case FEXCore::IR::Break_InvalidInstruction:
|
||||
{
|
||||
ResetStack();
|
||||
// First we must reset the stack
|
||||
ResetStack();
|
||||
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.UnimplementedInstructionHandler)));
|
||||
br(TMP1);
|
||||
LoadConstant(w1, 1);
|
||||
strb(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.FaultToTopAndGeneratedException)));
|
||||
LoadConstant(w1, Op->Reason.Signal);
|
||||
strb(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.Signal)));
|
||||
LoadConstant(w1, Op->Reason.TrapNumber);
|
||||
str(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.TrapNo)));
|
||||
LoadConstant(w1, Op->Reason.si_code);
|
||||
str(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.si_code)));
|
||||
LoadConstant(x1, Op->Reason.ErrorRegister);
|
||||
str(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.err_code)));
|
||||
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Break reason: {}", Op->Reason);
|
||||
switch (Op->Reason.Signal) {
|
||||
case SIGILL:
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGILL)));
|
||||
br(TMP1);
|
||||
break;
|
||||
case SIGTRAP:
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP)));
|
||||
br(TMP1);
|
||||
break;
|
||||
case SIGSEGV:
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGSEGV)));
|
||||
br(TMP1);
|
||||
break;
|
||||
default:
|
||||
ldr(TMP1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP)));
|
||||
br(TMP1);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -97,7 +100,7 @@ DEF_OP(GetRoundingMode) {
|
||||
|
||||
DEF_OP(SetRoundingMode) {
|
||||
auto Op = IROp->C<IR::IROp_SetRoundingMode>();
|
||||
auto Src = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
auto Src = GetReg<RA_64>(Op->RoundMode.ID());
|
||||
|
||||
// Setup the rounding flags correctly
|
||||
and_(TMP1, Src, 0b11);
|
||||
@@ -132,15 +135,15 @@ DEF_OP(Print) {
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
|
||||
if (IsGPR(Op->Header.Args[0].ID())) {
|
||||
mov(x0, GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.PrintValue)));
|
||||
if (IsGPR(Op->Value.ID())) {
|
||||
mov(x0, GetReg<RA_64>(Op->Value.ID()));
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintValue)));
|
||||
}
|
||||
else {
|
||||
fmov(x0, GetSrc(Op->Header.Args[0].ID()).V1D());
|
||||
fmov(x0, GetSrc(Op->Value.ID()).V1D());
|
||||
// Bug in vixl that source vector needs to b V1D rather than V2D?
|
||||
fmov(x1, GetSrc(Op->Header.Args[0].ID()).V1D(), 1);
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.PrintVectorValue)));
|
||||
fmov(x1, GetSrc(Op->Value.ID()).V1D(), 1);
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintVectorValue)));
|
||||
}
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
@@ -231,6 +234,7 @@ void Arm64JITCore::RegisterMiscHandlers() {
|
||||
REGISTER_OP(CODEBLOCK, NoOp);
|
||||
REGISTER_OP(BEGINBLOCK, NoOp);
|
||||
REGISTER_OP(ENDBLOCK, NoOp);
|
||||
REGISTER_OP(GUESTOPCODE, GuestOpcode);
|
||||
REGISTER_OP(FENCE, Fence);
|
||||
REGISTER_OP(BREAK, Break);
|
||||
REGISTER_OP(PHI, NoOp);
|
||||
|
||||
@@ -15,13 +15,13 @@ DEF_OP(ExtractElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_ExtractElementPair>();
|
||||
switch (Op->Header.Size) {
|
||||
case 4: {
|
||||
auto Src = GetSrcPair<RA_32>(Op->Header.Args[0].ID());
|
||||
auto Src = GetSrcPair<RA_32>(Op->Pair.ID());
|
||||
std::array<aarch64::Register, 2> Regs = {Src.first, Src.second};
|
||||
mov (GetReg<RA_32>(Node), Regs[Op->Element]);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
auto Src = GetSrcPair<RA_64>(Op->Header.Args[0].ID());
|
||||
auto Src = GetSrcPair<RA_64>(Op->Pair.ID());
|
||||
std::array<aarch64::Register, 2> Regs = {Src.first, Src.second};
|
||||
mov (GetReg<RA_64>(Node), Regs[Op->Element]);
|
||||
break;
|
||||
@@ -40,15 +40,15 @@ DEF_OP(CreateElementPair) {
|
||||
switch (IROp->ElementSize) {
|
||||
case 4: {
|
||||
Dst = GetSrcPair<RA_32>(Node);
|
||||
RegFirst = GetReg<RA_32>(Op->Header.Args[0].ID());
|
||||
RegSecond = GetReg<RA_32>(Op->Header.Args[1].ID());
|
||||
RegFirst = GetReg<RA_32>(Op->Lower.ID());
|
||||
RegSecond = GetReg<RA_32>(Op->Upper.ID());
|
||||
RegTmp = w0;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
Dst = GetSrcPair<RA_64>(Node);
|
||||
RegFirst = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
RegSecond = GetReg<RA_64>(Op->Header.Args[1].ID());
|
||||
RegFirst = GetReg<RA_64>(Op->Lower.ID());
|
||||
RegSecond = GetReg<RA_64>(Op->Upper.ID());
|
||||
RegTmp = x0;
|
||||
break;
|
||||
}
|
||||
@@ -70,7 +70,7 @@ DEF_OP(CreateElementPair) {
|
||||
|
||||
DEF_OP(Mov) {
|
||||
auto Op = IROp->C<IR::IROp_Mov>();
|
||||
mov(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Header.Args[0].ID()));
|
||||
mov(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Value.ID()));
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
+415
-415
File diff suppressed because it is too large.
Load diff
@@ -16,9 +16,11 @@ class CPUBackend;
|
||||
[[nodiscard]] std::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
void InitializeX86JITSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
CPUBackendFeatures GetX86JITBackendFeatures();
|
||||
|
||||
[[nodiscard]] std::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
void InitializeArm64JITSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
CPUBackendFeatures GetArm64JITBackendFeatures();
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
+236
-238
File diff suppressed because it is too large.
Load diff
+18
-26
@@ -29,13 +29,6 @@ $end_info$
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(GuestCallDirect) {
|
||||
LogMan::Msg::DFmt("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(GuestCallIndirect) {
|
||||
LogMan::Msg::DFmt("Unimplemented");
|
||||
}
|
||||
|
||||
DEF_OP(SignalReturn) {
|
||||
// Adjust the stack first for a regular return
|
||||
@@ -43,7 +36,7 @@ DEF_OP(SignalReturn) {
|
||||
add(rsp, SpillSlots * 16); // + 8 to consume return address
|
||||
}
|
||||
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.SignalReturnHandler)]);
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SignalReturnHandler)]);
|
||||
}
|
||||
|
||||
DEF_OP(CallbackReturn) {
|
||||
@@ -53,7 +46,7 @@ DEF_OP(CallbackReturn) {
|
||||
}
|
||||
|
||||
// Make sure to adjust the refcounter so we don't clear the cache now
|
||||
sub(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.SignalHandlerRefCountPointer)], 1);
|
||||
sub(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter)], 1);
|
||||
|
||||
// We need to adjust an additional 8 bytes to get back to the original "misaligned" RSP state
|
||||
add(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])], 8);
|
||||
@@ -91,14 +84,15 @@ DEF_OP(ExitFunction) {
|
||||
jmp(qword[rax]);
|
||||
|
||||
L(l_BranchHost);
|
||||
dq(Dispatcher->ExitFunctionLinkerAddress);
|
||||
//FEX_TODO(this is not per thread)
|
||||
dq(ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker);
|
||||
L(l_BranchGuest);
|
||||
dq(NewRIP);
|
||||
} else {
|
||||
Xbyak::Reg RipReg = GetSrc<RA_64>(Op->NewRIP.ID());
|
||||
|
||||
// L1 Cache
|
||||
mov(rcx, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.L1Pointer)]);
|
||||
mov(rcx, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer)]);
|
||||
|
||||
mov(rax, RipReg);
|
||||
|
||||
@@ -113,7 +107,7 @@ DEF_OP(ExitFunction) {
|
||||
|
||||
L(FullLookup);
|
||||
mov(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.rip)], RipReg);
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.DispatcherLoopTop)]);
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop)]);
|
||||
}
|
||||
|
||||
#ifdef BLOCKSTATS
|
||||
@@ -123,9 +117,9 @@ DEF_OP(ExitFunction) {
|
||||
|
||||
DEF_OP(Jump) {
|
||||
const auto Op = IROp->C<IR::IROp_Jump>();
|
||||
const auto ArgID = Op->Args(0).ID();
|
||||
const auto Target = Op->TargetBlock.ID();
|
||||
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(ArgID).first->second;
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(Target).first->second;
|
||||
}
|
||||
|
||||
#define GRCMP(Node) (Op->CompareSize == 4 ? GetSrc<RA_32>(Node) : GetSrc<RA_64>(Node))
|
||||
@@ -181,13 +175,13 @@ DEF_OP(Syscall) {
|
||||
}
|
||||
|
||||
mov(rsi, STATE); // Move thread in to rsi
|
||||
mov(rdi, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.SyscallHandlerObj)]);
|
||||
mov(rdi, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SyscallHandlerObj)]);
|
||||
mov(rdx, rsp);
|
||||
|
||||
if (NumPush & 1)
|
||||
sub(rsp, 8); // Align
|
||||
// {rdi, rsi, rdx}
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.SyscallHandlerFunc)]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SyscallHandlerFunc)]);
|
||||
|
||||
if (NumPush & 1)
|
||||
add(rsp, 8); // Align
|
||||
@@ -215,7 +209,7 @@ DEF_OP(Thunk) {
|
||||
if (NumPush & 1)
|
||||
sub(rsp, 8); // Align
|
||||
|
||||
mov(rdi, GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
mov(rdi, GetSrc<RA_64>(Op->ArgPtr.ID()));
|
||||
|
||||
auto thunkFn = ThreadState->CTX->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
|
||||
@@ -259,7 +253,7 @@ DEF_OP(ValidateCode) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(RemoveThreadCodeEntry) {
|
||||
DEF_OP(ThreadRemoveCodeEntry) {
|
||||
auto NumPush = RA64.size();
|
||||
|
||||
for (auto &Reg : RA64)
|
||||
@@ -272,7 +266,7 @@ DEF_OP(RemoveThreadCodeEntry) {
|
||||
mov(rax, Entry); // imm64 move
|
||||
mov(rsi, rax);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.RemoveThreadCodeEntryFromJIT)]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.ThreadRemoveCodeEntryFromJIT)]);
|
||||
|
||||
if (NumPush & 1)
|
||||
add(rsp, 8); // Align
|
||||
@@ -294,9 +288,9 @@ DEF_OP(CPUID) {
|
||||
// Result: RAX, RDX. 4xi32
|
||||
|
||||
// rsi can be in the source registers, so copy argument to edx first
|
||||
mov (edx, GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
mov (esi, GetSrc<RA_32>(Op->Header.Args[0].ID()));
|
||||
mov (rdi, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.CPUIDObj)]);
|
||||
mov (edx, GetSrc<RA_32>(Op->Leaf.ID()));
|
||||
mov (esi, GetSrc<RA_32>(Op->Function.ID()));
|
||||
mov (rdi, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj)]);
|
||||
|
||||
auto NumPush = RA64.size();
|
||||
|
||||
@@ -304,7 +298,7 @@ DEF_OP(CPUID) {
|
||||
sub(rsp, 8); // Align
|
||||
|
||||
// {rdi, rsi, rdx}
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.CPUIDFunction)]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDFunction)]);
|
||||
|
||||
if (NumPush & 1)
|
||||
add(rsp, 8); // Align
|
||||
@@ -320,8 +314,6 @@ DEF_OP(CPUID) {
|
||||
#undef DEF_OP
|
||||
void X86JITCore::RegisterBranchHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
|
||||
REGISTER_OP(GUESTCALLDIRECT, GuestCallDirect);
|
||||
REGISTER_OP(GUESTCALLINDIRECT, GuestCallIndirect);
|
||||
REGISTER_OP(SIGNALRETURN, SignalReturn);
|
||||
REGISTER_OP(CALLBACKRETURN, CallbackReturn);
|
||||
REGISTER_OP(EXITFUNCTION, ExitFunction);
|
||||
@@ -330,7 +322,7 @@ void X86JITCore::RegisterBranchHandlers() {
|
||||
REGISTER_OP(SYSCALL, Syscall);
|
||||
REGISTER_OP(THUNK, Thunk);
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(REMOVETHREADCODEENTRY, RemoveThreadCodeEntry);
|
||||
REGISTER_OP(THREADREMOVECODEENTRY, ThreadRemoveCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
|
||||
@@ -45,18 +45,18 @@ DEF_OP(VCastFromGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VCastFromGPR>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1:
|
||||
movzx(rax, GetSrc<RA_8>(Op->Header.Args[0].ID()));
|
||||
movzx(rax, GetSrc<RA_8>(Op->Src.ID()));
|
||||
vmovq(GetDst(Node), rax);
|
||||
break;
|
||||
case 2:
|
||||
movzx(rax, GetSrc<RA_16>(Op->Header.Args[0].ID()));
|
||||
movzx(rax, GetSrc<RA_16>(Op->Src.ID()));
|
||||
vmovq(GetDst(Node), rax);
|
||||
break;
|
||||
case 4:
|
||||
vmovd(GetDst(Node), GetSrc<RA_32>(Op->Header.Args[0].ID()).cvt32());
|
||||
vmovd(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()).cvt32());
|
||||
break;
|
||||
case 8:
|
||||
vmovq(GetDst(Node), GetSrc<RA_64>(Op->Header.Args[0].ID()).cvt64());
|
||||
vmovq(GetDst(Node), GetSrc<RA_64>(Op->Src.ID()).cvt64());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown VCastFromGPR element size: {}", Op->Header.ElementSize);
|
||||
}
|
||||
@@ -64,22 +64,23 @@ DEF_OP(VCastFromGPR) {
|
||||
|
||||
DEF_OP(Float_FromGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0404: { // Float <- int32_t
|
||||
cvtsi2ss(GetDst(Node), GetSrc<RA_32>(Op->Header.Args[0].ID()));
|
||||
cvtsi2ss(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- int64_t
|
||||
cvtsi2ss(GetDst(Node), GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
cvtsi2ss(GetDst(Node), GetSrc<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- int32_t
|
||||
cvtsi2sd(GetDst(Node), GetSrc<RA_32>(Op->Header.Args[0].ID()));
|
||||
cvtsi2sd(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 0x0808: { // Double <- int64_t
|
||||
cvtsi2sd(GetDst(Node), GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
cvtsi2sd(GetDst(Node), GetSrc<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
}
|
||||
@@ -87,14 +88,15 @@ DEF_OP(Float_FromGPR_S) {
|
||||
|
||||
DEF_OP(Float_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FToF>();
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- Float
|
||||
cvtss2sd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
cvtss2sd(GetDst(Node), GetSrc(Op->Scalar.ID()));
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
cvtsd2ss(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
cvtsd2ss(GetDst(Node), GetSrc(Op->Scalar.ID()));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Float_FToF sizes: 0x{:x}", Conv);
|
||||
@@ -105,7 +107,7 @@ DEF_OP(Vector_SToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
cvtdq2ps(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
cvtdq2ps(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
case 8:
|
||||
// This operation is a bit disgusting in x86
|
||||
@@ -113,8 +115,8 @@ DEF_OP(Vector_SToF) {
|
||||
// 1) First extract the top 64bits
|
||||
// 2) Do a scalar conversion on each
|
||||
// 3) Make sure to merge them together at the end
|
||||
pextrq(rax, GetSrc(Op->Header.Args[0].ID()), 1);
|
||||
pextrq(rcx, GetSrc(Op->Header.Args[0].ID()), 0);
|
||||
pextrq(rax, GetSrc(Op->Vector.ID()), 1);
|
||||
pextrq(rcx, GetSrc(Op->Vector.ID()), 0);
|
||||
cvtsi2sd(GetDst(Node), rcx);
|
||||
cvtsi2sd(xmm15, rax);
|
||||
movlhps(GetDst(Node), xmm15);
|
||||
@@ -127,10 +129,10 @@ DEF_OP(Vector_FToZS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
cvttps2dq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
cvttps2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
case 8:
|
||||
cvttpd2dq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
cvttpd2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", Op->Header.ElementSize);
|
||||
}
|
||||
@@ -140,10 +142,10 @@ DEF_OP(Vector_FToS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
cvtps2dq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
cvtps2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
case 8:
|
||||
cvtpd2dq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
cvtpd2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", Op->Header.ElementSize);
|
||||
}
|
||||
@@ -151,15 +153,15 @@ DEF_OP(Vector_FToS) {
|
||||
|
||||
DEF_OP(Vector_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- Float
|
||||
cvtps2pd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
cvtps2pd(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
cvtpd2ps(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
cvtpd2ps(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToF conversion type : 0x{:04x}", Conv); break;
|
||||
@@ -190,10 +192,10 @@ DEF_OP(Vector_FToI) {
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
roundps(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), RoundMode);
|
||||
roundps(GetDst(Node), GetSrc(Op->Vector.ID()), RoundMode);
|
||||
break;
|
||||
case 8:
|
||||
roundpd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), RoundMode);
|
||||
roundpd(GetDst(Node), GetSrc(Op->Vector.ID()), RoundMode);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -17,44 +17,44 @@ namespace FEXCore::CPU {
|
||||
|
||||
DEF_OP(AESImc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESImc>();
|
||||
vaesimc(GetDst(Node), GetSrc(Op->Header.Args[0].ID()));
|
||||
vaesimc(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(AESEnc) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
vaesenc(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()));
|
||||
vaesenc(GetDst(Node), GetSrc(Op->State.ID()), GetSrc(Op->Key.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(AESEncLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
vaesenclast(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()));
|
||||
vaesenclast(GetDst(Node), GetSrc(Op->State.ID()), GetSrc(Op->Key.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(AESDec) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
vaesdec(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()));
|
||||
vaesdec(GetDst(Node), GetSrc(Op->State.ID()), GetSrc(Op->Key.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(AESDecLast) {
|
||||
auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
vaesdeclast(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()));
|
||||
vaesdeclast(GetDst(Node), GetSrc(Op->State.ID()), GetSrc(Op->Key.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(AESKeyGenAssist) {
|
||||
auto Op = IROp->C<IR::IROp_VAESKeyGenAssist>();
|
||||
vaeskeygenassist(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), Op->RCON);
|
||||
vaeskeygenassist(GetDst(Node), GetSrc(Op->Src.ID()), Op->RCON);
|
||||
}
|
||||
|
||||
DEF_OP(CRC32) {
|
||||
auto Op = IROp->C<IR::IROp_CRC32>();
|
||||
switch (IROp->Size) {
|
||||
case 4:
|
||||
mov(TMP1, GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
mov(GetDst<RA_32>(Node), GetSrc<RA_32>(Op->Header.Args[0].ID()));
|
||||
mov(TMP1, GetSrc<RA_32>(Op->Src2.ID()));
|
||||
mov(GetDst<RA_32>(Node), GetSrc<RA_32>(Op->Src1.ID()));
|
||||
break;
|
||||
case 8:
|
||||
mov(TMP1, GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
mov(GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
mov(TMP1, GetSrc<RA_64>(Op->Src2.ID()));
|
||||
mov(GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Src1.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown CRC32 size: {}", IROp->Size);
|
||||
}
|
||||
@@ -75,16 +75,37 @@ DEF_OP(CRC32) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(PCLMUL) {
|
||||
auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
|
||||
auto Dst = GetDst(Node);
|
||||
auto Src1 = GetSrc(Op->Src1.ID());
|
||||
auto Src2 = GetSrc(Op->Src2.ID());
|
||||
|
||||
switch (Op->Selector) {
|
||||
case 0b00000000:
|
||||
case 0b00000001:
|
||||
case 0b00010000:
|
||||
case 0b00010001:
|
||||
vpclmulqdq(Dst, Src1, Src2, Op->Selector);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown PCLMUL selector: {}", Op->Selector);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void X86JITCore::RegisterEncryptionHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
|
||||
REGISTER_OP(VAESIMC, AESImc);
|
||||
REGISTER_OP(VAESENC, AESEnc);
|
||||
REGISTER_OP(VAESENCLAST, AESEncLast);
|
||||
REGISTER_OP(VAESDEC, AESDec);
|
||||
REGISTER_OP(VAESDECLAST, AESDecLast);
|
||||
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
|
||||
REGISTER_OP(VAESIMC, AESImc);
|
||||
REGISTER_OP(VAESENC, AESEnc);
|
||||
REGISTER_OP(VAESENCLAST, AESEncLast);
|
||||
REGISTER_OP(VAESDEC, AESDec);
|
||||
REGISTER_OP(VAESDECLAST, AESDecLast);
|
||||
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
|
||||
REGISTER_OP(CRC32, CRC32);
|
||||
REGISTER_OP(PCLMUL, PCLMUL);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
@@ -18,7 +18,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(GetHostFlag) {
|
||||
auto Op = IROp->C<IR::IROp_GetHostFlag>();
|
||||
|
||||
mov(rax, GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
mov(rax, GetSrc<RA_64>(Op->Value.ID()));
|
||||
shr(rax, Op->Flag);
|
||||
and_(rax, 1);
|
||||
mov(GetDst<RA_64>(Node), rax);
|
||||
|
||||
+84
-172
@@ -43,6 +43,9 @@ $end_info$
|
||||
// #define DEBUG_RA 1
|
||||
// #define DEBUG_CYCLES
|
||||
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 256;
|
||||
|
||||
namespace {
|
||||
static void PrintValue(uint64_t Value) {
|
||||
LogMan::Msg::DFmt("Value: 0x{:x}", Value);
|
||||
@@ -55,26 +58,6 @@ static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) {
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
CodeBuffer AllocateNewCodeBuffer(FEXCore::Context::Context *CTX, size_t Size) {
|
||||
CodeBuffer Buffer;
|
||||
Buffer.Size = Size;
|
||||
Buffer.Ptr = static_cast<uint8_t*>(
|
||||
FEXCore::Allocator::mmap(nullptr,
|
||||
Buffer.Size,
|
||||
PROT_READ | PROT_WRITE | PROT_EXEC,
|
||||
MAP_PRIVATE | MAP_ANONYMOUS,
|
||||
-1, 0));
|
||||
LOGMAN_THROW_A_FMT(Buffer.Ptr != reinterpret_cast<uint8_t*>(~0ULL), "Couldn't allocate code buffer");
|
||||
if (CTX->Config.GlobalJITNaming()) {
|
||||
CTX->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
return Buffer;
|
||||
}
|
||||
|
||||
void FreeCodeBuffer(CodeBuffer Buffer) {
|
||||
FEXCore::Allocator::munmap(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
|
||||
void X86JITCore::PushRegs() {
|
||||
sub(rsp, 16 * RAXMM_x.size());
|
||||
for (size_t i = 0; i < RAXMM_x.size(); ++i) {
|
||||
@@ -115,7 +98,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
case FABI_VOID_U16: {
|
||||
PushRegs();
|
||||
mov(edi, GetSrc<RA_32>(IROp->Args[0].ID()));
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
break;
|
||||
@@ -124,7 +107,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushRegs();
|
||||
|
||||
movss(xmm0, GetSrc(IROp->Args[0].ID()));
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -138,7 +121,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushRegs();
|
||||
|
||||
movsd(xmm0, GetSrc(IROp->Args[0].ID()));
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -153,7 +136,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushRegs();
|
||||
|
||||
mov(edi, GetSrc<RA_32>(IROp->Args[0].ID()));
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -169,7 +152,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -183,7 +166,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -196,7 +179,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
|
||||
movsd(xmm0, GetSrc(IROp->Args[0].ID()));
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -210,7 +193,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
movsd(xmm0, GetSrc(IROp->Args[0].ID()));
|
||||
movsd(xmm1, GetSrc(IROp->Args[1].ID()));
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -224,7 +207,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -237,7 +220,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -250,7 +233,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -266,7 +249,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
movq(rdx, GetSrc(IROp->Args[1].ID()));
|
||||
pextrq(rcx, GetSrc(IROp->Args[1].ID()), 1);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -279,7 +262,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -297,7 +280,7 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
movq(rdx, GetSrc(IROp->Args[1].ID()));
|
||||
pextrq(rcx, GetSrc(IROp->Args[1].ID()), 1);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
|
||||
@@ -317,17 +300,34 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
}
|
||||
}
|
||||
|
||||
static uint64_t X86JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
|
||||
auto Thread = Frame->Thread;
|
||||
auto GuestRip = record[1];
|
||||
|
||||
auto HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
|
||||
if (!HostCode) {
|
||||
Thread->CurrentFrame->State.rip = GuestRip;
|
||||
return Frame->Pointers.Common.DispatcherLoopTop;
|
||||
}
|
||||
|
||||
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
|
||||
Context::Context::ThreadAddBlockLink(Thread, GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
// undo the link
|
||||
record[0] = LinkerAddress;
|
||||
});
|
||||
|
||||
record[0] = HostCode;
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
void X86JITCore::Op_NoOp(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
}
|
||||
|
||||
X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, CodeBuffer Buffer)
|
||||
: CodeGenerator(Buffer.Size, Buffer.Ptr, nullptr)
|
||||
, CTX {ctx}
|
||||
, ThreadState {Thread}
|
||||
, InitialCodeBuffer {Buffer}
|
||||
{
|
||||
CurrentCodeBuffer = &InitialCodeBuffer;
|
||||
EmitDetectionString();
|
||||
X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
|
||||
: CPUBackend(Thread, INITIAL_CODE_SIZE, MAX_CODE_SIZE)
|
||||
, CodeGenerator(0, this, nullptr) // this is not used here
|
||||
, CTX {ctx} {
|
||||
|
||||
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
|
||||
@@ -356,69 +356,39 @@ X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalTh
|
||||
RegisterVectorHandlers();
|
||||
RegisterEncryptionHandlers();
|
||||
|
||||
DispatcherConfig config;
|
||||
config.ExitFunctionLink = reinterpret_cast<uintptr_t>(&ExitFunctionLink);
|
||||
config.ExitFunctionLinkThis = reinterpret_cast<uintptr_t>(this);
|
||||
config.StaticRegisterAssignment = ctx->Config.StaticRegisterAllocation;
|
||||
|
||||
Dispatcher = std::make_unique<X86Dispatcher>(CTX, ThreadState, config);
|
||||
DispatchPtr = Dispatcher->DispatchPtr;
|
||||
CallbackPtr = Dispatcher->CallbackPtr;
|
||||
|
||||
{
|
||||
// Set up pointers that the JIT needs to load
|
||||
auto &Pointers = ThreadState->CurrentFrame->Pointers.X86;
|
||||
// Process specific
|
||||
Pointers.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
|
||||
Pointers.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
|
||||
Pointers.RemoveThreadCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::Context::RemoveThreadCodeEntryFromJit);
|
||||
Pointers.CPUIDObj = reinterpret_cast<uint64_t>(&CTX->CPUID);
|
||||
auto &Common = ThreadState->CurrentFrame->Pointers.Common;
|
||||
|
||||
Common.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
|
||||
Common.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
|
||||
Common.ThreadRemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::Context::ThreadRemoveCodeEntryFromJit);
|
||||
Common.CPUIDObj = reinterpret_cast<uint64_t>(&CTX->CPUID);
|
||||
|
||||
{
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::CPUIDEmu::RunFunction);
|
||||
Pointers.CPUIDFunction = PMF.GetConvertedPointer();
|
||||
Common.CPUIDFunction = PMF.GetConvertedPointer();
|
||||
}
|
||||
|
||||
Pointers.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
|
||||
Pointers.SyscallHandlerFunc = reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall);
|
||||
Common.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
|
||||
Common.SyscallHandlerFunc = reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall);
|
||||
Common.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Context::Context::ThreadExitFunctionLink<X86JITCore_ExitFunctionLink>);
|
||||
|
||||
// Fill in the fallback handlers
|
||||
InterpreterOps::FillFallbackIndexPointers(Pointers.FallbackHandlerPointers);
|
||||
|
||||
// Thread Specific
|
||||
Pointers.SignalHandlerRefCountPointer = reinterpret_cast<uint64_t>(&Dispatcher->SignalHandlerRefCounter);
|
||||
InterpreterOps::FillFallbackIndexPointers(Common.FallbackHandlerPointers);
|
||||
}
|
||||
|
||||
// Must be done after Dispatcher init
|
||||
ClearCache();
|
||||
}
|
||||
|
||||
void X86JITCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGILL, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
X86JITCore *Core = reinterpret_cast<X86JITCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleSIGILL(Signal, info, ucontext);
|
||||
return Thread->CTX->Dispatcher->HandleSIGILL(Thread, Signal, info, ucontext);
|
||||
}, true);
|
||||
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SignalDelegator::SIGNAL_FOR_PAUSE, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
X86JITCore *Core = reinterpret_cast<X86JITCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleSignalPause(Signal, info, ucontext);
|
||||
}, true);
|
||||
|
||||
auto GuestSignalHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) -> bool {
|
||||
X86JITCore *Core = reinterpret_cast<X86JITCore*>(Thread->CPUBackend.get());
|
||||
return Core->Dispatcher->HandleGuestSignal(Signal, info, ucontext, GuestAction, GuestStack);
|
||||
};
|
||||
|
||||
for (uint32_t Signal = 0; Signal <= SignalDelegator::MAX_SIGNALS; ++Signal) {
|
||||
CTX->SignalDelegation->RegisterHostSignalHandlerForGuest(Signal, GuestSignalHandler);
|
||||
}
|
||||
}
|
||||
|
||||
X86JITCore::~X86JITCore() {
|
||||
for (auto CodeBuffer : CodeBuffers) {
|
||||
FreeCodeBuffer(CodeBuffer);
|
||||
}
|
||||
CodeBuffers.clear();
|
||||
|
||||
|
||||
FreeCodeBuffer(InitialCodeBuffer);
|
||||
}
|
||||
|
||||
void X86JITCore::EmitDetectionString() {
|
||||
@@ -429,51 +399,15 @@ void X86JITCore::EmitDetectionString() {
|
||||
}
|
||||
|
||||
void X86JITCore::ClearCache() {
|
||||
if (Dispatcher->SignalHandlerRefCounter == 0) {
|
||||
if (!CodeBuffers.empty()) {
|
||||
// If we have more than one code buffer we are tracking then walk them and delete
|
||||
// This is a cleanup step
|
||||
for (auto CodeBuffer : CodeBuffers) {
|
||||
FreeCodeBuffer(CodeBuffer);
|
||||
}
|
||||
CodeBuffers.clear();
|
||||
|
||||
// Set the current code buffer to the initial
|
||||
setNewBuffer(InitialCodeBuffer.Ptr, InitialCodeBuffer.Size);
|
||||
CurrentCodeBuffer = &InitialCodeBuffer;
|
||||
}
|
||||
|
||||
if (CurrentCodeBuffer->Size == MAX_CODE_SIZE) {
|
||||
// Rewind to the start of the code cache start
|
||||
reset();
|
||||
}
|
||||
else {
|
||||
FreeCodeBuffer(InitialCodeBuffer);
|
||||
|
||||
// Resize the code buffer and reallocate our code size
|
||||
CurrentCodeBuffer->Size *= 1.5;
|
||||
CurrentCodeBuffer->Size = std::min(CurrentCodeBuffer->Size, MAX_CODE_SIZE);
|
||||
|
||||
InitialCodeBuffer = AllocateNewCodeBuffer(CTX, CurrentCodeBuffer->Size);
|
||||
setNewBuffer(InitialCodeBuffer.Ptr, InitialCodeBuffer.Size);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// We have signal handlers that have generated code
|
||||
// This means that we can not safely clear the code at this point in time
|
||||
// Allocate some new code buffers that we can switch over to instead
|
||||
auto NewCodeBuffer = AllocateNewCodeBuffer(CTX, X86JITCore::INITIAL_CODE_SIZE);
|
||||
EmplaceNewCodeBuffer(NewCodeBuffer);
|
||||
setNewBuffer(NewCodeBuffer.Ptr, NewCodeBuffer.Size);
|
||||
}
|
||||
|
||||
auto CodeBuffer = GetEmptyCodeBuffer();
|
||||
setNewBuffer(CodeBuffer->Ptr, CodeBuffer->Size);
|
||||
EmitDetectionString();
|
||||
}
|
||||
|
||||
IR::PhysicalRegister X86JITCore::GetPhys(IR::NodeID Node) const {
|
||||
auto PhyReg = RAData->GetNodeRegister(Node);
|
||||
|
||||
LOGMAN_THROW_A_FMT(PhyReg.Raw != 255, "Couldn't Allocate register for node: ssa{}. Class: {}", Node, PhyReg.Class);
|
||||
LOGMAN_THROW_AA_FMT(PhyReg.Raw != 255, "Couldn't Allocate register for node: ssa{}. Class: {}", Node, PhyReg.Class);
|
||||
|
||||
return PhyReg;
|
||||
}
|
||||
@@ -635,43 +569,30 @@ std::tuple<X86JITCore::SetCC, X86JITCore::CMovCC, X86JITCore::JCC> X86JITCore::G
|
||||
return { &CodeGenerator::sete , &CodeGenerator::cmove , &CodeGenerator::je };
|
||||
}
|
||||
|
||||
void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) {
|
||||
JumpTargets.clear();
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
|
||||
this->Entry = Entry;
|
||||
this->RAData = RAData;
|
||||
this->DebugData = DebugData;
|
||||
|
||||
// Fairly excessive buffer range to make sure we don't overflow
|
||||
uint32_t BufferRange = SSACount * 16;
|
||||
uint32_t BufferRange = SSACount * 16 + GDBEnabled * Dispatcher::MaxGDBPauseCheckSize;
|
||||
if ((getSize() + BufferRange) > CurrentCodeBuffer->Size) {
|
||||
ThreadState->CTX->ClearCodeCache(ThreadState, false);
|
||||
CTX->ClearCodeCache(ThreadState);
|
||||
}
|
||||
|
||||
void *GuestEntry = getCurr<void*>();
|
||||
GuestEntry = getCurr<uint8_t*>();
|
||||
CursorEntry = getSize();
|
||||
this->IR = IR;
|
||||
|
||||
if (CTX->GetGdbServerStatus()) {
|
||||
Label RunBlock;
|
||||
|
||||
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
|
||||
// This happens when single stepping
|
||||
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
|
||||
mov(rax, reinterpret_cast<uint64_t>(CTX));
|
||||
|
||||
// If the value == 0 then branch to the top
|
||||
cmp(dword [rax + (offsetof(FEXCore::Context::Context, Config.RunningMode))], 0);
|
||||
je(RunBlock);
|
||||
// Else we need to pause now
|
||||
mov(rax, Dispatcher->ThreadPauseHandlerAddress);
|
||||
jmp(rax);
|
||||
ud2();
|
||||
|
||||
L(RunBlock);
|
||||
if (GDBEnabled) {
|
||||
auto GDBSize = CTX->Dispatcher->GenerateGDBPauseCheck(GuestEntry, Entry);
|
||||
setSize(getSize() + GDBSize);
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(RAData != nullptr, "Needs RA");
|
||||
LOGMAN_THROW_AA_FMT(RAData != nullptr, "Needs RA");
|
||||
|
||||
SpillSlots = RAData->SpillSlots();
|
||||
|
||||
@@ -726,12 +647,13 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
|
||||
using namespace FEXCore::IR;
|
||||
{
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
auto BlockIROp = BlockHeader->CW<IROp_CodeBlock>();
|
||||
LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
auto BlockIROp = BlockHeader->CW<IROp_CodeBlock>();
|
||||
LOGMAN_THROW_AA_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
|
||||
#endif
|
||||
|
||||
auto BlockStartHostCode = getCurr<uint8_t *>();
|
||||
{
|
||||
const auto Node = IR->GetID(BlockNode);
|
||||
const auto IsTarget = JumpTargets.try_emplace(Node).first;
|
||||
|
||||
@@ -786,6 +708,13 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
|
||||
OpHandler Handler = OpHandlers[IROp->Op];
|
||||
(this->*Handler)(IROp, ID);
|
||||
}
|
||||
|
||||
if (DebugData) {
|
||||
DebugData->Subblocks.push_back({
|
||||
static_cast<uint32_t>(BlockStartHostCode - GuestEntry),
|
||||
static_cast<uint32_t>(getCurr<uint8_t *>() - BlockStartHostCode)
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Make sure last branch is generated. It certainly can't be eliminated here.
|
||||
@@ -808,29 +737,12 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
|
||||
return GuestEntry;
|
||||
}
|
||||
|
||||
uint64_t X86JITCore::ExitFunctionLink(X86JITCore *core, FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
|
||||
auto Thread = Frame->Thread;
|
||||
auto GuestRip = record[1];
|
||||
|
||||
auto HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
|
||||
if (!HostCode) {
|
||||
Thread->CurrentFrame->State.rip = GuestRip;
|
||||
return core->Dispatcher->AbsoluteLoopTopAddress;
|
||||
}
|
||||
|
||||
auto LinkerAddress = core->Dispatcher->ExitFunctionLinkerAddress;
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
// undo the link
|
||||
record[0] = LinkerAddress;
|
||||
});
|
||||
|
||||
record[0] = HostCode;
|
||||
return HostCode;
|
||||
std::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return std::make_unique<X86JITCore>(ctx, Thread);
|
||||
}
|
||||
|
||||
std::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
|
||||
return std::make_unique<X86JITCore>(ctx, Thread, AllocateNewCodeBuffer(ctx, X86JITCore::INITIAL_CODE_SIZE));
|
||||
CPUBackendFeatures GetX86JITBackendFeatures() {
|
||||
return CPUBackendFeatures { };
|
||||
}
|
||||
|
||||
void InitializeX86JITSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
|
||||
+13
-40
@@ -6,6 +6,7 @@ $end_info$
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include "Interface/Core/BlockSamplingData.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/ObjectCache/Relocations.h"
|
||||
@@ -25,13 +26,6 @@ using namespace Xbyak;
|
||||
#include <tuple>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
struct CodeBuffer {
|
||||
uint8_t *Ptr;
|
||||
size_t Size;
|
||||
};
|
||||
|
||||
[[nodiscard]] CodeBuffer AllocateNewCodeBuffer(size_t Size);
|
||||
void FreeCodeBuffer(CodeBuffer Buffer);
|
||||
|
||||
// Temp registers
|
||||
// rax, rcx, rdx, rsi, r8, r9,
|
||||
@@ -58,8 +52,7 @@ const std::array<Xbyak::Xmm, 11> RAXMM_x = { xmm1, xmm2, xmm3, xmm4, xmm5, xmm6
|
||||
class X86JITCore final : public CPUBackend, public Xbyak::CodeGenerator {
|
||||
public:
|
||||
explicit X86JITCore(FEXCore::Context::Context *ctx,
|
||||
FEXCore::Core::InternalThreadState *Thread,
|
||||
CodeBuffer Buffer);
|
||||
FEXCore::Core::InternalThreadState *Thread);
|
||||
~X86JITCore() override;
|
||||
|
||||
[[nodiscard]] std::string GetName() override { return "JIT"; }
|
||||
@@ -67,7 +60,7 @@ public:
|
||||
[[nodiscard]] void *CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData) override;
|
||||
FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) override;
|
||||
|
||||
[[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
|
||||
|
||||
@@ -75,13 +68,6 @@ public:
|
||||
|
||||
void ClearCache() override;
|
||||
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 256;
|
||||
|
||||
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true) const override {
|
||||
return Dispatcher->IsAddressInJITCode(Address, IncludeDispatcher);
|
||||
}
|
||||
|
||||
static void InitializeSignalHandlers(FEXCore::Context::Context *CTX);
|
||||
|
||||
void ClearRelocations() override { Relocations.clear(); }
|
||||
@@ -150,9 +136,7 @@ private:
|
||||
|
||||
Label* PendingTargetLabel{};
|
||||
FEXCore::Context::Context *CTX;
|
||||
FEXCore::Core::InternalThreadState *ThreadState;
|
||||
FEXCore::IR::IRListView const *IR;
|
||||
std::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
uint64_t Entry;
|
||||
|
||||
std::unordered_map<IR::NodeID, Label> JumpTargets;
|
||||
@@ -205,33 +189,23 @@ private:
|
||||
|
||||
IR::RegisterAllocationPass *RAPass;
|
||||
FEXCore::IR::RegisterAllocationData *RAData;
|
||||
FEXCore::Core::DebugData *DebugData;
|
||||
|
||||
#ifdef BLOCKSTATS
|
||||
bool GetSamplingData {true};
|
||||
#endif
|
||||
|
||||
void EmplaceNewCodeBuffer(CodeBuffer Buffer) {
|
||||
CurrentCodeBuffer = &CodeBuffers.emplace_back(Buffer);
|
||||
}
|
||||
|
||||
static uint64_t ExitFunctionLink(X86JITCore* code, FEXCore::Core::CpuStateFrame *Frame, uint64_t *record);
|
||||
static uint64_t ExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record);
|
||||
|
||||
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
|
||||
void EmitDetectionString();
|
||||
|
||||
// This is the initial code buffer that we will fall back to
|
||||
// In a program without signals and code clearing, we will typically
|
||||
// only have this code buffer
|
||||
CodeBuffer InitialCodeBuffer{};
|
||||
// This is the array of /additional/ code buffers that we may need to allocate
|
||||
// Allocation only occurs when we've hit signals and need to clear code cache
|
||||
// For code safety we can't delete code buffers until outside of all signals
|
||||
std::vector<CodeBuffer> CodeBuffers{};
|
||||
|
||||
// This is the current code buffer that we are tracking
|
||||
CodeBuffer *CurrentCodeBuffer{};
|
||||
|
||||
uint32_t SpillSlots{};
|
||||
/**
|
||||
* @brief Current guest RIP entrypoint
|
||||
*/
|
||||
uint8_t *GuestEntry{};
|
||||
|
||||
using SetCC = void (X86JITCore::*)(const Operand& op);
|
||||
using CMovCC = void (X86JITCore::*)(const Reg& reg, const Operand& op);
|
||||
using JCC = void (X86JITCore::*)(const Label& label, LabelType type);
|
||||
@@ -330,8 +304,6 @@ private:
|
||||
DEF_OP(AtomicFetchNeg);
|
||||
|
||||
///< Branch ops
|
||||
DEF_OP(GuestCallDirect);
|
||||
DEF_OP(GuestCallIndirect);
|
||||
DEF_OP(SignalReturn);
|
||||
DEF_OP(CallbackReturn);
|
||||
DEF_OP(ExitFunction);
|
||||
@@ -340,7 +312,7 @@ private:
|
||||
DEF_OP(Syscall);
|
||||
DEF_OP(Thunk);
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(RemoveThreadCodeEntry);
|
||||
DEF_OP(ThreadRemoveCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
|
||||
///< Conversion ops
|
||||
@@ -375,7 +347,7 @@ private:
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
///< Misc ops
|
||||
DEF_OP(EndBlock);
|
||||
DEF_OP(GuestOpcode);
|
||||
DEF_OP(Fence);
|
||||
DEF_OP(Break);
|
||||
DEF_OP(Phi);
|
||||
@@ -491,6 +463,7 @@ private:
|
||||
DEF_OP(AESDecLast);
|
||||
DEF_OP(AESKeyGenAssist);
|
||||
DEF_OP(CRC32);
|
||||
DEF_OP(PCLMUL);
|
||||
#undef DEF_OP
|
||||
};
|
||||
|
||||
|
||||
@@ -4,6 +4,7 @@ tags: backend|x86-64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
|
||||
+35
-53
@@ -7,6 +7,7 @@ $end_info$
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
@@ -20,6 +21,12 @@ $end_info$
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(GuestOpcode) {
|
||||
auto Op = IROp->C<IR::IROp_GuestOpcode>();
|
||||
// metadata
|
||||
DebugData->GuestOpcodes.push_back({Op->GuestEntryOffset, getCurr<uint8_t*>() - GuestEntry});
|
||||
}
|
||||
|
||||
DEF_OP(Fence) {
|
||||
auto Op = IROp->C<IR::IROp_Fence>();
|
||||
switch (Op->Fence) {
|
||||
@@ -38,56 +45,30 @@ DEF_OP(Fence) {
|
||||
|
||||
DEF_OP(Break) {
|
||||
auto Op = IROp->C<IR::IROp_Break>();
|
||||
switch (Op->Reason) {
|
||||
case FEXCore::IR::Break_Unimplemented: // Hard fault
|
||||
case FEXCore::IR::Break_Interrupt: // Guest ud2
|
||||
ud2();
|
||||
break;
|
||||
case FEXCore::IR::Break_Overflow: // overflow
|
||||
// Need to be outside of JIT cache space to ensure cache clearing correctness
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.OverflowExceptionHandler)]);
|
||||
break;
|
||||
case FEXCore::IR::Break_Halt: { // HLT
|
||||
// Time to quit
|
||||
// Set our stack to the starting stack location
|
||||
mov(rsp, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)]);
|
||||
|
||||
// Now we need to jump to the thread stop handler
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.ThreadStopHandler)]);
|
||||
break;
|
||||
}
|
||||
case FEXCore::IR::Break_Interrupt3: // INT3
|
||||
{
|
||||
if (CTX->GetGdbServerStatus()) {
|
||||
// Adjust the stack first for a regular return
|
||||
if (SpillSlots) {
|
||||
add(rsp, SpillSlots * 16);
|
||||
}
|
||||
if (SpillSlots) {
|
||||
add(rsp, SpillSlots * 16);
|
||||
}
|
||||
|
||||
// This jump target needs to be a constant offset here
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.ThreadPauseHandler)]);
|
||||
}
|
||||
else {
|
||||
// If we don't have a gdb server attached then....crash?
|
||||
// Treat this case like HLT
|
||||
mov(rsp, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)]);
|
||||
mov(byte [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.FaultToTopAndGeneratedException)], 1);
|
||||
mov(byte [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.Signal)], Op->Reason.Signal);
|
||||
mov(dword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.TrapNo)], Op->Reason.TrapNumber);
|
||||
mov(dword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.err_code)], Op->Reason.ErrorRegister);
|
||||
mov(dword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.si_code)], Op->Reason.si_code);
|
||||
|
||||
// Now we need to jump to the thread stop handler
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.ThreadStopHandler)]);
|
||||
}
|
||||
switch (Op->Reason.Signal) {
|
||||
case SIGILL:
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGILL)]);
|
||||
break;
|
||||
case SIGTRAP:
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP)]);
|
||||
break;
|
||||
case SIGSEGV:
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGSEGV)]);
|
||||
break;
|
||||
default:
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP)]);
|
||||
break;
|
||||
}
|
||||
case FEXCore::IR::Break_InvalidInstruction:
|
||||
{
|
||||
if (SpillSlots) {
|
||||
add(rsp, SpillSlots * 16);
|
||||
}
|
||||
|
||||
// Need to be outside of JIT cache space to ensure cache clearing correctness
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.UnimplementedInstructionHandler)]);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Break reason: {}", Op->Reason);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -103,7 +84,7 @@ DEF_OP(GetRoundingMode) {
|
||||
|
||||
DEF_OP(SetRoundingMode) {
|
||||
auto Op = IROp->C<IR::IROp_SetRoundingMode>();
|
||||
auto Src = GetSrc<RA_32>(Op->Header.Args[0].ID());
|
||||
auto Src = GetSrc<RA_32>(Op->RoundMode.ID());
|
||||
|
||||
// Load old mxcsr
|
||||
// Only stores to memory
|
||||
@@ -128,15 +109,15 @@ DEF_OP(Print) {
|
||||
auto Op = IROp->C<IR::IROp_Print>();
|
||||
|
||||
PushRegs();
|
||||
if (IsGPR(Op->Header.Args[0].ID())) {
|
||||
mov (rdi, GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.PrintValue)]);
|
||||
if (IsGPR(Op->Value.ID())) {
|
||||
mov (rdi, GetSrc<RA_64>(Op->Value.ID()));
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintValue)]);
|
||||
}
|
||||
else {
|
||||
pextrq(rdi, GetSrc(Op->Header.Args[0].ID()), 0);
|
||||
pextrq(rsi, GetSrc(Op->Header.Args[0].ID()), 1);
|
||||
pextrq(rdi, GetSrc(Op->Value.ID()), 0);
|
||||
pextrq(rsi, GetSrc(Op->Value.ID()), 1);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.PrintVectorValue)]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintVectorValue)]);
|
||||
}
|
||||
|
||||
PopRegs();
|
||||
@@ -178,6 +159,7 @@ void X86JITCore::RegisterMiscHandlers() {
|
||||
REGISTER_OP(CODEBLOCK, NoOp);
|
||||
REGISTER_OP(BEGINBLOCK, NoOp);
|
||||
REGISTER_OP(ENDBLOCK, NoOp);
|
||||
REGISTER_OP(GUESTOPCODE, GuestOpcode);
|
||||
REGISTER_OP(FENCE, Fence);
|
||||
REGISTER_OP(BREAK, Break);
|
||||
REGISTER_OP(PHI, NoOp);
|
||||
|
||||
@@ -20,13 +20,13 @@ DEF_OP(ExtractElementPair) {
|
||||
auto Op = IROp->C<IR::IROp_ExtractElementPair>();
|
||||
switch (Op->Header.Size) {
|
||||
case 4: {
|
||||
auto Src = GetSrcPair<RA_32>(Op->Header.Args[0].ID());
|
||||
auto Src = GetSrcPair<RA_32>(Op->Pair.ID());
|
||||
std::array<Xbyak::Reg, 2> Regs = {Src.first, Src.second};
|
||||
mov (GetDst<RA_32>(Node), Regs[Op->Element]);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
auto Src = GetSrcPair<RA_64>(Op->Header.Args[0].ID());
|
||||
auto Src = GetSrcPair<RA_64>(Op->Pair.ID());
|
||||
std::array<Xbyak::Reg, 2> Regs = {Src.first, Src.second};
|
||||
mov (GetDst<RA_64>(Node), Regs[Op->Element]);
|
||||
break;
|
||||
@@ -45,15 +45,15 @@ DEF_OP(CreateElementPair) {
|
||||
switch (IROp->ElementSize) {
|
||||
case 4: {
|
||||
Dst = GetSrcPair<RA_32>(Node);
|
||||
RegFirst = GetSrc<RA_32>(Op->Header.Args[0].ID());
|
||||
RegSecond = GetSrc<RA_32>(Op->Header.Args[1].ID());
|
||||
RegFirst = GetSrc<RA_32>(Op->Lower.ID());
|
||||
RegSecond = GetSrc<RA_32>(Op->Upper.ID());
|
||||
RegTmp = eax;
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
Dst = GetSrcPair<RA_64>(Node);
|
||||
RegFirst = GetSrc<RA_64>(Op->Header.Args[0].ID());
|
||||
RegSecond = GetSrc<RA_64>(Op->Header.Args[1].ID());
|
||||
RegFirst = GetSrc<RA_64>(Op->Lower.ID());
|
||||
RegSecond = GetSrc<RA_64>(Op->Upper.ID());
|
||||
RegTmp = rax;
|
||||
break;
|
||||
}
|
||||
@@ -75,7 +75,7 @@ DEF_OP(CreateElementPair) {
|
||||
|
||||
DEF_OP(Mov) {
|
||||
auto Op = IROp->C<IR::IROp_Mov>();
|
||||
mov (GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
mov (GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Value.ID()));
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
+302
-302
File diff suppressed because it is too large.
Load diff
@@ -4,6 +4,7 @@ tags: backend|x86-64
|
||||
desc: relocation logic of the x86-64 splatter backend
|
||||
$end_info$
|
||||
*/
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/JIT/x86_64/JITClass.h"
|
||||
#include "Interface/HLE/Thunks/Thunks.h"
|
||||
|
||||
@@ -11,7 +12,7 @@ namespace FEXCore::CPU {
|
||||
uint64_t X86JITCore::GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
switch (Op) {
|
||||
case FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER:
|
||||
return Dispatcher->ExitFunctionLinkerAddress;
|
||||
return ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker;
|
||||
break;
|
||||
default:
|
||||
ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op));
|
||||
@@ -90,7 +91,7 @@ bool X86JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint6
|
||||
size_t DataIndex{};
|
||||
for (size_t j = 0; j < NumRelocations; ++j) {
|
||||
const FEXCore::CPU::Relocation *Reloc = reinterpret_cast<const FEXCore::CPU::Relocation *>(&EntryRelocations[DataIndex]);
|
||||
LOGMAN_THROW_A_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
|
||||
LOGMAN_THROW_AA_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
|
||||
|
||||
switch (Reloc->Header.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
|
||||
|
||||
+2
-10
@@ -37,11 +37,11 @@ LookupCache::LookupCache(FEXCore::Context::Context *CTX)
|
||||
// We currently limit to 128MB of real memory for caching for the total cache size.
|
||||
// Can end up being inefficient if we compile a small number of blocks per page
|
||||
PageMemory = reinterpret_cast<uintptr_t>(FEXCore::Allocator::mmap(nullptr, CODE_SIZE, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
|
||||
LOGMAN_THROW_A_FMT(PageMemory != -1ULL, "Failed to allocate page memory");
|
||||
LOGMAN_THROW_AA_FMT(PageMemory != -1ULL, "Failed to allocate page memory");
|
||||
|
||||
// L1 Cache
|
||||
L1Pointer = reinterpret_cast<uintptr_t>(FEXCore::Allocator::mmap(nullptr, L1_SIZE, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
|
||||
LOGMAN_THROW_A_FMT(L1Pointer != -1ULL, "Failed to allocate L1Pointer");
|
||||
LOGMAN_THROW_AA_FMT(L1Pointer != -1ULL, "Failed to allocate L1Pointer");
|
||||
|
||||
VirtualMemSize = ctx->Config.VirtualMemSize;
|
||||
}
|
||||
@@ -52,14 +52,6 @@ LookupCache::~LookupCache() {
|
||||
FEXCore::Allocator::munmap(reinterpret_cast<void*>(L1Pointer), L1_SIZE);
|
||||
}
|
||||
|
||||
void LookupCache::HintUsedRange(uint64_t Address, uint64_t Size) {
|
||||
// Tell the kernel we will definitely need [Address, Address+Size) mapped for the page pointer
|
||||
// Page Pointer is allocated per page, so shift by page size
|
||||
Address >>= 12;
|
||||
Size >>= 12;
|
||||
madvise(reinterpret_cast<void*>(PagePointer + Address), Size, MADV_WILLNEED);
|
||||
}
|
||||
|
||||
void LookupCache::ClearL2Cache() {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
// Clear out the page memory
|
||||
|
||||
+17
-18
@@ -25,9 +25,6 @@ public:
|
||||
LookupCache(FEXCore::Context::Context *CTX);
|
||||
~LookupCache();
|
||||
|
||||
using LookupCacheIter = uintptr_t;
|
||||
uintptr_t End() { return 0; }
|
||||
|
||||
uintptr_t FindBlock(uint64_t Address) {
|
||||
// Try L1, no lock needed
|
||||
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
@@ -72,30 +69,34 @@ public:
|
||||
|
||||
std::map<uint64_t, std::vector<uint64_t>> CodePages;
|
||||
|
||||
// Appends Block {Address} to CodePages [Start, Start + Length)
|
||||
// Returns true if new pages are marked as containing code
|
||||
bool AddBlockMapping(uint64_t Address, void *HostCode, uint64_t Start, uint64_t Length) {
|
||||
bool AddBlockExecutableRange(uint64_t Address, uint64_t Start, uint64_t Length) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
auto InsertPoint =
|
||||
#endif
|
||||
BlockList.emplace(Address, (uintptr_t)HostCode);
|
||||
LOGMAN_THROW_A_FMT(InsertPoint.second == true, "Dupplicate block mapping added");
|
||||
|
||||
bool rv = false;
|
||||
|
||||
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length) >> 12; CurrentPage <= EndPage; CurrentPage++) {
|
||||
rv |= CodePages[CurrentPage].size() == 0;
|
||||
CodePages[CurrentPage].push_back(Address);
|
||||
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length -1) >> 12; CurrentPage <= EndPage; CurrentPage++) {
|
||||
auto &CodePage = CodePages[CurrentPage];
|
||||
rv |= CodePage.size() == 0;
|
||||
CodePage.push_back(Address);
|
||||
}
|
||||
|
||||
return rv;
|
||||
}
|
||||
|
||||
// Adds to Guest -> Host code mapping
|
||||
void AddBlockMapping(uint64_t Address, void *HostCode) {
|
||||
std::lock_guard<std::recursive_mutex> lk(WriteLock);
|
||||
|
||||
[[maybe_unused]] auto Inserted = BlockList.emplace(Address, (uintptr_t)HostCode).second;
|
||||
LOGMAN_THROW_AA_FMT(Inserted, "Duplicate block mapping added");
|
||||
|
||||
// There is no need to update L1 or L2, they will get updated on first lookup
|
||||
// However, adding to L1 here increases performance
|
||||
auto &L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
L1Entry.GuestCode = Address;
|
||||
L1Entry.HostCode = (uintptr_t)HostCode;
|
||||
|
||||
return rv;
|
||||
}
|
||||
|
||||
void Erase(uint64_t Address) {
|
||||
@@ -149,8 +150,6 @@ public:
|
||||
void ClearCache();
|
||||
void ClearL2Cache();
|
||||
|
||||
void HintUsedRange(uint64_t Address, uint64_t Size);
|
||||
|
||||
uintptr_t GetL1Pointer() const { return L1Pointer; }
|
||||
uintptr_t GetPagePointer() const { return PagePointer; }
|
||||
uintptr_t GetVirtualMemorySize() const { return VirtualMemSize; }
|
||||
@@ -158,7 +157,7 @@ public:
|
||||
constexpr static size_t L1_ENTRIES = 1 * 1024 * 1024; // Must be a power of 2
|
||||
constexpr static size_t L1_ENTRIES_MASK = L1_ENTRIES - 1;
|
||||
|
||||
// This needs to be taken before reads or writes to L2, L3, CodePages, Thread::LocalIRCache,
|
||||
// This needs to be taken before reads or writes to L2, L3, CodePages, Thread::DebugStore,
|
||||
// and before writes to L1. Concurrent access from a thread that this LookupCache doesn't belong to
|
||||
// may only happen during cross thread invalidation (::Erase).
|
||||
// All other operations must be done from the owning thread.
|
||||
|
||||
+97
-37
@@ -17,6 +17,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <algorithm>
|
||||
@@ -350,7 +351,7 @@ void OpDispatchBuilder::SecondaryALUOp(OpcodeArgs) {
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Atomic IR Op: {}", IROp);
|
||||
LOGMAN_MSG_A_FMT("Unknown Atomic IR Op: {}", ToUnderlying(IROp));
|
||||
break;
|
||||
}
|
||||
}
|
||||
@@ -1589,7 +1590,12 @@ void OpDispatchBuilder::MOVSegOp(OpcodeArgs) {
|
||||
case 2: // CS
|
||||
case FEXCore::X86State::REG_R9: // CS
|
||||
// CPL3 can't write to this
|
||||
_Break(FEXCore::IR::Break_InvalidInstruction, 0);
|
||||
_Break(FEXCore::IR::BreakDefinition {
|
||||
.ErrorRegister = 0,
|
||||
.Signal = SIGILL,
|
||||
.TrapNumber = 0,
|
||||
.si_code = 0,
|
||||
});
|
||||
break;
|
||||
case 3: // SS
|
||||
case FEXCore::X86State::REG_R10: // SS
|
||||
@@ -4503,7 +4509,7 @@ void OpDispatchBuilder::Finalize() {
|
||||
|
||||
[[maybe_unused]] const FEXCore::IR::IROp_Header *IROp =
|
||||
RealNode->Op(DualListData.DataBegin());
|
||||
LOGMAN_THROW_A_FMT(IROp->Op == OP_IRHEADER, "First op in function must be our header");
|
||||
LOGMAN_THROW_AA_FMT(IROp->Op == OP_IRHEADER, "First op in function must be our header");
|
||||
|
||||
// Let's walk the jump blocks and see if we have handled every block target
|
||||
for (auto &Handler : JumpTargets) {
|
||||
@@ -4529,7 +4535,7 @@ uint8_t OpDispatchBuilder::GetDstSize(X86Tables::DecodedOp Op) const {
|
||||
|
||||
const uint32_t DstSizeFlag = X86Tables::DecodeFlags::GetSizeDstFlags(Op->Flags);
|
||||
const uint8_t Size = Sizes[DstSizeFlag];
|
||||
LOGMAN_THROW_A_FMT(Size != 0, "Invalid destination size for op");
|
||||
LOGMAN_THROW_AA_FMT(Size != 0, "Invalid destination size for op");
|
||||
return Size;
|
||||
}
|
||||
|
||||
@@ -4547,7 +4553,7 @@ uint8_t OpDispatchBuilder::GetSrcSize(X86Tables::DecodedOp Op) const {
|
||||
|
||||
const uint32_t SrcSizeFlag = X86Tables::DecodeFlags::GetSizeSrcFlags(Op->Flags);
|
||||
const uint8_t Size = Sizes[SrcSizeFlag];
|
||||
LOGMAN_THROW_A_FMT(Size != 0, "Invalid destination size for op");
|
||||
LOGMAN_THROW_AA_FMT(Size != 0, "Invalid destination size for op");
|
||||
return Size;
|
||||
}
|
||||
|
||||
@@ -4641,7 +4647,14 @@ OrderedNode *OpDispatchBuilder::LoadSource_WithOpSize(FEXCore::IR::RegisterClass
|
||||
Src = _LoadContext(OpSize, FPRClass, offsetof(FEXCore::Core::CPUState, mm[gpr - FEXCore::X86State::REG_MM_0]));
|
||||
}
|
||||
else if (gpr >= FEXCore::X86State::REG_XMM_0) {
|
||||
Src = _LoadContext(OpSize, FPRClass, offsetof(FEXCore::Core::CPUState, xmm[gpr - FEXCore::X86State::REG_XMM_0][Operand.Data.GPR.HighBits ? 1 : 0]));
|
||||
const auto gprIndex = gpr - X86State::REG_XMM_0;
|
||||
const auto highIndex = Operand.Data.GPR.HighBits ? 1 : 0;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAVX) {
|
||||
Src = _LoadContext(OpSize, FPRClass, offsetof(Core::CPUState, xmm.avx.data[gprIndex][highIndex]));
|
||||
} else {
|
||||
Src = _LoadContext(OpSize, FPRClass, offsetof(Core::CPUState, xmm.sse.data[gprIndex][highIndex]));
|
||||
}
|
||||
}
|
||||
else {
|
||||
Src = _LoadContext(OpSize, GPRClass, offsetof(FEXCore::Core::CPUState, gregs[gpr]) + (Operand.Data.GPR.HighBits ? 1 : 0));
|
||||
@@ -4783,7 +4796,14 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
_StoreContext(OpSize, Class, Src, offsetof(FEXCore::Core::CPUState, mm[gpr - FEXCore::X86State::REG_MM_0]));
|
||||
}
|
||||
else if (gpr >= FEXCore::X86State::REG_XMM_0) {
|
||||
_StoreContext(OpSize, Class, Src, offsetof(FEXCore::Core::CPUState, xmm[gpr - FEXCore::X86State::REG_XMM_0][Operand.Data.GPR.HighBits ? 1 : 0]));
|
||||
const auto gprIndex = gpr - X86State::REG_XMM_0;
|
||||
const auto highIndex = Operand.Data.GPR.HighBits ? 1 : 0;
|
||||
|
||||
if (CTX->HostFeatures.SupportsAVX) {
|
||||
_StoreContext(OpSize, Class, Src, offsetof(Core::CPUState, xmm.avx.data[gprIndex][highIndex]));
|
||||
} else {
|
||||
_StoreContext(OpSize, Class, Src, offsetof(Core::CPUState, xmm.sse.data[gprIndex][highIndex]));
|
||||
}
|
||||
}
|
||||
else {
|
||||
if (GPRSize == 8 && OpSize == 4) {
|
||||
@@ -4791,11 +4811,11 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
// For all other sizes, the upper bits are guaranteed to already be zero
|
||||
OrderedNode *Value = GetOpSize(Src) == 8 ? _Bfe(4, 32, 0, Src) : Src;
|
||||
|
||||
LOGMAN_THROW_A_FMT(!Operand.Data.GPR.HighBits, "Can't handle 32bit store to high 8bit register");
|
||||
LOGMAN_THROW_AA_FMT(!Operand.Data.GPR.HighBits, "Can't handle 32bit store to high 8bit register");
|
||||
_StoreContext(GPRSize, Class, Value, offsetof(FEXCore::Core::CPUState, gregs[gpr]));
|
||||
}
|
||||
else {
|
||||
LOGMAN_THROW_A_FMT(!(GPRSize == 4 && OpSize > 4), "Oops had a {} GPR load", OpSize);
|
||||
LOGMAN_THROW_AA_FMT(!(GPRSize == 4 && OpSize > 4), "Oops had a {} GPR load", OpSize);
|
||||
_StoreContext(std::min(GPRSize, OpSize), Class, Src, offsetof(FEXCore::Core::CPUState, gregs[gpr]) + (Operand.Data.GPR.HighBits ? 1 : 0));
|
||||
}
|
||||
}
|
||||
@@ -5029,7 +5049,7 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs) {
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Atomic IR Op: {}", IROp);
|
||||
LOGMAN_MSG_A_FMT("Unknown Atomic IR Op: {}", ToUnderlying(IROp));
|
||||
break;
|
||||
}
|
||||
}
|
||||
@@ -5069,37 +5089,58 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::INTOp(OpcodeArgs) {
|
||||
FEXCore::IR::BreakReason Reason{};
|
||||
uint8_t Literal{};
|
||||
bool setRIP = false;
|
||||
IR::BreakDefinition Reason;
|
||||
bool SetRIPToNext = false;
|
||||
|
||||
switch (Op->OP) {
|
||||
case 0xCD:
|
||||
Reason = FEXCore::IR::Break_Interrupt;
|
||||
Literal = Op->Src[0].Data.Literal.Value;
|
||||
case 0xCD: { // INT imm8
|
||||
uint8_t Literal = Op->Src[0].Data.Literal.Value;
|
||||
|
||||
if (Literal == 0x80) {
|
||||
// Syscall on linux
|
||||
SyscallOp(Op);
|
||||
return;
|
||||
}
|
||||
|
||||
Reason.ErrorRegister = Literal << 3 | (0b010);
|
||||
Reason.Signal = SIGSEGV;
|
||||
// GP is raised when task-gate isn't setup to be valid
|
||||
Reason.TrapNumber = X86State::X86_TRAPNO_GP;
|
||||
Reason.si_code = 0x80;
|
||||
break;
|
||||
}
|
||||
case 0xCE: // INTO
|
||||
Reason.ErrorRegister = 0;
|
||||
Reason.Signal = SIGSEGV;
|
||||
Reason.TrapNumber = X86State::X86_TRAPNO_OF;
|
||||
Reason.si_code = 0x80;
|
||||
break;
|
||||
case 0xF1: // INT1
|
||||
Reason.ErrorRegister = 0;
|
||||
Reason.Signal = SIGTRAP;
|
||||
Reason.TrapNumber = X86State::X86_TRAPNO_DB;
|
||||
Reason.si_code = 1;
|
||||
SetRIPToNext = true;
|
||||
break;
|
||||
case 0xCE:
|
||||
Reason = FEXCore::IR::Break_Overflow;
|
||||
break;
|
||||
case 0xF1:
|
||||
Reason = FEXCore::IR::Break_Interrupt;
|
||||
break;
|
||||
case 0xF4: {
|
||||
Reason = FEXCore::IR::Break_Halt;
|
||||
setRIP = true;
|
||||
case 0xF4: { // HLT
|
||||
Reason.ErrorRegister = 0;
|
||||
Reason.Signal = SIGSEGV;
|
||||
Reason.TrapNumber = X86State::X86_TRAPNO_GP;
|
||||
Reason.si_code = 0x80;
|
||||
break;
|
||||
}
|
||||
case 0x0B:
|
||||
Reason = FEXCore::IR::Break_Interrupt;
|
||||
case 0x0B: // UD2
|
||||
Reason.ErrorRegister = 0;
|
||||
Reason.Signal = SIGILL;
|
||||
Reason.TrapNumber = X86State::X86_TRAPNO_UD;
|
||||
Reason.si_code = 2;
|
||||
break;
|
||||
case 0xCC:
|
||||
Reason = FEXCore::IR::Break_Interrupt3;
|
||||
setRIP = true;
|
||||
case 0xCC: // INT3
|
||||
Reason.ErrorRegister = 0;
|
||||
Reason.Signal = SIGTRAP;
|
||||
Reason.TrapNumber = X86State::X86_TRAPNO_BP;
|
||||
Reason.si_code = 0x80;
|
||||
SetRIPToNext = true;
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -5108,13 +5149,17 @@ void OpDispatchBuilder::INTOp(OpcodeArgs) {
|
||||
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
|
||||
if (setRIP) {
|
||||
BlockSetRIP = setRIP;
|
||||
if (SetRIPToNext) {
|
||||
BlockSetRIP = SetRIPToNext;
|
||||
|
||||
// We want to set RIP to the next instruction after HLT/INT3
|
||||
// We want to set RIP to the next instruction after INT3/INT1
|
||||
auto NewRIP = GetRelocatedPC(Op);
|
||||
_StoreContext(GPRSize, GPRClass, NewRIP, offsetof(FEXCore::Core::CPUState, rip));
|
||||
}
|
||||
else if (Op->OP != 0xCE) {
|
||||
auto NewRIP = GetRelocatedPC(Op, -Op->InstSize);
|
||||
_StoreContext(GPRSize, GPRClass, NewRIP, offsetof(FEXCore::Core::CPUState, rip));
|
||||
}
|
||||
|
||||
if (Op->OP == 0xCE) { // Conditional to only break if Overflow == 1
|
||||
auto Flag = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC);
|
||||
@@ -5127,7 +5172,7 @@ void OpDispatchBuilder::INTOp(OpcodeArgs) {
|
||||
|
||||
auto NewRIP = GetRelocatedPC(Op);
|
||||
_StoreContext(GPRSize, GPRClass, NewRIP, offsetof(FEXCore::Core::CPUState, rip));
|
||||
_Break(Reason, Literal);
|
||||
_Break(Reason);
|
||||
|
||||
// Make sure to start a new block after ending this one
|
||||
auto JumpTarget = CreateNewCodeBlockAfter(FalseBlock);
|
||||
@@ -5135,7 +5180,8 @@ void OpDispatchBuilder::INTOp(OpcodeArgs) {
|
||||
SetCurrentCodeBlock(JumpTarget);
|
||||
}
|
||||
else {
|
||||
_Break(Reason, Literal);
|
||||
BlockSetRIP = true;
|
||||
_Break(Reason);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -5238,7 +5284,13 @@ void OpDispatchBuilder::UnimplementedOp(OpcodeArgs) {
|
||||
// We don't actually support this instruction
|
||||
// Multiblock may hit it though
|
||||
_StoreContext(GPRSize, GPRClass, GetRelocatedPC(Op, -Op->InstSize), offsetof(FEXCore::Core::CPUState, rip));
|
||||
_Break(FEXCore::IR::Break_Unimplemented, 0);
|
||||
_Break(FEXCore::IR::BreakDefinition {
|
||||
.ErrorRegister = 0,
|
||||
.Signal = SIGILL,
|
||||
.TrapNumber = 0,
|
||||
.si_code = 0,
|
||||
});
|
||||
|
||||
BlockSetRIP = true;
|
||||
|
||||
if (Multiblock) {
|
||||
@@ -5256,7 +5308,12 @@ void OpDispatchBuilder::InvalidOp(OpcodeArgs) {
|
||||
// We don't actually support this instruction
|
||||
// Multiblock may hit it though
|
||||
_StoreContext(GPRSize, GPRClass, GetRelocatedPC(Op, -Op->InstSize), offsetof(FEXCore::Core::CPUState, rip));
|
||||
_Break(FEXCore::IR::Break_InvalidInstruction, 0);
|
||||
_Break(FEXCore::IR::BreakDefinition {
|
||||
.ErrorRegister = 0,
|
||||
.Signal = SIGILL,
|
||||
.TrapNumber = 0,
|
||||
.si_code = 0,
|
||||
});
|
||||
BlockSetRIP = true;
|
||||
}
|
||||
|
||||
@@ -6566,6 +6623,7 @@ constexpr uint16_t PF_F2 = 3;
|
||||
{OPD(0, PF_3A_66, 0x40), 1, &OpDispatchBuilder::DPPOp<4>},
|
||||
{OPD(0, PF_3A_66, 0x41), 1, &OpDispatchBuilder::DPPOp<8>},
|
||||
{OPD(0, PF_3A_66, 0x42), 1, &OpDispatchBuilder::MPSADBWOp},
|
||||
{OPD(0, PF_3A_66, 0x44), 1, &OpDispatchBuilder::PCLMULQDQOp},
|
||||
|
||||
{OPD(0, PF_3A_NONE, 0xCC), 1, &OpDispatchBuilder::SHA1RNDS4Op},
|
||||
};
|
||||
@@ -6649,6 +6707,8 @@ constexpr uint16_t PF_F2 = 3;
|
||||
{OPD(2, 0b10, 0xF7), 1, &OpDispatchBuilder::BMI2Shift},
|
||||
{OPD(2, 0b11, 0xF7), 1, &OpDispatchBuilder::BMI2Shift},
|
||||
|
||||
{OPD(3, 0b01, 0x44), 1, &OpDispatchBuilder::VPCLMULQDQOp},
|
||||
|
||||
{OPD(3, 0b11, 0xF0), 1, &OpDispatchBuilder::RORX},
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
@@ -76,8 +76,10 @@ public:
|
||||
OrderedNode* flagsOpSrcSigned{};
|
||||
|
||||
FEXCore::Context::Context *CTX{};
|
||||
|
||||
// Used during new op bringup
|
||||
bool ShouldDump {false};
|
||||
|
||||
|
||||
struct JumpTargetInfo {
|
||||
OrderedNode* BlockEntry;
|
||||
bool HaveEmitted;
|
||||
@@ -621,6 +623,8 @@ public:
|
||||
void DPPOp(OpcodeArgs);
|
||||
|
||||
void MPSADBWOp(OpcodeArgs);
|
||||
void PCLMULQDQOp(OpcodeArgs);
|
||||
void VPCLMULQDQOp(OpcodeArgs);
|
||||
|
||||
void CRC32(OpcodeArgs);
|
||||
|
||||
@@ -661,7 +665,7 @@ private:
|
||||
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, OrderedNode *const Src, int8_t Align, MemoryAccessType AccessType = MemoryAccessType::ACCESS_DEFAULT);
|
||||
|
||||
[[nodiscard]] static uint32_t GPROffset(X86State::X86Reg reg) {
|
||||
LOGMAN_THROW_A_FMT(reg <= X86State::X86Reg::REG_R15, "Invalid reg used");
|
||||
LOGMAN_THROW_AA_FMT(reg <= X86State::X86Reg::REG_R15, "Invalid reg used");
|
||||
return static_cast<uint32_t>(offsetof(Core::CPUState, gregs[static_cast<size_t>(reg)]));
|
||||
}
|
||||
|
||||
@@ -1236,14 +1240,14 @@ private:
|
||||
uint64_t Entry;
|
||||
|
||||
OrderedNode* _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *Addr, OrderedNode *Value, uint8_t Align = 1) {
|
||||
if (CTX->Config.TSOEnabled)
|
||||
if (CTX->IsTSOEnabled())
|
||||
return _StoreMemTSO(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
else
|
||||
return _StoreMem(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
OrderedNode* _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *ssa0, uint8_t Align = 1) {
|
||||
if (CTX->Config.TSOEnabled)
|
||||
if (CTX->IsTSOEnabled())
|
||||
return _LoadMemTSO(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
else
|
||||
return _LoadMem(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
|
||||
@@ -221,7 +221,7 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *XMM0 = _LoadContext(16, FPRClass, offsetof(FEXCore::Core::CPUState, xmm[0]));
|
||||
OrderedNode *XMM0 = _LoadContext(16, FPRClass, offsetof(FEXCore::Core::CPUState, xmm.avx.data[0]));
|
||||
|
||||
auto A0 = _VExtractToGPR(16, 4, Src, 3);
|
||||
auto B0 = _VExtractToGPR(16, 4, Src, 2);
|
||||
@@ -306,4 +306,26 @@ void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Res, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Selector needs to be literal here");
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[1].Data.Literal.Value);
|
||||
|
||||
auto Res = _PCLMUL(Dest, Src, Selector);
|
||||
StoreResult(FPRClass, Op, Res, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[2].IsLiteral(), "Selector needs to be literal here");
|
||||
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[2].Data.Literal.Value);
|
||||
|
||||
auto Res = _PCLMUL(Src1, Src2, Selector);
|
||||
StoreResult(FPRClass, Op, Res, -1);
|
||||
}
|
||||
|
||||
}
|
||||
@@ -440,10 +440,16 @@ void OpDispatchBuilder::MOVQOp(OpcodeArgs) {
|
||||
// This instruction is a bit special that if the destination is a register then it'll ZEXT the 64bit source to 128bit
|
||||
if (Op->Dest.IsGPR()) {
|
||||
const auto gpr = Op->Dest.Data.GPR.GPR;
|
||||
const auto gprIndex = gpr - X86State::REG_XMM_0;
|
||||
|
||||
_StoreContext(8, FPRClass, Src, offsetof(FEXCore::Core::CPUState, xmm[gpr - FEXCore::X86State::REG_XMM_0][0]));
|
||||
const auto fprLowOffset = CTX->HostFeatures.SupportsAVX ? offsetof(Core::CPUState, xmm.avx.data[gprIndex][0])
|
||||
: offsetof(Core::CPUState, xmm.sse.data[gprIndex][0]);
|
||||
const auto fprHighOffset = CTX->HostFeatures.SupportsAVX ? offsetof(Core::CPUState, xmm.avx.data[gprIndex][1])
|
||||
: offsetof(Core::CPUState, xmm.sse.data[gprIndex][1]);
|
||||
|
||||
_StoreContext(8, FPRClass, Src, fprLowOffset);
|
||||
auto Const = _Constant(0);
|
||||
_StoreContext(8, GPRClass, Const, offsetof(FEXCore::Core::CPUState, xmm[gpr - FEXCore::X86State::REG_XMM_0][1]));
|
||||
_StoreContext(8, GPRClass, Const, fprHighOffset);
|
||||
}
|
||||
else {
|
||||
// This is simple, just store the result
|
||||
@@ -562,7 +568,7 @@ void OpDispatchBuilder::PSHUFBOp(OpcodeArgs) {
|
||||
|
||||
template<size_t ElementSize, bool HalfSize, bool Low>
|
||||
void OpDispatchBuilder::PSHUFDOp(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(ElementSize != 0, "What. No element size?");
|
||||
LOGMAN_THROW_AA_FMT(ElementSize != 0, "What. No element size?");
|
||||
const auto Size = GetSrcSize(Op);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
uint8_t Shuffle = Op->Src[1].Data.Literal.Value;
|
||||
@@ -599,7 +605,7 @@ void OpDispatchBuilder::PSHUFDOp<4, false, true>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::SHUFOp(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(ElementSize != 0, "What. No element size?");
|
||||
LOGMAN_THROW_AA_FMT(ElementSize != 0, "What. No element size?");
|
||||
const auto Size = GetSrcSize(Op);
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
@@ -1390,10 +1396,18 @@ void OpDispatchBuilder::FXSaveOp(OpcodeArgs) {
|
||||
|
||||
_StoreMem(FPRClass, 16, MemLocation, MMReg, 16);
|
||||
}
|
||||
unsigned NumRegs = CTX->Config.Is64BitMode ? 16 : 8;
|
||||
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
const auto GetXMMOffset = [this](size_t i) {
|
||||
if (CTX->HostFeatures.SupportsAVX) {
|
||||
return offsetof(Core::CPUState, xmm.avx.data[i]);
|
||||
} else {
|
||||
return offsetof(Core::CPUState, xmm.sse.data[i]);
|
||||
}
|
||||
};
|
||||
|
||||
for (unsigned i = 0; i < NumRegs; ++i) {
|
||||
OrderedNode *XMMReg = _LoadContext(16, FPRClass, offsetof(FEXCore::Core::CPUState, xmm[i]));
|
||||
OrderedNode *XMMReg = _LoadContext(16, FPRClass, GetXMMOffset(i));
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 160));
|
||||
|
||||
_StoreMem(FPRClass, 16, MemLocation, XMMReg, 16);
|
||||
@@ -1439,12 +1453,20 @@ void OpDispatchBuilder::FXRStoreOp(OpcodeArgs) {
|
||||
auto MMReg = _LoadMem(FPRClass, 16, MemLocation, 16);
|
||||
_StoreContext(16, FPRClass, MMReg, offsetof(FEXCore::Core::CPUState, mm[i]));
|
||||
}
|
||||
unsigned NumRegs = CTX->Config.Is64BitMode ? 16 : 8;
|
||||
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
const auto GetXMMOffset = [this](size_t i) {
|
||||
if (CTX->HostFeatures.SupportsAVX) {
|
||||
return offsetof(Core::CPUState, xmm.avx.data[i]);
|
||||
} else {
|
||||
return offsetof(Core::CPUState, xmm.sse.data[i]);
|
||||
}
|
||||
};
|
||||
|
||||
for (unsigned i = 0; i < NumRegs; ++i) {
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 160));
|
||||
auto XMMReg = _LoadMem(FPRClass, 16, MemLocation, 16);
|
||||
_StoreContext(16, FPRClass, XMMReg, offsetof(FEXCore::Core::CPUState, xmm[i]));
|
||||
_StoreContext(16, FPRClass, XMMReg, GetXMMOffset(i));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1590,8 +1612,12 @@ void OpDispatchBuilder::MOVQ2DQ(OpcodeArgs) {
|
||||
|
||||
// This instruction is a bit special in that if the source is MMX then it zexts to 128bit
|
||||
if constexpr (ToXMM) {
|
||||
const auto Index = Op->Dest.Data.GPR.GPR - FEXCore::X86State::REG_XMM_0;
|
||||
const auto Offset = CTX->HostFeatures.SupportsAVX ? offsetof(FEXCore::Core::CPUState, xmm.avx.data[Index][0])
|
||||
: offsetof(FEXCore::Core::CPUState, xmm.sse.data[Index][0]);
|
||||
|
||||
Src = _VMov(16, Src);
|
||||
_StoreContext(16, FPRClass, Src, offsetof(FEXCore::Core::CPUState, xmm[Op->Dest.Data.GPR.GPR - FEXCore::X86State::REG_XMM_0][0]));
|
||||
_StoreContext(16, FPRClass, Src, Offset);
|
||||
}
|
||||
else {
|
||||
// This is simple, just store the result
|
||||
@@ -2356,7 +2382,7 @@ void OpDispatchBuilder::VectorVariableBlend(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
// The mask is hardcoded to be xmm0 in this instruction
|
||||
OrderedNode *Mask = _LoadContext(16, FPRClass, offsetof(FEXCore::Core::CPUState, xmm[0]));
|
||||
OrderedNode *Mask = _LoadContext(16, FPRClass, offsetof(FEXCore::Core::CPUState, xmm.avx.data[0]));
|
||||
// Each element is selected by the high bit of that element size
|
||||
// Dest[ElementIdx] = Xmm0[ElementIndex][HighBit] ? Src : Dest;
|
||||
//
|
||||
|
||||
@@ -42,7 +42,7 @@ void InitializeH0F3ATables(Context::OperatingMode Mode) {
|
||||
{OPD(0, PF_3A_66, 0x40), 1, X86InstInfo{"DPPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x41), 1, X86InstInfo{"DPPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x42), 1, X86InstInfo{"MPSADBW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x44), 1, X86InstInfo{"PCLMULQDQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x44), 1, X86InstInfo{"PCLMULQDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x60), 1, X86InstInfo{"PCMPESTRM", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x61), 1, X86InstInfo{"PCMPESTRI", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
@@ -441,7 +441,7 @@ void InitializeVEXTables() {
|
||||
{OPD(3, 0b01, 0x40), 1, X86InstInfo{"VDPPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x41), 1, X86InstInfo{"VDPPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x42), 1, X86InstInfo{"VMPSADBW", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x44), 1, X86InstInfo{"VPCLMULQDQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x44), 1, X86InstInfo{"VPCLMULQDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x46), 1, X86InstInfo{"VPERM2I128", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(3, 0b01, 0x48), 1, X86InstInfo{"VPERMILzz2PS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
@@ -33,7 +33,7 @@ static inline void GenerateTable(X86InstInfo *FinalTable, X86TablesInfoStruct<Op
|
||||
auto OpNum = Op.first;
|
||||
X86InstInfo const &Info = Op.Info;
|
||||
for (uint32_t i = 0; i < Op.second; ++i) {
|
||||
LOGMAN_THROW_A_FMT(FinalTable[OpNum + i].Type == TYPE_UNKNOWN, "Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
|
||||
LOGMAN_THROW_AA_FMT(FinalTable[OpNum + i].Type == TYPE_UNKNOWN, "Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
|
||||
FinalTable[OpNum + i] = Info;
|
||||
#ifndef NDEBUG
|
||||
++Total;
|
||||
@@ -51,7 +51,7 @@ static inline void GenerateTableWithCopy(X86InstInfo *FinalTable, X86TablesInfoS
|
||||
auto OpNum = Op.first;
|
||||
X86InstInfo const &Info = Op.Info;
|
||||
for (uint32_t i = 0; i < Op.second; ++i) {
|
||||
LOGMAN_THROW_A_FMT(FinalTable[OpNum + i].Type == TYPE_UNKNOWN, "Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
|
||||
LOGMAN_THROW_AA_FMT(FinalTable[OpNum + i].Type == TYPE_UNKNOWN, "Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
|
||||
if (Info.Type == TYPE_COPY_OTHER) {
|
||||
FinalTable[OpNum + i] = OtherLocal[OpNum + i];
|
||||
}
|
||||
@@ -74,7 +74,7 @@ static inline void GenerateX87Table(X86InstInfo *FinalTable, X86TablesInfoStruct
|
||||
auto OpNum = Op.first;
|
||||
X86InstInfo const &Info = Op.Info;
|
||||
for (uint32_t i = 0; i < Op.second; ++i) {
|
||||
LOGMAN_THROW_A_FMT(FinalTable[OpNum + i].Type == TYPE_UNKNOWN, "Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
|
||||
LOGMAN_THROW_AA_FMT(FinalTable[OpNum + i].Type == TYPE_UNKNOWN, "Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
|
||||
if ((OpNum & 0b11'000'000) == 0b11'000'000) {
|
||||
// If the mod field is 0b11 then it is a regular op
|
||||
FinalTable[OpNum + i] = Info;
|
||||
@@ -82,7 +82,7 @@ static inline void GenerateX87Table(X86InstInfo *FinalTable, X86TablesInfoStruct
|
||||
else {
|
||||
// If the mod field is !0b11 then this instruction is duplicated through the whole mod [0b00, 0b10] range
|
||||
// and the modrm.rm space because that is used part of the instruction encoding
|
||||
LOGMAN_THROW_A_FMT((OpNum & 0b11'000'000) == 0, "Only support mod field of zero in this path");
|
||||
LOGMAN_THROW_AA_FMT((OpNum & 0b11'000'000) == 0, "Only support mod field of zero in this path");
|
||||
for (uint16_t mod = 0b00'000'000; mod < 0b11'000'000; mod += 0b01'000'000) {
|
||||
for (uint16_t rm = 0b000; rm < 0b1'000; ++rm) {
|
||||
FinalTable[(OpNum | mod | rm) + i] = Info;
|
||||
|
||||
+127
@@ -0,0 +1,127 @@
|
||||
#include "GDBJIT.h"
|
||||
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/HLE/SourcecodeResolver.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#if defined(GDB_SYMBOLS_ENABLED)
|
||||
|
||||
#include <FEXCore/Debug/GDBReaderInterface.h>
|
||||
|
||||
extern "C" {
|
||||
enum jit_actions_t { JIT_NOACTION = 0, JIT_REGISTER_FN, JIT_UNREGISTER_FN };
|
||||
|
||||
struct jit_code_entry {
|
||||
jit_code_entry *next_entry;
|
||||
jit_code_entry *prev_entry;
|
||||
const char *symfile_addr;
|
||||
uint64_t symfile_size;
|
||||
};
|
||||
|
||||
struct jit_descriptor {
|
||||
uint32_t version;
|
||||
/* This type should be jit_actions_t, but we use uint32_t
|
||||
to be explicit about the bitwidth. */
|
||||
uint32_t action_flag;
|
||||
jit_code_entry *relevant_entry;
|
||||
jit_code_entry *first_entry;
|
||||
};
|
||||
|
||||
/* Make sure to specify the version statically, because the
|
||||
debugger may check the version before we can set it. */
|
||||
|
||||
constinit jit_descriptor __jit_debug_descriptor = {.version = 1};
|
||||
|
||||
/* GDB puts a breakpoint in this function. */
|
||||
void __attribute__((noinline)) __jit_debug_register_code() {
|
||||
asm volatile("" ::"r"(&__jit_debug_descriptor));
|
||||
};
|
||||
}
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
void GDBJITRegister(FEXCore::IR::AOTIRCacheEntry *Entry, uintptr_t VAFileStart,
|
||||
uint64_t GuestRIP, uintptr_t HostEntry,
|
||||
FEXCore::Core::DebugData *DebugData) {
|
||||
auto map = Entry->SourcecodeMap.get();
|
||||
|
||||
if (map) {
|
||||
auto FileOffset = GuestRIP - VAFileStart;
|
||||
|
||||
auto Sym = map->FindSymbolMapping(FileOffset);
|
||||
|
||||
std::string SymName = HLE::SourcecodeSymbolMapping::SymName(
|
||||
Sym, Entry->Filename, HostEntry, FileOffset);
|
||||
|
||||
std::vector<gdb_line_mapping> Lines;
|
||||
for (const auto &GuestOpcode : DebugData->GuestOpcodes) {
|
||||
auto Line = map->FindLineMapping(GuestRIP + GuestOpcode.GuestEntryOffset -
|
||||
VAFileStart);
|
||||
if (Line) {
|
||||
Lines.push_back(
|
||||
{Line->LineNumber, HostEntry + GuestOpcode.HostEntryOffset});
|
||||
}
|
||||
}
|
||||
|
||||
size_t size = sizeof(info_t) + 1 * sizeof(blocks_t) +
|
||||
Lines.size() * sizeof(gdb_line_mapping);
|
||||
|
||||
auto mem = (uint8_t *)malloc(size);
|
||||
auto base = mem;
|
||||
info_t *info = (info_t *)mem;
|
||||
mem += sizeof(info_t);
|
||||
|
||||
strncpy(info->filename, map->SourceFile.c_str(), 511);
|
||||
|
||||
info->nblocks = 1;
|
||||
|
||||
auto blocks = (blocks_t *)mem;
|
||||
info->blocks_ofs = mem - base;
|
||||
|
||||
mem += info->nblocks * sizeof(blocks_t);
|
||||
|
||||
for (int i = 0; i < info->nblocks; i++) {
|
||||
strncpy(blocks[i].name, SymName.c_str(), 511);
|
||||
blocks[i].start = HostEntry;
|
||||
blocks[i].end = HostEntry + DebugData->HostCodeSize;
|
||||
}
|
||||
|
||||
info->nlines = Lines.size();
|
||||
|
||||
auto lines = (gdb_line_mapping *)mem;
|
||||
info->lines_ofs = mem - base;
|
||||
mem += info->nlines * sizeof(gdb_line_mapping);
|
||||
|
||||
if (info->nlines) {
|
||||
memcpy(lines, &Lines.at(0), info->nlines * sizeof(gdb_line_mapping));
|
||||
}
|
||||
|
||||
auto entry = new jit_code_entry{0, 0, 0, 0};
|
||||
|
||||
entry->symfile_addr = (const char *)info;
|
||||
entry->symfile_size = size;
|
||||
|
||||
if (__jit_debug_descriptor.first_entry) {
|
||||
__jit_debug_descriptor.relevant_entry->next_entry = entry;
|
||||
entry->prev_entry = __jit_debug_descriptor.relevant_entry;
|
||||
} else {
|
||||
__jit_debug_descriptor.first_entry = entry;
|
||||
}
|
||||
|
||||
__jit_debug_descriptor.relevant_entry = entry;
|
||||
__jit_debug_descriptor.action_flag = JIT_REGISTER_FN;
|
||||
__jit_debug_register_code();
|
||||
}
|
||||
}
|
||||
} // namespace FEXCore
|
||||
#else
|
||||
namespace FEXCore {
|
||||
void GDBJITRegister([[maybe_unused]] FEXCore::IR::AOTIRCacheEntry *Entry,
|
||||
[[maybe_unused]] uintptr_t VAFileStart,
|
||||
[[maybe_unused]] uint64_t GuestRIP,
|
||||
[[maybe_unused]] uintptr_t HostEntry,
|
||||
[[maybe_unused]] FEXCore::Core::DebugData *DebugData) {
|
||||
ERROR_AND_DIE_FMT("GDBSymbols support not compiled in");
|
||||
}
|
||||
} // namespace FEXCore
|
||||
#endif
|
||||
@@ -0,0 +1,7 @@
|
||||
|
||||
|
||||
#include <Interface/IR/AOTIR.h>
|
||||
|
||||
namespace FEXCore {
|
||||
void GDBJITRegister(FEXCore::IR::AOTIRCacheEntry *Entry, uintptr_t VAFileStart, uint64_t GuestRIP, uintptr_t HostEntry, FEXCore::Core::DebugData *DebugData);
|
||||
}
|
||||
+313
-16
@@ -10,42 +10,146 @@ $end_info$
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include "FEXCore/Utils/CompilerDefs.h"
|
||||
#include "Thunks.h"
|
||||
|
||||
#include <cstdint>
|
||||
#include <dlfcn.h>
|
||||
|
||||
#include <Interface/Context/Context.h>
|
||||
#include "FEXCore/Core/X86Enums.h"
|
||||
#include <malloc.h>
|
||||
#include <map>
|
||||
#include <mutex>
|
||||
#include <unordered_map>
|
||||
#include <memory>
|
||||
#include <shared_mutex>
|
||||
#include <stdint.h>
|
||||
#include <string>
|
||||
#include <utility>
|
||||
|
||||
#include "jemalloc/jemalloc.h"
|
||||
|
||||
struct LoadlibArgs {
|
||||
const char *Name;
|
||||
uintptr_t CallbackThunks;
|
||||
};
|
||||
|
||||
static thread_local FEXCore::Core::InternalThreadState *Thread;
|
||||
|
||||
static __attribute__((aligned(16), naked, section("HostToGuestTrampolineTemplate"))) void HostToGuestTrampolineTemplate() {
|
||||
#if defined(_M_X86_64)
|
||||
asm(
|
||||
"lea 0f(%rip), %r11 \n"
|
||||
"jmpq *0f(%rip) \n"
|
||||
".align 8 \n"
|
||||
"0: \n"
|
||||
".quad 0, 0, 0, 0 \n" // TrampolineInstanceInfo
|
||||
);
|
||||
#elif defined(_M_ARM_64)
|
||||
asm(
|
||||
"adr x11, 0f \n"
|
||||
"ldr x16, [x11] \n"
|
||||
"br x16 \n"
|
||||
// Manually align to the next 8-byte boundary
|
||||
// NOTE: GCC over-aligns to a full page when using .align directives on ARM (last tested on GCC 11.2)
|
||||
"nop \n"
|
||||
"0: \n"
|
||||
".quad 0, 0, 0, 0 \n" // TrampolineInstanceInfo
|
||||
);
|
||||
#else
|
||||
#error Unsupported host architecture
|
||||
#endif
|
||||
}
|
||||
|
||||
extern char __start_HostToGuestTrampolineTemplate[];
|
||||
extern char __stop_HostToGuestTrampolineTemplate[];
|
||||
|
||||
namespace FEXCore {
|
||||
struct ExportEntry { uint8_t *sha256; ThunkedFunction* Fn; };
|
||||
|
||||
class ThunkHandler_impl final: public ThunkHandler {
|
||||
struct TrampolineInstanceInfo {
|
||||
void* HostPacker;
|
||||
uintptr_t CallCallback;
|
||||
uintptr_t GuestUnpacker;
|
||||
uintptr_t GuestTarget;
|
||||
};
|
||||
|
||||
// Opaque type pointing to an instance of HostToGuestTrampolineTemplate and its
|
||||
// embedded TrampolineInstanceInfo
|
||||
struct HostToGuestTrampolinePtr;
|
||||
const auto HostToGuestTrampolineSize = __stop_HostToGuestTrampolineTemplate - __start_HostToGuestTrampolineTemplate;
|
||||
|
||||
static TrampolineInstanceInfo& GetInstanceInfo(HostToGuestTrampolinePtr* Trampoline) {
|
||||
const auto Length = __stop_HostToGuestTrampolineTemplate - __start_HostToGuestTrampolineTemplate;
|
||||
const auto InstanceInfoOffset = Length - sizeof(TrampolineInstanceInfo);
|
||||
return *reinterpret_cast<TrampolineInstanceInfo*>(reinterpret_cast<char*>(Trampoline) + InstanceInfoOffset);
|
||||
}
|
||||
|
||||
struct GuestcallInfo {
|
||||
uintptr_t GuestUnpacker;
|
||||
uintptr_t GuestTarget;
|
||||
|
||||
bool operator==(const GuestcallInfo&) const noexcept = default;
|
||||
};
|
||||
|
||||
struct GuestcallInfoHash {
|
||||
size_t operator()(const GuestcallInfo& x) const noexcept {
|
||||
// Hash only the target address, which is generally unique.
|
||||
// For the unlikely case of a hash collision, std::unordered_map still picks the correct bucket entry.
|
||||
return std::hash<uintptr_t>{}(x.GuestTarget);
|
||||
}
|
||||
};
|
||||
|
||||
// Bits in a SHA256 sum are already randomly distributed, so truncation yields a suitable hash function
|
||||
struct TruncatingSHA256Hash {
|
||||
size_t operator()(const FEXCore::IR::SHA256Sum& SHA256Sum) const noexcept {
|
||||
return (const size_t&)SHA256Sum;
|
||||
}
|
||||
};
|
||||
|
||||
HostToGuestTrampolinePtr* MakeHostTrampolineForGuestFunction(void* HostPacker, uintptr_t GuestTarget, uintptr_t GuestUnpacker);
|
||||
|
||||
struct ThunkHandler_impl final: public ThunkHandler {
|
||||
std::shared_mutex ThunksMutex;
|
||||
|
||||
std::map<IR::SHA256Sum, ThunkedFunction*> Thunks = {
|
||||
std::unordered_map<IR::SHA256Sum, ThunkedFunction*, TruncatingSHA256Hash> Thunks = {
|
||||
{
|
||||
// sha256(fex:loadlib)
|
||||
{ 0x27, 0x7e, 0xb7, 0x69, 0x5b, 0xe9, 0xab, 0x12, 0x6e, 0xf7, 0x85, 0x9d, 0x4b, 0xc9, 0xa2, 0x44, 0x46, 0xcf, 0xbd, 0xb5, 0x87, 0x43, 0xef, 0x28, 0xa2, 0x65, 0xba, 0xfc, 0x89, 0x0f, 0x77, 0x80},
|
||||
{ 0x27, 0x7e, 0xb7, 0x69, 0x5b, 0xe9, 0xab, 0x12, 0x6e, 0xf7, 0x85, 0x9d, 0x4b, 0xc9, 0xa2, 0x44, 0x46, 0xcf, 0xbd, 0xb5, 0x87, 0x43, 0xef, 0x28, 0xa2, 0x65, 0xba, 0xfc, 0x89, 0x0f, 0x77, 0x80 },
|
||||
&LoadLib
|
||||
},
|
||||
{
|
||||
// sha256(fex:is_lib_loaded)
|
||||
{ 0xee, 0x57, 0xba, 0x0c, 0x5f, 0x6e, 0xef, 0x2a, 0x8c, 0xb5, 0x19, 0x81, 0xc9, 0x23, 0xe6, 0x51, 0xae, 0x65, 0x02, 0x8f, 0x2b, 0x5d, 0x59, 0x90, 0x6a, 0x7e, 0xe2, 0xe7, 0x1c, 0x33, 0x8a, 0xff },
|
||||
&IsLibLoaded
|
||||
},
|
||||
{
|
||||
// sha256(fex:is_host_heap_allocation)
|
||||
{ 0xf5, 0x77, 0x68, 0x43, 0xbb, 0x6b, 0x28, 0x18, 0x40, 0xb0, 0xdb, 0x8a, 0x66, 0xfb, 0x0e, 0x2d, 0x98, 0xc2, 0xad, 0xe2, 0x5a, 0x18, 0x5a, 0x37, 0x2e, 0x13, 0xc9, 0xe7, 0xb9, 0x8c, 0xa9, 0x3e },
|
||||
&IsHostHeapAllocation
|
||||
},
|
||||
{
|
||||
// sha256(fex:link_address_to_function)
|
||||
{ 0xe6, 0xa8, 0xec, 0x1c, 0x7b, 0x74, 0x35, 0x27, 0xe9, 0x4f, 0x5b, 0x6e, 0x2d, 0xc9, 0xa0, 0x27, 0xd6, 0x1f, 0x2b, 0x87, 0x8f, 0x2d, 0x35, 0x50, 0xea, 0x16, 0xb8, 0xc4, 0x5e, 0x42, 0xfd, 0x77 },
|
||||
&LinkAddressToGuestFunction
|
||||
},
|
||||
{
|
||||
// sha256(fex:allocate_host_trampoline_for_guest_function)
|
||||
{ 0x9b, 0xb2, 0xf4, 0xb4, 0x83, 0x7d, 0x28, 0x93, 0x40, 0xcb, 0xf4, 0x7a, 0x0b, 0x47, 0x85, 0x87, 0xf9, 0xbc, 0xb5, 0x27, 0xca, 0xa6, 0x93, 0xa5, 0xc0, 0x73, 0x27, 0x24, 0xae, 0xc8, 0xb8, 0x5a },
|
||||
&AllocateHostTrampolineForGuestFunction
|
||||
}
|
||||
};
|
||||
|
||||
// Can't be a string_view. We need to keep a copy of the library name in-case string_view pointer goes away.
|
||||
// Ideally we track when a library has been unloaded and remove it from this set before the memory backing goes away.
|
||||
std::set<std::string> Libs;
|
||||
|
||||
std::unordered_map<GuestcallInfo, HostToGuestTrampolinePtr*, GuestcallInfoHash> GuestcallToHostTrampoline;
|
||||
|
||||
uint8_t *HostTrampolineInstanceDataPtr;
|
||||
size_t HostTrampolineInstanceDataAvailable = 0;
|
||||
|
||||
|
||||
/*
|
||||
Set arg0/1 to arg regs, use CTX::HandleCallback to handle the callback
|
||||
*/
|
||||
@@ -56,13 +160,101 @@ namespace FEXCore {
|
||||
Thread->CTX->HandleCallback(Thread, (uintptr_t)callback);
|
||||
}
|
||||
|
||||
/**
|
||||
* Instructs the Core to redirect calls to functions at the given
|
||||
* address to another function. The original callee address is passed
|
||||
* to the target function through an implicit argument stored in r11.
|
||||
*
|
||||
* The primary use case of this is ensuring that host function pointers
|
||||
* returned from thunked APIs can safely be called by the guest.
|
||||
*/
|
||||
static void LinkAddressToGuestFunction(void* argsv) {
|
||||
struct args_t {
|
||||
uintptr_t original_callee;
|
||||
uintptr_t target_addr; // Guest function to call when branching to original_callee
|
||||
};
|
||||
|
||||
auto args = reinterpret_cast<args_t*>(argsv);
|
||||
auto CTX = Thread->CTX;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(args->original_callee, "Tried to link null pointer address to guest function");
|
||||
LOGMAN_THROW_AA_FMT(args->target_addr, "Tried to link address to null pointer guest function");
|
||||
if (!CTX->Config.Is64BitMode) {
|
||||
LOGMAN_THROW_AA_FMT((args->original_callee >> 32) == 0, "Tried to link 64-bit address in 32-bit mode");
|
||||
LOGMAN_THROW_AA_FMT((args->target_addr >> 32) == 0, "Tried to link 64-bit address in 32-bit mode");
|
||||
}
|
||||
|
||||
LogMan::Msg::DFmt("Thunks: Adding guest trampoline from address {:#x} to guest function {:#x}",
|
||||
args->original_callee, args->target_addr);
|
||||
|
||||
auto Result = Thread->CTX->AddCustomIREntrypoint(
|
||||
args->original_callee,
|
||||
[CTX, GuestThunkEntrypoint = args->target_addr](uintptr_t Entrypoint, FEXCore::IR::IREmitter *emit) {
|
||||
auto IRHeader = emit->_IRHeader(emit->Invalid(), 0);
|
||||
auto Block = emit->CreateCodeNode();
|
||||
IRHeader.first->Blocks = emit->WrapNode(Block);
|
||||
emit->SetCurrentCodeBlock(Block);
|
||||
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
|
||||
emit->_StoreContext(GPRSize, IR::GPRClass, emit->_Constant(Entrypoint), offsetof(Core::CPUState, gregs[X86State::REG_R11]));
|
||||
emit->_ExitFunction(emit->_Constant(GuestThunkEntrypoint));
|
||||
}, CTX->ThunkHandler.get(), (void*)args->target_addr);
|
||||
|
||||
if (!Result) {
|
||||
if (Result.Creator != CTX->ThunkHandler.get()) {
|
||||
ERROR_AND_DIE_FMT("Input address for LinkAddressToGuestFunction is already linked by another module");
|
||||
}
|
||||
if (Result.Data != (void*)args->target_addr) {
|
||||
// NOTE: This may happen in Vulkan thunks if the Vulkan driver resolves two different symbols
|
||||
// to the same function (e.g. vkGetPhysicalDeviceFeatures2/vkGetPhysicalDeviceFeatures2KHR)
|
||||
LogMan::Msg::EFmt("Input address for LinkAddressToGuestFunction is already linked elsewhere");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Guest-side helper to initiate creation of a host trampoline for
|
||||
* calling guest functions. This must be followed by a host-side call
|
||||
* to FinalizeHostTrampolineForGuestFunction to make the trampoline
|
||||
* usable.
|
||||
*
|
||||
* This two-step initialization is equivalent to a host-side call to
|
||||
* MakeHostTrampolineForGuestFunction. The split is needed if the
|
||||
* host doesn't have all information needed to create the trampoline
|
||||
* on its own.
|
||||
*/
|
||||
static void AllocateHostTrampolineForGuestFunction(void* ArgsRV) {
|
||||
struct ArgsRV_t {
|
||||
uintptr_t GuestUnpacker;
|
||||
uintptr_t GuestTarget;
|
||||
uintptr_t rv; // Pointer to host trampoline + TrampolineInstanceInfo
|
||||
} *args = reinterpret_cast<ArgsRV_t*>(ArgsRV);
|
||||
|
||||
args->rv = (uintptr_t)MakeHostTrampolineForGuestFunction(nullptr, args->GuestTarget, args->GuestUnpacker);
|
||||
}
|
||||
|
||||
/**
|
||||
* Checks if the given pointer is allocated on the host heap.
|
||||
*
|
||||
* This is useful for thunking APIs that need to work with both guest
|
||||
* and host heap pointers.
|
||||
*/
|
||||
static void IsHostHeapAllocation(void* ArgsRV) {
|
||||
struct ArgsRV_t {
|
||||
void* ptr;
|
||||
bool rv;
|
||||
} *args = reinterpret_cast<ArgsRV_t*>(ArgsRV);
|
||||
|
||||
args->rv = je_is_known_allocation(args->ptr);
|
||||
}
|
||||
|
||||
static void LoadLib(void *ArgsV) {
|
||||
auto CTX = Thread->CTX;
|
||||
|
||||
auto Args = reinterpret_cast<LoadlibArgs*>(ArgsV);
|
||||
|
||||
auto Name = Args->Name;
|
||||
auto CallbackThunks = Args->CallbackThunks;
|
||||
|
||||
auto SOName = CTX->Config.ThunkHostLibsPath() + "/" + (const char*)Name + "-host.so";
|
||||
|
||||
@@ -75,13 +267,13 @@ namespace FEXCore {
|
||||
|
||||
const auto InitSym = std::string("fexthunks_exports_") + Name;
|
||||
|
||||
ExportEntry* (*InitFN)(void *, uintptr_t);
|
||||
ExportEntry* (*InitFN)();
|
||||
(void*&)InitFN = dlsym(Handle, InitSym.c_str());
|
||||
if (!InitFN) {
|
||||
ERROR_AND_DIE_FMT("LoadLib: Failed to find export {}", InitSym);
|
||||
}
|
||||
|
||||
auto Exports = InitFN((void*)&CallCallback, CallbackThunks);
|
||||
auto Exports = InitFN();
|
||||
if (!Exports) {
|
||||
ERROR_AND_DIE_FMT("LoadLib: Failed to initialize thunk library {}. "
|
||||
"Check if the corresponding host library is installed "
|
||||
@@ -91,7 +283,9 @@ namespace FEXCore {
|
||||
auto That = reinterpret_cast<ThunkHandler_impl*>(CTX->ThunkHandler.get());
|
||||
|
||||
{
|
||||
std::unique_lock lk(That->ThunksMutex);
|
||||
std::lock_guard lk(That->ThunksMutex);
|
||||
|
||||
That->Libs.insert(Name);
|
||||
|
||||
int i;
|
||||
for (i = 0; Exports[i].sha256; i++) {
|
||||
@@ -102,7 +296,22 @@ namespace FEXCore {
|
||||
}
|
||||
}
|
||||
|
||||
public:
|
||||
static void IsLibLoaded(void* ArgsRV) {
|
||||
struct ArgsRV_t {
|
||||
const char *Name;
|
||||
bool rv;
|
||||
};
|
||||
|
||||
auto &[Name, rv] = *reinterpret_cast<ArgsRV_t*>(ArgsRV);
|
||||
|
||||
auto CTX = Thread->CTX;
|
||||
auto That = reinterpret_cast<ThunkHandler_impl*>(CTX->ThunkHandler.get());
|
||||
|
||||
{
|
||||
std::shared_lock lk(That->ThunksMutex);
|
||||
rv = That->Libs.contains(Name);
|
||||
}
|
||||
}
|
||||
|
||||
ThunkedFunction* LookupThunk(const IR::SHA256Sum &sha256) {
|
||||
|
||||
@@ -120,15 +329,103 @@ namespace FEXCore {
|
||||
void RegisterTLSState(FEXCore::Core::InternalThreadState *Thread) {
|
||||
::Thread = Thread;
|
||||
}
|
||||
|
||||
ThunkHandler_impl() {
|
||||
}
|
||||
|
||||
~ThunkHandler_impl() {
|
||||
}
|
||||
};
|
||||
|
||||
ThunkHandler* ThunkHandler::Create() {
|
||||
return new ThunkHandler_impl();
|
||||
}
|
||||
|
||||
/**
|
||||
* Generates a host-callable trampoline to call guest functions via the host ABI.
|
||||
*
|
||||
* This trampoline uses the same calling convention as the given HostPacker. Trampolines
|
||||
* are cached, so it's safe to call this function repeatedly on the same arguments without
|
||||
* leaking memory.
|
||||
*
|
||||
* Invoking the returned trampoline has the effect of:
|
||||
* - packing the arguments (using the HostPacker identified by its SHA256)
|
||||
* - performing a host->guest transition
|
||||
* - unpacking the arguments via GuestUnpacker
|
||||
* - calling the function at GuestTarget
|
||||
*
|
||||
* The primary use case of this is ensuring that guest function pointers ("callbacks")
|
||||
* passed to thunked APIs can safely be called by the native host library.
|
||||
*
|
||||
* Returns a pointer to the generated host trampoline and its TrampolineInstanceInfo.
|
||||
*
|
||||
* If HostPacker is zero, the trampoline will be partially initialized and needs to be
|
||||
* finalized with a call to FinalizeHostTrampolineForGuestFunction. A typical use case
|
||||
* is to allocate the trampoline for a given GuestTarget/GuestUnpacker on the guest-side,
|
||||
* and provide the HostPacker host-side.
|
||||
*/
|
||||
__attribute__((visibility("default")))
|
||||
HostToGuestTrampolinePtr* MakeHostTrampolineForGuestFunction(void* HostPacker, uintptr_t GuestTarget, uintptr_t GuestUnpacker) {
|
||||
LOGMAN_THROW_AA_FMT(GuestTarget, "Tried to create host-trampoline to null pointer guest function");
|
||||
|
||||
const auto CTX = Thread->CTX;
|
||||
const auto ThunkHandler = reinterpret_cast<ThunkHandler_impl *>(CTX->ThunkHandler.get());
|
||||
|
||||
const GuestcallInfo gci = { GuestUnpacker, GuestTarget };
|
||||
|
||||
// Try first with shared_lock
|
||||
{
|
||||
std::shared_lock lk(ThunkHandler->ThunksMutex);
|
||||
|
||||
auto found = ThunkHandler->GuestcallToHostTrampoline.find(gci);
|
||||
if (found != ThunkHandler->GuestcallToHostTrampoline.end()) {
|
||||
return found->second;
|
||||
}
|
||||
}
|
||||
|
||||
std::lock_guard lk(ThunkHandler->ThunksMutex);
|
||||
|
||||
// Retry lookup with full lock before making a new trampoline to avoid double trampolines
|
||||
{
|
||||
auto found = ThunkHandler->GuestcallToHostTrampoline.find(gci);
|
||||
if (found != ThunkHandler->GuestcallToHostTrampoline.end()) {
|
||||
return found->second;
|
||||
}
|
||||
}
|
||||
|
||||
LogMan::Msg::DFmt("Thunks: Adding host trampoline for guest function {:#x} via unpacker {:#x}",
|
||||
GuestTarget, GuestUnpacker);
|
||||
|
||||
if (ThunkHandler->HostTrampolineInstanceDataAvailable < HostToGuestTrampolineSize) {
|
||||
const auto allocation_step = 16 * 1024;
|
||||
ThunkHandler->HostTrampolineInstanceDataAvailable = allocation_step;
|
||||
ThunkHandler->HostTrampolineInstanceDataPtr = (uint8_t *)mmap(
|
||||
0, ThunkHandler->HostTrampolineInstanceDataAvailable,
|
||||
PROT_READ | PROT_WRITE | PROT_EXEC,
|
||||
MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ThunkHandler->HostTrampolineInstanceDataPtr != MAP_FAILED, "Failed to mmap HostTrampolineInstanceDataPtr");
|
||||
}
|
||||
|
||||
auto HostTrampoline = reinterpret_cast<HostToGuestTrampolinePtr* const>(ThunkHandler->HostTrampolineInstanceDataPtr);
|
||||
ThunkHandler->HostTrampolineInstanceDataAvailable -= HostToGuestTrampolineSize;
|
||||
ThunkHandler->HostTrampolineInstanceDataPtr += HostToGuestTrampolineSize;
|
||||
memcpy(HostTrampoline, (void*)&HostToGuestTrampolineTemplate, HostToGuestTrampolineSize);
|
||||
GetInstanceInfo(HostTrampoline) = TrampolineInstanceInfo {
|
||||
.HostPacker = HostPacker,
|
||||
.CallCallback = (uintptr_t)&ThunkHandler_impl::CallCallback,
|
||||
.GuestUnpacker = GuestUnpacker,
|
||||
.GuestTarget = GuestTarget
|
||||
};
|
||||
|
||||
ThunkHandler->GuestcallToHostTrampoline[gci] = HostTrampoline;
|
||||
return HostTrampoline;
|
||||
}
|
||||
|
||||
__attribute__((visibility("default")))
|
||||
void FinalizeHostTrampolineForGuestFunction(HostToGuestTrampolinePtr* TrampolineAddress, void* HostPacker) {
|
||||
auto& Trampoline = GetInstanceInfo(TrampolineAddress);
|
||||
|
||||
LOGMAN_THROW_A_FMT(Trampoline.CallCallback == (uintptr_t)&ThunkHandler_impl::CallCallback,
|
||||
"Invalid trampoline at {} passed to {}", fmt::ptr(TrampolineAddress), __FUNCTION__);
|
||||
|
||||
if (!Trampoline.HostPacker) {
|
||||
LogMan::Msg::DFmt("Thunks: Finalizing trampoline at {} with host packer {}", fmt::ptr(TrampolineAddress), fmt::ptr(HostPacker));
|
||||
Trampoline.HostPacker = HostPacker;
|
||||
}
|
||||
}
|
||||
}
|
||||
+29
-19
@@ -6,6 +6,7 @@
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <Interface/Core/LookupCache.h>
|
||||
#include <Interface/GDBJIT/GDBJIT.h>
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
@@ -17,6 +18,7 @@
|
||||
#include <unistd.h>
|
||||
#include <xxhash.h>
|
||||
|
||||
|
||||
namespace FEXCore::IR {
|
||||
AOTIRInlineEntry *AOTIRInlineIndex::GetInlineEntry(uint64_t DataOffset) {
|
||||
uintptr_t This = (uintptr_t)this;
|
||||
@@ -125,7 +127,7 @@ namespace FEXCore::IR {
|
||||
|
||||
auto Array = (AOTIRInlineIndex *)((char*)FilePtr + IndexOffset);
|
||||
|
||||
LOGMAN_THROW_A_FMT(Entry->Array == nullptr && Entry->FilePtr == nullptr, "Entry must not be initialized here");
|
||||
LOGMAN_THROW_AA_FMT(Entry->Array == nullptr && Entry->FilePtr == nullptr, "Entry must not be initialized here");
|
||||
Entry->Array = Array;
|
||||
Entry->FilePtr = FilePtr;
|
||||
Entry->Size = Size;
|
||||
@@ -242,9 +244,9 @@ namespace FEXCore::IR {
|
||||
|
||||
AOTIRCaptureCache::PreGenerateIRFetchResult AOTIRCaptureCache::PreGenerateIRFetch(uint64_t GuestRIP, FEXCore::IR::IRListView *IRList) {
|
||||
auto AOTIRCacheEntry = CTX->SyscallHandler->LookupAOTIRCacheEntry(GuestRIP);
|
||||
|
||||
|
||||
PreGenerateIRFetchResult Result{};
|
||||
|
||||
|
||||
if (AOTIRCacheEntry.Entry) {
|
||||
AOTIRCacheEntry.Entry->ContainsCode = true;
|
||||
|
||||
@@ -253,7 +255,7 @@ namespace FEXCore::IR {
|
||||
|
||||
if (Mod != nullptr)
|
||||
{
|
||||
auto AOTEntry = Mod->Find(GuestRIP - AOTIRCacheEntry.Offset);
|
||||
auto AOTEntry = Mod->Find(GuestRIP - AOTIRCacheEntry.VAFileStart);
|
||||
|
||||
if (AOTEntry) {
|
||||
// verify hash
|
||||
@@ -263,7 +265,7 @@ namespace FEXCore::IR {
|
||||
Result.IRList = AOTEntry->GetIRData();
|
||||
//LogMan::Msg::DFmt("using {} + {:x} -> {:x}\n", file->second.fileid, AOTEntry->first, GuestRIP);
|
||||
|
||||
Result.RAData = AOTEntry->GetRAData();;
|
||||
Result.RAData = AOTEntry->GetRAData()->CreateCopy();
|
||||
Result.DebugData = new FEXCore::Core::DebugData();
|
||||
Result.StartAddr = MappedStart;
|
||||
Result.Length = AOTEntry->GuestLength;
|
||||
@@ -287,12 +289,13 @@ namespace FEXCore::IR {
|
||||
uint64_t GuestRIP,
|
||||
uint64_t StartAddr,
|
||||
uint64_t Length,
|
||||
FEXCore::IR::RegisterAllocationData *RAData,
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData,
|
||||
FEXCore::IR::IRListView *IRList,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
bool GeneratedIR) {
|
||||
|
||||
// Both generated ir and LibraryJITName need a named region lookup
|
||||
if (GeneratedIR || CTX->Config.LibraryJITNaming()) {
|
||||
if (GeneratedIR || CTX->Config.LibraryJITNaming() || CTX->Config.GDBSymbols()) {
|
||||
|
||||
auto AOTIRCacheEntry = CTX->SyscallHandler->LookupAOTIRCacheEntry(GuestRIP);
|
||||
|
||||
@@ -301,18 +304,26 @@ namespace FEXCore::IR {
|
||||
CTX->Symbols.RegisterNamedRegion(CodePtr, DebugData->HostCodeSize, AOTIRCacheEntry.Entry->Filename);
|
||||
}
|
||||
|
||||
if (CTX->Config.GDBSymbols()) {
|
||||
GDBJITRegister(AOTIRCacheEntry.Entry, AOTIRCacheEntry.VAFileStart, GuestRIP, (uintptr_t)CodePtr, DebugData);
|
||||
}
|
||||
|
||||
// Add to AOT cache if aot generation is enabled
|
||||
if (GeneratedIR && RAData &&
|
||||
(CTX->Config.AOTIRCapture() || CTX->Config.AOTIRGenerate())) {
|
||||
|
||||
auto hash = XXH3_64bits((void*)StartAddr, Length);
|
||||
|
||||
auto LocalRIP = GuestRIP - AOTIRCacheEntry.Offset;
|
||||
auto LocalStartAddr = StartAddr - AOTIRCacheEntry.Offset;
|
||||
auto LocalRIP = GuestRIP - AOTIRCacheEntry.VAFileStart;
|
||||
auto LocalStartAddr = StartAddr - AOTIRCacheEntry.VAFileStart;
|
||||
auto FileId = AOTIRCacheEntry.Entry->FileId;
|
||||
// The underlying pointer and the unique_ptr deleter for RAData must
|
||||
// be marshalled separately to the lambda below. Otherwise, the
|
||||
// lambda can't be used as an std::function due to being non-copyable
|
||||
auto RADataCopy = RAData->CreateCopy();
|
||||
auto RADataCopyDeleter = RADataCopy.get_deleter();
|
||||
auto IRListCopy = IRList->CreateCopy();
|
||||
AOTIRCaptureCacheWriteoutQueue_Append([this, LocalRIP, LocalStartAddr, Length, hash, IRListCopy, RADataCopy, FileId]() {
|
||||
AOTIRCaptureCacheWriteoutQueue_Append([this, LocalRIP, LocalStartAddr, Length, hash, IRListCopy, RADataCopy=RADataCopy.release(), RADataCopyDeleter, FileId]() {
|
||||
|
||||
// It is guaranteed via AOTIRCaptureCacheWriteoutLock and AOTIRCaptureCacheWriteoutFlusing that this will not run concurrently
|
||||
// Memory coherency is guaranteed via AOTIRCaptureCacheWriteoutLock
|
||||
@@ -325,7 +336,7 @@ namespace FEXCore::IR {
|
||||
AotFile->Stream->write((char*)&tag, sizeof(tag));
|
||||
}
|
||||
AotFile->AppendAOTIRCaptureCache(LocalRIP, LocalStartAddr, Length, hash, IRListCopy, RADataCopy);
|
||||
FEXCore::Allocator::free(RADataCopy);
|
||||
RADataCopyDeleter(RADataCopy);
|
||||
delete IRListCopy;
|
||||
});
|
||||
|
||||
@@ -339,18 +350,17 @@ namespace FEXCore::IR {
|
||||
|
||||
// Insert to caches if we generated IR
|
||||
if (GeneratedIR) {
|
||||
if (Thread->CPUBackend->NeedsRetainedIRCopy()) {
|
||||
if (CTX->GetGdbServerStatus()) {
|
||||
// Add to thread local ir cache
|
||||
Core::LocalIREntry Entry = {StartAddr, Length, decltype(Entry.IR)(IRList), decltype(Entry.RAData)(RAData), decltype(Entry.DebugData)(DebugData)};
|
||||
Core::LocalIREntry Entry = {StartAddr, Length, decltype(Entry.IR)(IRList), std::move(RAData), decltype(Entry.DebugData)(DebugData)};
|
||||
|
||||
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
|
||||
Thread->LocalIRCache.insert({GuestRIP, std::move(Entry)});
|
||||
Thread->DebugStore.insert({GuestRIP, std::move(Entry)});
|
||||
}
|
||||
else {
|
||||
// If the IR doesn't need to be retained then we can just delete it now
|
||||
delete DebugData;
|
||||
delete RAData;
|
||||
delete IRList;
|
||||
if (IRList->IsCopy()) delete IRList;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -374,10 +384,10 @@ namespace FEXCore::IR {
|
||||
|
||||
std::unique_lock lk(AOTIRCacheLock);
|
||||
|
||||
auto Inserted = AOTIRCache.insert({fileid, AOTIRCacheEntry{0, 0, 0, fileid, filename, false}});
|
||||
auto Inserted = AOTIRCache.insert({fileid, AOTIRCacheEntry { .FileId = fileid, .Filename = filename }});
|
||||
auto Entry = &(Inserted.first->second);
|
||||
|
||||
LOGMAN_THROW_A_FMT(Entry->Array == nullptr, "Duplicate LoadAOTIRCacheEntry");
|
||||
LOGMAN_THROW_AA_FMT(Entry->Array == nullptr, "Duplicate LoadAOTIRCacheEntry");
|
||||
|
||||
if (CTX->Config.AOTIRLoad && AOTIRLoader) {
|
||||
auto streamfd = AOTIRLoader(fileid);
|
||||
@@ -393,7 +403,7 @@ namespace FEXCore::IR {
|
||||
}
|
||||
|
||||
void AOTIRCaptureCache::UnloadAOTIRCacheEntry(AOTIRCacheEntry *Entry) {
|
||||
LOGMAN_THROW_A_FMT(Entry != nullptr, "Removing not existing entry");
|
||||
LOGMAN_THROW_AA_FMT(Entry != nullptr, "Removing not existing entry");
|
||||
|
||||
if (Entry->Array) {
|
||||
FEXCore::Allocator::munmap(Entry->FilePtr, Entry->Size);
|
||||
|
||||
+5
-2
@@ -1,5 +1,6 @@
|
||||
#pragma once
|
||||
|
||||
#include "FEXCore/IR/RegisterAllocationData.h"
|
||||
#include <FEXCore/Config/Config.h>
|
||||
|
||||
#include <atomic>
|
||||
@@ -11,6 +12,7 @@
|
||||
#include <unordered_map>
|
||||
#include <shared_mutex>
|
||||
#include <queue>
|
||||
#include <FEXCore/HLE/SourcecodeResolver.h>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct DebugData;
|
||||
@@ -74,6 +76,7 @@ namespace FEXCore::IR {
|
||||
AOTIRInlineIndex *Array;
|
||||
void *FilePtr;
|
||||
size_t Size;
|
||||
std::unique_ptr<FEXCore::HLE::SourcecodeMap> SourcecodeMap;
|
||||
std::string FileId;
|
||||
std::string Filename;
|
||||
bool ContainsCode;
|
||||
@@ -93,7 +96,7 @@ namespace FEXCore::IR {
|
||||
|
||||
struct PreGenerateIRFetchResult {
|
||||
FEXCore::IR::IRListView *IRList {};
|
||||
FEXCore::IR::RegisterAllocationData *RAData {};
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData {};
|
||||
FEXCore::Core::DebugData *DebugData {};
|
||||
uint64_t StartAddr {};
|
||||
uint64_t Length {};
|
||||
@@ -106,7 +109,7 @@ namespace FEXCore::IR {
|
||||
uint64_t GuestRIP,
|
||||
uint64_t StartAddr,
|
||||
uint64_t Length,
|
||||
FEXCore::IR::RegisterAllocationData *RAData,
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData,
|
||||
FEXCore::IR::IRListView *IRList,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
bool GeneratedIR);
|
||||
|
||||
+33
-24
@@ -123,12 +123,12 @@
|
||||
"constexpr FEXCore::IR::MemOffsetType MEM_OFFSET_UXTW {1}",
|
||||
"constexpr FEXCore::IR::MemOffsetType MEM_OFFSET_SXTW {2}",
|
||||
|
||||
"constexpr FEXCore::IR::BreakReason Break_Unimplemented {0}",
|
||||
"constexpr FEXCore::IR::BreakReason Break_Interrupt {1}",
|
||||
"constexpr FEXCore::IR::BreakReason Break_Interrupt3 {2}",
|
||||
"constexpr FEXCore::IR::BreakReason Break_Halt {3}",
|
||||
"constexpr FEXCore::IR::BreakReason Break_Overflow {4}",
|
||||
"constexpr FEXCore::IR::BreakReason Break_InvalidInstruction {5}"
|
||||
"struct BreakDefinition {",
|
||||
" uint16_t ErrorRegister;",
|
||||
" uint8_t Signal;",
|
||||
" uint8_t TrapNumber;",
|
||||
" uint8_t si_code;",
|
||||
"};"
|
||||
],
|
||||
"IRTypes" : {
|
||||
"i1": "bool",
|
||||
@@ -150,7 +150,7 @@
|
||||
"SyscallFlags": "FEXCore::IR::SyscallFlags",
|
||||
"SHA256Sum": "SHA256Sum",
|
||||
"MemOffsetType": "MemOffsetType",
|
||||
"BreakReason": "BreakReason",
|
||||
"BreakDefinition": "BreakDefinition",
|
||||
"RoundType": "RoundType"
|
||||
},
|
||||
"Ops": {
|
||||
@@ -181,13 +181,18 @@
|
||||
"RAOverride": "0"
|
||||
},
|
||||
|
||||
"GuestOpcode u32:$GuestEntryOffset": {
|
||||
"Desc": ["Marks the beginning of a guest opcode"],
|
||||
"HasSideEffects": true
|
||||
},
|
||||
|
||||
"GPR = ValidateCode u64:$CodeOriginalLow, u64:$CodeOriginalhigh, i64:$Offset, u8:$CodeLength": {
|
||||
"HasSideEffects": true,
|
||||
"HasDest": true,
|
||||
"DestSize": "8"
|
||||
},
|
||||
|
||||
"RemoveThreadCodeEntry": {
|
||||
"ThreadRemoveCodeEntry": {
|
||||
"HasSideEffects": true
|
||||
},
|
||||
|
||||
@@ -256,7 +261,7 @@
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "GetOpSize(_NewRIP)"
|
||||
},
|
||||
"Break BreakReason:$Reason, u8:$Literal": {
|
||||
"Break BreakDefinition:$Reason": {
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"SignalReturn": {
|
||||
@@ -294,12 +299,6 @@
|
||||
],
|
||||
"DestSize": "16",
|
||||
"NumElements": "2"
|
||||
},
|
||||
"GuestCallDirect u64:$RIP, u64:$NextRIP": {
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"GuestCallIndirect GPR:$RIP, u64:$NextRIP": {
|
||||
"HasSideEffects": true
|
||||
}
|
||||
},
|
||||
"Moves": {
|
||||
@@ -943,7 +942,7 @@
|
||||
},
|
||||
|
||||
"FPR = VBitcast u8:#RegisterSize, u8:#ElementSize, FPR:$Source": {
|
||||
"Dest": ["Workaround for issue with LLVM breaking when loading scalar elements to vectors"],
|
||||
"Desc": ["Workaround for issue with LLVM breaking when loading scalar elements to vectors"],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -1277,12 +1276,12 @@
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
},
|
||||
"FPR = VUMull2 u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Dest": "Multiplies the high elements with size extension",
|
||||
"Desc": "Multiplies the high elements with size extension",
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
},
|
||||
"FPR = VSMull2 u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Dest": "Multiplies the high elements with size extension",
|
||||
"Desc": "Multiplies the high elements with size extension",
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
},
|
||||
@@ -1454,33 +1453,43 @@
|
||||
},
|
||||
"Crypto": {
|
||||
"FPR = VAESImc FPR:$Vector": {
|
||||
"Dest": "Does a stage of the inverse mix column transformation",
|
||||
"Desc": "Does a stage of the inverse mix column transformation",
|
||||
"DestSize": "16"
|
||||
},
|
||||
"FPR = VAESEnc FPR:$State, FPR:$Key": {
|
||||
"Dest": "Does a step of AES encryption",
|
||||
"Desc": "Does a step of AES encryption",
|
||||
"DestSize": "16"
|
||||
},
|
||||
"FPR = VAESEncLast FPR:$State, FPR:$Key": {
|
||||
"Dest": "Does the last step of AES encryption",
|
||||
"Desc": "Does the last step of AES encryption",
|
||||
"DestSize": "16"
|
||||
},
|
||||
"FPR = VAESDec FPR:$State, FPR:$Key": {
|
||||
"Dest": "Does a step of AES decryption",
|
||||
"Desc": "Does a step of AES decryption",
|
||||
"DestSize": "16"
|
||||
},
|
||||
"FPR = VAESDecLast FPR:$State, FPR:$Key": {
|
||||
"Dest": "Does the last step of AES decryption",
|
||||
"Desc": "Does the last step of AES decryption",
|
||||
"DestSize": "16"
|
||||
},
|
||||
"FPR = VAESKeyGenAssist FPR:$Src, u8:$RCON": {
|
||||
"Dest": "Assists in key generation",
|
||||
"Desc": "Assists in key generation",
|
||||
"DestSize": "16"
|
||||
},
|
||||
"GPR = CRC32 GPR:$Src1, GPR:$Src2, u8:$SrcSize": {
|
||||
"Desc": ["CRC32 using polynomial 0x1EDC6F41"
|
||||
],
|
||||
"DestSize": "std::max<uint8_t>(4, GetOpSize(_Src1))"
|
||||
},
|
||||
"FPR = PCLMUL FPR:$Src1, FPR:$Src2, u8:$Selector": {
|
||||
"Desc": [
|
||||
"Performs carryless multiplication of 64-bit elements depending on the selector.",
|
||||
"Selector = 0b00000000: Uses low 64-bit elements from both input vectors",
|
||||
"Selector = 0b00000001: Uses high 64-bit element from Src1 and low 64-bit element from Src2",
|
||||
"Selector = 0b00010000: Uses low 64-bit element from Src1 and high 64-bit element from Src2",
|
||||
"Selector = 0b00010001: Uses high 64-bit elements from both input vectors"
|
||||
],
|
||||
"DestSize": "16"
|
||||
}
|
||||
},
|
||||
"F64": {
|
||||
|
||||
+6
-1
@@ -182,7 +182,12 @@ static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView const*
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView const* IR, FEXCore::IR::BreakDefinition Arg) {
|
||||
*out << "{" << Arg.ErrorRegister << ".";
|
||||
*out << static_cast<uint32_t>(Arg.Signal) << ".";
|
||||
*out << static_cast<uint32_t>(Arg.TrapNumber) << ".";
|
||||
*out << static_cast<uint32_t>(Arg.si_code) << "}";
|
||||
}
|
||||
|
||||
void Dump(std::stringstream *out, IRListView const* IR, IR::RegisterAllocationData *RAData) {
|
||||
auto HeaderOp = IR->GetHeader();
|
||||
|
||||
+25
-2
@@ -8,6 +8,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <array>
|
||||
@@ -16,6 +17,27 @@ $end_info$
|
||||
#include <vector>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
bool IsFragmentExit(FEXCore::IR::IROps Op) {
|
||||
switch (Op) {
|
||||
case OP_EXITFUNCTION:
|
||||
case OP_BREAK:
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
bool IsBlockExit(FEXCore::IR::IROps Op) {
|
||||
switch(Op) {
|
||||
case OP_JUMP:
|
||||
case OP_CONDJUMP:
|
||||
return true;
|
||||
default:
|
||||
return IsFragmentExit(Op);
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::IR::RegisterClassType IREmitter::WalkFindRegClass(OrderedNode *Node) {
|
||||
auto Class = GetOpRegClass(Node);
|
||||
switch (Class) {
|
||||
@@ -66,7 +88,8 @@ FEXCore::IR::RegisterClassType IREmitter::WalkFindRegClass(OrderedNode *Node) {
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled op type: {} {} in argument class validation", IROp->Op, GetOpName(Node));
|
||||
LOGMAN_MSG_A_FMT("Unhandled op type: {} {} in argument class validation",
|
||||
ToUnderlying(IROp->Op), GetOpName(Node));
|
||||
break;
|
||||
}
|
||||
return InvalidClass;
|
||||
@@ -146,7 +169,7 @@ IREmitter::IRPair<IROp_CodeBlock> IREmitter::CreateNewCodeBlockAfter(OrderedNode
|
||||
if (insertAfter) {
|
||||
LinkCodeBlocks(insertAfter, CodeNode);
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(CurrentCodeBlock != nullptr, "CurrentCodeBlock must not be null here");
|
||||
LOGMAN_THROW_AA_FMT(CurrentCodeBlock != nullptr, "CurrentCodeBlock must not be null here");
|
||||
|
||||
// Find last block
|
||||
auto LastBlock = CurrentCodeBlock;
|
||||
|
||||
+28
-14
@@ -252,22 +252,36 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
}
|
||||
|
||||
template<>
|
||||
std::pair<DecodeFailure, FEXCore::IR::BreakReason> DecodeValue(const std::string &Arg) {
|
||||
static constexpr std::array<std::string_view, 6> Names = {
|
||||
"Unimplemented",
|
||||
"Interrupt",
|
||||
"Interrupt3",
|
||||
"Halt",
|
||||
"Overfloat",
|
||||
"InvalidInstruction",
|
||||
};
|
||||
std::pair<DecodeFailure, FEXCore::IR::BreakDefinition> DecodeValue(const std::string &Arg) {
|
||||
uint32_t tmp{};
|
||||
std::stringstream ss{Arg};
|
||||
BreakDefinition Reason{};
|
||||
|
||||
for (size_t i = 0; i < Names.size(); ++i) {
|
||||
if (Names[i] == Arg) {
|
||||
return {DecodeFailure::DECODE_OKAY, BreakReason{static_cast<uint8_t>(i)}};
|
||||
}
|
||||
// Seek past '{'
|
||||
ss.seekg(1, std::ios::cur);
|
||||
ss >> Reason.ErrorRegister;
|
||||
|
||||
// Seek past '.'
|
||||
ss.seekg(1, std::ios::cur);
|
||||
ss >> tmp;
|
||||
Reason.Signal = tmp;
|
||||
|
||||
// Seek past '.'
|
||||
ss.seekg(1, std::ios::cur);
|
||||
ss >> tmp;
|
||||
Reason.TrapNumber = tmp;
|
||||
|
||||
// Seek past '.'
|
||||
ss.seekg(1, std::ios::cur);
|
||||
ss >> tmp;
|
||||
Reason.si_code = tmp;
|
||||
|
||||
if (ss.fail()) {
|
||||
return {DecodeFailure::DECODE_INVALIDCHAR, {}};
|
||||
}
|
||||
else {
|
||||
return {DecodeFailure::DECODE_OKAY, Reason};
|
||||
}
|
||||
return {DecodeFailure::DECODE_INVALID_BREAKTYPE, {}};
|
||||
}
|
||||
|
||||
template<>
|
||||
|
||||
+6
-6
@@ -20,7 +20,7 @@ void PassManager::AddDefaultPasses(FEXCore::Context::Context *ctx, bool InlineCo
|
||||
FEX_CONFIG_OPT(DisablePasses, O0);
|
||||
|
||||
if (!DisablePasses()) {
|
||||
InsertPass(CreateContextLoadStoreElimination());
|
||||
InsertPass(CreateContextLoadStoreElimination(ctx->HostFeatures.SupportsAVX));
|
||||
|
||||
if (Is64BitMode()) {
|
||||
// This needs to run after RCLSE
|
||||
@@ -28,7 +28,7 @@ void PassManager::AddDefaultPasses(FEXCore::Context::Context *ctx, bool InlineCo
|
||||
InsertPass(CreateLongDivideEliminationPass());
|
||||
}
|
||||
|
||||
InsertPass(CreateDeadStoreElimination());
|
||||
InsertPass(CreateDeadStoreElimination(ctx->HostFeatures.SupportsAVX));
|
||||
InsertPass(CreatePassDeadCodeElimination());
|
||||
InsertPass(CreateConstProp(InlineConstants, ctx->HostFeatures.SupportsTSOImm9));
|
||||
|
||||
@@ -39,12 +39,12 @@ void PassManager::AddDefaultPasses(FEXCore::Context::Context *ctx, bool InlineCo
|
||||
|
||||
// only do SRA if enabled and JIT
|
||||
if (InlineConstants && StaticRegisterAllocation)
|
||||
InsertPass(CreateStaticRegisterAllocationPass());
|
||||
InsertPass(CreateStaticRegisterAllocationPass(ctx->HostFeatures.SupportsAVX));
|
||||
}
|
||||
else {
|
||||
// only do SRA if enabled and JIT
|
||||
if (InlineConstants && StaticRegisterAllocation)
|
||||
InsertPass(CreateStaticRegisterAllocationPass());
|
||||
InsertPass(CreateStaticRegisterAllocationPass(ctx->HostFeatures.SupportsAVX));
|
||||
}
|
||||
|
||||
// If the IR is compacted post-RA then the node indexing gets messed up and the backend isn't able to find the register assigned to a node
|
||||
@@ -61,8 +61,8 @@ void PassManager::AddDefaultValidationPasses() {
|
||||
#endif
|
||||
}
|
||||
|
||||
void PassManager::InsertRegisterAllocationPass(bool OptimizeSRA) {
|
||||
InsertPass(IR::CreateRegisterAllocationPass(GetPass("Compaction"), OptimizeSRA), "RA");
|
||||
void PassManager::InsertRegisterAllocationPass(bool OptimizeSRA, bool SupportsAVX) {
|
||||
InsertPass(IR::CreateRegisterAllocationPass(GetPass("Compaction"), OptimizeSRA, SupportsAVX), "RA");
|
||||
}
|
||||
|
||||
bool PassManager::Run(IREmitter *IREmit) {
|
||||
|
||||
+1
-1
@@ -52,7 +52,7 @@ public:
|
||||
return PassPtr;
|
||||
}
|
||||
|
||||
void InsertRegisterAllocationPass(bool OptimizeSRA);
|
||||
void InsertRegisterAllocationPass(bool OptimizeSRA, bool SupportsAVX);
|
||||
|
||||
bool Run(IREmitter *IREmit);
|
||||
|
||||
|
||||
+6
-4
@@ -12,14 +12,16 @@ class RegisterAllocationPass;
|
||||
class RegisterAllocationData;
|
||||
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool InlineConstants, bool SupportsTSOImm9);
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateContextLoadStoreElimination();
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateContextLoadStoreElimination(bool SupportsAVX);
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateSyscallOptimization();
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateDeadFlagCalculationEliminination();
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateDeadStoreElimination();
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateDeadStoreElimination(bool SupportsAVX);
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreatePassDeadCodeElimination();
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateIRCompaction(FEXCore::Utils::IntrusivePooledAllocator &Allocator);
|
||||
std::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass(FEXCore::IR::Pass* CompactionPass, bool OptimizeSRA);
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateStaticRegisterAllocationPass();
|
||||
std::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass(FEXCore::IR::Pass* CompactionPass,
|
||||
bool OptimizeSRA,
|
||||
bool SupportsAVX);
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateStaticRegisterAllocationPass(bool SupportsAVX);
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateLongDivideEliminationPass();
|
||||
|
||||
namespace Validation {
|
||||
|
||||
@@ -299,7 +299,7 @@ void ConstProp::FCMPOptimization(IREmitter *IREmit, const IRListView& CurrentIR)
|
||||
auto ghf = IROp->CW<IR::IROp_GetHostFlag>();
|
||||
|
||||
auto fcmp = IREmit->GetOpHeader(ghf->Value)->CW<IR::IROp_FCmp>();
|
||||
LOGMAN_THROW_A_FMT(fcmp->Header.Op == OP_FCMP || fcmp->Header.Op == OP_F80CMP, "Unexpected OP_GETHOSTFLAG source");
|
||||
LOGMAN_THROW_AA_FMT(fcmp->Header.Op == OP_FCMP || fcmp->Header.Op == OP_F80CMP, "Unexpected OP_GETHOSTFLAG source");
|
||||
if(fcmp->Header.Op == OP_FCMP) {
|
||||
fcmp->Flags |= 1 << ghf->Flag;
|
||||
}
|
||||
|
||||
+86
-81
@@ -25,7 +25,7 @@ $end_info$
|
||||
namespace {
|
||||
struct ContextMemberClassification {
|
||||
size_t Offset;
|
||||
uint8_t Size;
|
||||
uint16_t Size;
|
||||
};
|
||||
|
||||
enum LastAccessType {
|
||||
@@ -76,10 +76,7 @@ namespace {
|
||||
std::vector<ContextMemberInfo> ClassificationInfo;
|
||||
};
|
||||
|
||||
constexpr static std::array<LastAccessType, 16> DefaultAccess = {
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_INVALID, // PAD
|
||||
constexpr static std::array<LastAccessType, 15> DefaultAccess = {
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
@@ -88,14 +85,16 @@ namespace {
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_INVALID, // PAD
|
||||
ACCESS_NONE,
|
||||
ACCESS_INVALID, // SSE padding in non-AVX case
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
};
|
||||
|
||||
static void ClassifyContextStruct(ContextInfo *ContextClassificationInfo) {
|
||||
static void ClassifyContextStruct(ContextInfo *ContextClassificationInfo, bool SupportsAVX) {
|
||||
auto ContextClassification = &ContextClassificationInfo->ClassificationInfo;
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
@@ -107,43 +106,23 @@ namespace {
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
for (size_t i = 0; i < 16; ++i) {
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_GPRS; ++i) {
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, gregs[0]) + sizeof(FEXCore::Core::CPUState::gregs[0]) * i,
|
||||
sizeof(FEXCore::Core::CPUState::gregs[0]),
|
||||
FEXCore::Core::CPUState::GPR_REG_SIZE,
|
||||
},
|
||||
DefaultAccess[1],
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, gregs[16]),
|
||||
sizeof(uint64_t),
|
||||
},
|
||||
DefaultAccess[2], ///< NOP padding
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
for (size_t i = 0; i < 16; ++i) {
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, xmm[0][0]) + sizeof(FEXCore::Core::CPUState::xmm[0]) * i,
|
||||
sizeof(FEXCore::Core::CPUState::xmm[0]),
|
||||
},
|
||||
DefaultAccess[3],
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, es),
|
||||
sizeof(FEXCore::Core::CPUState::es),
|
||||
},
|
||||
DefaultAccess[5],
|
||||
DefaultAccess[2],
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
@@ -152,7 +131,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, cs),
|
||||
sizeof(FEXCore::Core::CPUState::cs),
|
||||
},
|
||||
DefaultAccess[6],
|
||||
DefaultAccess[3],
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
@@ -161,7 +140,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, ss),
|
||||
sizeof(FEXCore::Core::CPUState::ss),
|
||||
},
|
||||
DefaultAccess[7],
|
||||
DefaultAccess[4],
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
@@ -170,7 +149,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, ds),
|
||||
sizeof(FEXCore::Core::CPUState::ds),
|
||||
},
|
||||
DefaultAccess[8],
|
||||
DefaultAccess[5],
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
@@ -179,7 +158,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, gs),
|
||||
sizeof(FEXCore::Core::CPUState::gs),
|
||||
},
|
||||
DefaultAccess[4],
|
||||
DefaultAccess[6],
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
@@ -188,49 +167,73 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, fs),
|
||||
sizeof(FEXCore::Core::CPUState::fs),
|
||||
},
|
||||
DefaultAccess[9],
|
||||
DefaultAccess[7],
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
for (size_t i = 0; i < (sizeof(FEXCore::Core::CPUState::flags) / sizeof(FEXCore::Core::CPUState::flags[0])); ++i) {
|
||||
if (SupportsAVX) {
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; ++i) {
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.avx.data[0][0]) + FEXCore::Core::CPUState::XMM_AVX_REG_SIZE * i,
|
||||
FEXCore::Core::CPUState::XMM_AVX_REG_SIZE,
|
||||
},
|
||||
DefaultAccess[8],
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
} else {
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; ++i) {
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.sse.data[0][0]) + FEXCore::Core::CPUState::XMM_SSE_REG_SIZE * i,
|
||||
FEXCore::Core::CPUState::XMM_SSE_REG_SIZE,
|
||||
},
|
||||
DefaultAccess[8],
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.sse.pad[0][0]),
|
||||
static_cast<uint16_t>(FEXCore::Core::CPUState::XMM_SSE_REG_SIZE * FEXCore::Core::CPUState::NUM_XMMS),
|
||||
},
|
||||
DefaultAccess[9],
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_FLAGS; ++i) {
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, flags[0]) + sizeof(FEXCore::Core::CPUState::flags[0]) * i,
|
||||
sizeof(FEXCore::Core::CPUState::flags[0]),
|
||||
FEXCore::Core::CPUState::FLAG_SIZE,
|
||||
},
|
||||
DefaultAccess[10],
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, flags[48]),
|
||||
sizeof(uint64_t),
|
||||
},
|
||||
DefaultAccess[11], ///< NOP padding
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
for (size_t i = 0; i < 8; ++i) {
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_MMS; ++i) {
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, mm[0][0]) + sizeof(FEXCore::Core::CPUState::mm[0]) * i,
|
||||
sizeof(FEXCore::Core::CPUState::mm[0]),
|
||||
FEXCore::Core::CPUState::MM_REG_SIZE
|
||||
},
|
||||
DefaultAccess[12],
|
||||
DefaultAccess[11],
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
|
||||
// GDTs
|
||||
for (size_t i = 0; i < 32; ++i) {
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_GDTS; ++i) {
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, gdt[0]) + sizeof(FEXCore::Core::CPUState::gdt[0]) * i,
|
||||
sizeof(FEXCore::Core::CPUState::gdt[0]),
|
||||
},
|
||||
DefaultAccess[13],
|
||||
DefaultAccess[12],
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
@@ -241,7 +244,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, FCW),
|
||||
sizeof(FEXCore::Core::CPUState::FCW),
|
||||
},
|
||||
DefaultAccess[14],
|
||||
DefaultAccess[13],
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
@@ -251,7 +254,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, FTW),
|
||||
sizeof(FEXCore::Core::CPUState::FTW),
|
||||
},
|
||||
DefaultAccess[15],
|
||||
DefaultAccess[14],
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
@@ -266,7 +269,7 @@ namespace {
|
||||
ClassifiedStructSize += it.Class.Size;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(ClassifiedStructSize == sizeof(FEXCore::Core::CPUState),
|
||||
LOGMAN_THROW_AA_FMT(ClassifiedStructSize == sizeof(FEXCore::Core::CPUState),
|
||||
"Classified CPUStruct size doesn't match real CPUState struct size! {} (classified) != {} (real)",
|
||||
ClassifiedStructSize, sizeof(FEXCore::Core::CPUState));
|
||||
|
||||
@@ -275,7 +278,7 @@ namespace {
|
||||
ContextClassificationInfo->Lookup.size(), sizeof(FEXCore::Core::CPUState));
|
||||
}
|
||||
|
||||
static void ResetClassificationAccesses(ContextInfo *ContextClassificationInfo) {
|
||||
static void ResetClassificationAccesses(ContextInfo *ContextClassificationInfo, bool SupportsAVX) {
|
||||
auto ContextClassification = &ContextClassificationInfo->ClassificationInfo;
|
||||
|
||||
auto SetAccess = [&](size_t Offset, auto Access) {
|
||||
@@ -286,39 +289,39 @@ namespace {
|
||||
};
|
||||
size_t Offset = 0;
|
||||
SetAccess(Offset++, DefaultAccess[0]);
|
||||
for (size_t i = 0; i < 16; ++i) {
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_GPRS; ++i) {
|
||||
SetAccess(Offset++, DefaultAccess[1]);
|
||||
}
|
||||
|
||||
SetAccess(Offset++, DefaultAccess[2]);
|
||||
|
||||
for (size_t i = 0; i < 16; ++i) {
|
||||
SetAccess(Offset++, DefaultAccess[3]);
|
||||
}
|
||||
|
||||
SetAccess(Offset++, DefaultAccess[3]);
|
||||
SetAccess(Offset++, DefaultAccess[4]);
|
||||
SetAccess(Offset++, DefaultAccess[5]);
|
||||
SetAccess(Offset++, DefaultAccess[6]);
|
||||
SetAccess(Offset++, DefaultAccess[7]);
|
||||
SetAccess(Offset++, DefaultAccess[8]);
|
||||
SetAccess(Offset++, DefaultAccess[9]);
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; ++i) {
|
||||
SetAccess(Offset++, DefaultAccess[8]);
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < (sizeof(FEXCore::Core::CPUState::flags) / sizeof(FEXCore::Core::CPUState::flags[0])); ++i) {
|
||||
if (!SupportsAVX) {
|
||||
SetAccess(Offset++, DefaultAccess[9]);
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_FLAGS; ++i) {
|
||||
SetAccess(Offset++, DefaultAccess[10]);
|
||||
}
|
||||
|
||||
SetAccess(Offset++, DefaultAccess[11]);
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_MMS; ++i) {
|
||||
SetAccess(Offset++, DefaultAccess[11]);
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < 8; ++i) {
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_GDTS; ++i) {
|
||||
SetAccess(Offset++, DefaultAccess[12]);
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < 32; ++i) {
|
||||
SetAccess(Offset++, DefaultAccess[13]);
|
||||
}
|
||||
|
||||
SetAccess(Offset++, DefaultAccess[13]);
|
||||
SetAccess(Offset++, DefaultAccess[14]);
|
||||
SetAccess(Offset++, DefaultAccess[15]);
|
||||
}
|
||||
|
||||
struct BlockInfo {
|
||||
@@ -330,8 +333,8 @@ namespace {
|
||||
|
||||
class RCLSE final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
RCLSE() {
|
||||
ClassifyContextStruct(&ClassifiedStruct);
|
||||
explicit RCLSE(bool SupportsAVX_) : SupportsAVX{SupportsAVX_} {
|
||||
ClassifyContextStruct(&ClassifiedStruct, SupportsAVX);
|
||||
DCE = FEXCore::IR::CreatePassDeadCodeElimination();
|
||||
}
|
||||
bool Run(FEXCore::IR::IREmitter *IREmit) override;
|
||||
@@ -341,6 +344,8 @@ private:
|
||||
ContextInfo ClassifiedStruct;
|
||||
std::unordered_map<FEXCore::IR::NodeID, BlockInfo> OffsetToBlockMap;
|
||||
|
||||
bool SupportsAVX;
|
||||
|
||||
ContextMemberInfo *FindMemberInfo(ContextInfo *ClassifiedInfo, uint32_t Offset, uint8_t Size);
|
||||
ContextMemberInfo *RecordAccess(ContextMemberInfo *Info, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size, LastAccessType AccessType, FEXCore::IR::OrderedNode *Node, FEXCore::IR::OrderedNode *StoreNode = nullptr);
|
||||
ContextMemberInfo *RecordAccess(ContextInfo *ClassifiedInfo, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size, LastAccessType AccessType, FEXCore::IR::OrderedNode *Node, FEXCore::IR::OrderedNode *StoreNode = nullptr);
|
||||
@@ -355,8 +360,8 @@ ContextMemberInfo *RCLSE::FindMemberInfo(ContextInfo *ContextClassificationInfo,
|
||||
}
|
||||
|
||||
ContextMemberInfo *RCLSE::RecordAccess(ContextMemberInfo *Info, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size, LastAccessType AccessType, FEXCore::IR::OrderedNode *Node, FEXCore::IR::OrderedNode *StoreNode) {
|
||||
LOGMAN_THROW_A_FMT((Offset + Size) <= (Info->Class.Offset + Info->Class.Size), "Access to context item went over member size");
|
||||
LOGMAN_THROW_A_FMT(Info->Accessed != ACCESS_INVALID, "Tried to access invalid member");
|
||||
LOGMAN_THROW_AA_FMT((Offset + Size) <= (Info->Class.Offset + Info->Class.Size), "Access to context item went over member size");
|
||||
LOGMAN_THROW_AA_FMT(Info->Accessed != ACCESS_INVALID, "Tried to access invalid member");
|
||||
|
||||
// If we aren't fully overwriting the member then it is a partial write that we need to track
|
||||
if (Size < Info->Class.Size) {
|
||||
@@ -483,7 +488,7 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
auto BlockOp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
auto BlockEnd = IREmit->GetIterator(BlockOp->Last);
|
||||
|
||||
ResetClassificationAccesses(&LocalInfo);
|
||||
ResetClassificationAccesses(&LocalInfo, SupportsAVX);
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
if (IROp->Op == OP_STORECONTEXT) {
|
||||
@@ -623,7 +628,7 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
auto Op = IROp->CW<IR::IROp_InvalidateFlags>();
|
||||
|
||||
// Loop through non-reserved flag stores and eliminate unused ones.
|
||||
for (unsigned F = 0; F < 32; F++) {
|
||||
for (size_t F = 0; F < Core::CPUState::NUM_EFLAG_BITS; F++) {
|
||||
if (!(Op->Flags & (1ULL << F))) {
|
||||
continue;
|
||||
}
|
||||
@@ -675,14 +680,14 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::OPTIMIZETHROUGH) != FEXCore::IR::SyscallFlags::OPTIMIZETHROUGH) {
|
||||
// We can't track through these
|
||||
ResetClassificationAccesses(&LocalInfo);
|
||||
ResetClassificationAccesses(&LocalInfo, SupportsAVX);
|
||||
}
|
||||
}
|
||||
else if (IROp->Op == OP_STORECONTEXTINDEXED ||
|
||||
IROp->Op == OP_LOADCONTEXTINDEXED ||
|
||||
IROp->Op == OP_BREAK) {
|
||||
// We can't track through these
|
||||
ResetClassificationAccesses(&LocalInfo);
|
||||
ResetClassificationAccesses(&LocalInfo, SupportsAVX);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -710,8 +715,8 @@ bool RCLSE::Run(FEXCore::IR::IREmitter *IREmit) {
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateContextLoadStoreElimination() {
|
||||
return std::make_unique<RCLSE>();
|
||||
std::unique_ptr<FEXCore::IR::Pass> CreateContextLoadStoreElimination(bool SupportsAVX) {
|
||||
return std::make_unique<RCLSE>(SupportsAVX);
|
||||
}
|
||||
|
||||
}
|
||||
Loaded 100 of 361 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user