mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-07 17:00:19 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8c956e6ce1 | ||
|
|
832d013c92 | ||
|
|
1c24206117 | ||
|
|
f1979c15a2 | ||
|
|
556a1dab24 | ||
|
|
caffad8562 | ||
|
|
5978143141 | ||
|
|
594c70b5e0 | ||
|
|
655e6989ca | ||
|
|
afeb228a89 | ||
|
|
d308a438ea | ||
|
|
260fc8ba52 | ||
|
|
ade0d0f241 | ||
|
|
11a5105547 | ||
|
|
10ad5db686 | ||
|
|
68c441575d | ||
|
|
334a8ef87c | ||
|
|
b7a76af72f | ||
|
|
4c92b562b8 | ||
|
|
9d08451903 | ||
|
|
c252f8bfc5 | ||
|
|
dc7ec6377b | ||
|
|
ce6f4edaaa | ||
|
|
983c35ea3b | ||
|
|
70aaa1117a | ||
|
|
59e9859087 | ||
|
|
d9453ff639 | ||
|
|
70754991d1 | ||
|
|
57ebfceb48 | ||
|
|
9e224d2bb0 | ||
|
|
e43bd04901 | ||
|
|
48762e03a6 | ||
|
|
858924309e | ||
|
|
cab02d1e65 | ||
|
|
2a64f80567 | ||
|
|
174ddea99d | ||
|
|
6fb0e3c0cf | ||
|
|
2b044bbdf4 | ||
|
|
ea76de0fd2 | ||
|
|
9c8642e0dc | ||
|
|
13f35f7b79 | ||
|
|
e2798e370e | ||
|
|
d9149548b5 | ||
|
|
4823933f79 | ||
|
|
ee04067424 | ||
|
|
e4aef26ef5 | ||
|
|
a41dc8eafa | ||
|
|
f41cd8deff | ||
|
|
2a0c3cce30 | ||
|
|
e817f5d98c | ||
|
|
8b8cda9b80 | ||
|
|
8e3893df07 | ||
|
|
51b335914c | ||
|
|
8835d57ae3 | ||
|
|
eba1b65fb0 | ||
|
|
a3a138ef7e | ||
|
|
62d9a494cd | ||
|
|
6744a06a53 | ||
|
|
140e9824b7 | ||
|
|
bad84f61fa | ||
|
|
2296126af3 | ||
|
|
0a8717d9a8 | ||
|
|
4e2220c27f | ||
|
|
c87e11cee9 | ||
|
|
7768f6965a | ||
|
|
023aaaae0c | ||
|
|
a2aa9f3fc1 | ||
|
|
e46ec9a0ce | ||
|
|
784cbdd973 | ||
|
|
6022715a9b | ||
|
|
9eb5ba5ad1 | ||
|
|
82e5977709 | ||
|
|
7c08b67dff | ||
|
|
609587f9ee | ||
|
|
5dda3a1599 | ||
|
|
73aaa4c3a6 | ||
|
|
3ba5371d36 | ||
|
|
bf581decde | ||
|
|
250504502a | ||
|
|
a0e826feaf | ||
|
|
eb17edec05 | ||
|
|
228aed98c7 | ||
|
|
6ba4aec88e | ||
|
|
b8d6b2cd4a | ||
|
|
d4e2f42f90 | ||
|
|
3055c23365 | ||
|
|
cb7feaefbb | ||
|
|
5a5a498ed6 | ||
|
|
285ed8f1e0 | ||
|
|
a68da468a5 | ||
|
|
d59aa6874e | ||
|
|
19fd89d2bf | ||
|
|
e36beb8dbe | ||
|
|
70f447b265 | ||
|
|
a0026c92a8 | ||
|
|
8fc5f66a5b | ||
|
|
affbb40cc1 | ||
|
|
b276750c8a | ||
|
|
6d5322b349 | ||
|
|
f836ff0897 | ||
|
|
0bf577654d | ||
|
|
ac63a2bac4 | ||
|
|
6b8e6c18ee | ||
|
|
93dccffc81 | ||
|
|
22810208b4 | ||
|
|
154468c172 | ||
|
|
0e45b98fb2 | ||
|
|
2d573f7bf9 | ||
|
|
01587270d9 | ||
|
|
0a63a15287 | ||
|
|
a65e7a09e7 | ||
|
|
7e5c561849 | ||
|
|
a45bc7598c | ||
|
|
9f8071ce8f | ||
|
|
0808599813 | ||
|
|
340699ca33 | ||
|
|
00a572c00c | ||
|
|
d300181ab9 | ||
|
|
5146651908 | ||
|
|
875bae41a3 | ||
|
|
95bd309db9 | ||
|
|
6f1e7e7a4c | ||
|
|
b078f51199 | ||
|
|
d03856f0b7 | ||
|
|
98210fdf47 | ||
|
|
3d1cef4afe | ||
|
|
060ab98378 | ||
|
|
1d5bd1520a | ||
|
|
9eca823e84 | ||
|
|
cd4269f4e8 | ||
|
|
2f1d44f838 | ||
|
|
95eb456065 | ||
|
|
d4b31cd4c6 | ||
|
|
c611eb228f | ||
|
|
bc120b9f97 | ||
|
|
995672921f | ||
|
|
514b6a822f | ||
|
|
0945c728e9 | ||
|
|
3a3e2776ba | ||
|
|
b72245a2d5 | ||
|
|
9c49bd3c8e | ||
|
|
c691d70919 | ||
|
|
f3a27a57f1 | ||
|
|
edce981824 | ||
|
|
9bffaeea40 | ||
|
|
88ce9b5cd2 | ||
|
|
72e8a997f6 | ||
|
|
d708cbad5e | ||
|
|
152eaff00b | ||
|
|
2079f6b3c7 | ||
|
|
2c31080fd4 | ||
|
|
ef7b77dff7 | ||
|
|
777aadb73e | ||
|
|
ccd06e2097 | ||
|
|
09a5f8c6b5 | ||
|
|
8f835678f5 | ||
|
|
8f2cd39802 | ||
|
|
059e8a9099 | ||
|
|
5fae07f6ee | ||
|
|
02005e71ca | ||
|
|
59cc5228b8 | ||
|
|
117cbde226 | ||
|
|
565d1e27d7 | ||
|
|
22ecff05c9 | ||
|
|
17480c0e2d | ||
|
|
0608d95322 | ||
|
|
47bd47950b | ||
|
|
9aab7e07f4 | ||
|
|
f2fd9d9e3f | ||
|
|
45430e9569 | ||
|
|
be3e3a351a | ||
|
|
2cee9e5d4b | ||
|
|
03cc35f341 | ||
|
|
3aead5ef65 | ||
|
|
b9abd091d5 | ||
|
|
43dc232e5a | ||
|
|
ea73c9d7ea | ||
|
|
fa6f1b1d90 | ||
|
|
e24eb7a72d | ||
|
|
e05d116b05 | ||
|
|
ee821b9cbf | ||
|
|
c29e563836 | ||
|
|
fd59fb1a7a | ||
|
|
04deeb3911 | ||
|
|
4a7c65b20d | ||
|
|
23cb0dea00 | ||
|
|
bcad7a9eea | ||
|
|
0a3a270a63 | ||
|
|
06e4a5a5b7 | ||
|
|
6ff80670b3 | ||
|
|
38eea80b8d | ||
|
|
c43af0e10f | ||
|
|
2fa8cd7e97 | ||
|
|
5d0734a7f2 | ||
|
|
3e0e922fef | ||
|
|
e2004f4999 | ||
|
|
13f37efe12 | ||
|
|
80e66a36eb | ||
|
|
4998d35ec5 | ||
|
|
f0655874e3 | ||
|
|
9394e49c95 | ||
|
|
91984003b6 | ||
|
|
dbf571fdfb | ||
|
|
b6abcc5e3c | ||
|
|
e99a23cfc6 | ||
|
|
425d9323d5 | ||
|
|
a7c0997daf | ||
|
|
6eba3f331e | ||
|
|
bcc75e3312 | ||
|
|
30672e1517 | ||
|
|
b6e46fd44d | ||
|
|
32a0e37569 | ||
|
|
5b6175b702 | ||
|
|
0285e35c87 | ||
|
|
bbfb8713a7 | ||
|
|
c1d6967cb6 | ||
|
|
20e24eae9c | ||
|
|
b8d9027680 | ||
|
|
688ef9f5af | ||
|
|
181c9074ab | ||
|
|
6f85f64bfd | ||
|
|
ba6ee61db2 | ||
|
|
e735281b2c | ||
|
|
c8b2d714a8 | ||
|
|
bba0625a57 | ||
|
|
c9a7b36210 | ||
|
|
99bf02f27f | ||
|
|
a2c02ae51e | ||
|
|
c7b59143c6 | ||
|
|
1e23e61572 | ||
|
|
f4dd1a895e | ||
|
|
3a2f9a4d46 | ||
|
|
4a848d7202 | ||
|
|
1b58ed9f57 | ||
|
|
1c86f7ed36 | ||
|
|
2733b2ee1e | ||
|
|
4ceb2dfdf2 | ||
|
|
bd380e0f15 | ||
|
|
73ec786c60 | ||
|
|
e3a2c8dc80 | ||
|
|
d3b14df840 | ||
|
|
a12ab8f98a | ||
|
|
966b9a69d8 | ||
|
|
927d3d00e2 | ||
|
|
bfb9cabeb8 | ||
|
|
22466a973c | ||
|
|
c05e1c9797 | ||
|
|
c65be9f55d | ||
|
|
edc31fe0a3 | ||
|
|
d4655fbb17 | ||
|
|
5f0dfcd715 | ||
|
|
e62cb417c2 | ||
|
|
df486c0786 | ||
|
|
9d43904792 | ||
|
|
5654f9a030 | ||
|
|
d74cf6d8d8 | ||
|
|
84905d2856 | ||
|
|
92cb9d477e | ||
|
|
7ed6007252 | ||
|
|
b6499ac724 | ||
|
|
c73991f467 | ||
|
|
0bf0fe4779 | ||
|
|
097f48f3ff | ||
|
|
75e5df545a | ||
|
|
171d5f7263 | ||
|
|
970067d19b | ||
|
|
c26ff60949 | ||
|
|
04d830ed39 | ||
|
|
b9c49027c7 | ||
|
|
9860e8b71b | ||
|
|
a1e94a9863 | ||
|
|
2945c13dcb | ||
|
|
4f68821aef | ||
|
|
13087f8425 | ||
|
|
fd0424768b | ||
|
|
5c4112f103 | ||
|
|
e8670bbab2 | ||
|
|
13b14b857b | ||
|
|
48c7ff2a23 | ||
|
|
77a032a286 | ||
|
|
f2de640395 | ||
|
|
9a642158e0 | ||
|
|
dc44caa178 | ||
|
|
081a003c6c | ||
|
|
4c5fc6e813 | ||
|
|
253cdb552d | ||
|
|
6404aba6e2 | ||
|
|
45f919683a | ||
|
|
c3a6890a40 | ||
|
|
07be5a0bae | ||
|
|
d6e4da7e77 | ||
|
|
ed5d7f6a62 | ||
|
|
18b223811a | ||
|
|
e1d21f0bff | ||
|
|
d2783f2edd | ||
|
|
3377e5a50e |
No files matched your search
@@ -0,0 +1,45 @@
|
||||
---
|
||||
name: Potential Game Bug
|
||||
about: A bug in FEX-Emu that causes a problem in a game
|
||||
title: "[Game]: [Short Problem Description]"
|
||||
labels: Game related
|
||||
assignees: ''
|
||||
|
||||
---
|
||||
|
||||
**What Game**
|
||||
The game name.
|
||||
A link to the storefront where to get the game. GOG, Steam, Itch.io, etc
|
||||
|
||||
**Describe the bug**
|
||||
A clear and concise description of what the bug is.
|
||||
|
||||
**To Reproduce**
|
||||
Steps to reproduce the behavior:
|
||||
1. Go to '...'
|
||||
2. Click on '....'
|
||||
3. Scroll down to '....'
|
||||
4. See error
|
||||
|
||||
**Expected behavior**
|
||||
A clear and concise description of what you expected to happen.
|
||||
|
||||
**Screenshots and Video**
|
||||
If applicable, add screenshots and video to help explain your problem.
|
||||
|
||||
**System information:**
|
||||
- OS: [eg: Ubuntu 21.10]
|
||||
- CPU/SoC: [eg: Snapdragon 888, Intel Core i8-12900k]
|
||||
- Video driver version: [eg: OpenGL ES 3.2 Mesa 22.0.0-devel (git-9ff086052a)]
|
||||
- RootFS used: [eg: Ubuntu 21.10 Official Rootfs]
|
||||
- FEX version: (FEXGetConfig --version) [eg: FEX-2112-155-gc691d709]
|
||||
- Thunks Enabled: [Yes/No]
|
||||
|
||||
**Additional context**
|
||||
- Is this an x86 or x86-64 game: [x86/x86-64/Both]
|
||||
- Does this reproduce on x86-64 host with FEX: [Yes/No/Untested]
|
||||
- Does this reproduce on AArch64 with Radeon/Intel/Nvidia: [Yes/No/Untested]
|
||||
- Is this a Vulkan game: [Yes/No/Unknown]
|
||||
- If Yes, What is your Vulkan driver:
|
||||
|
||||
Add any other context about the problem here.
|
||||
@@ -31,7 +31,9 @@ jobs:
|
||||
|
||||
- name : submodule checkout
|
||||
# Need to update submodules
|
||||
run: git submodule update --init --depth 1
|
||||
run: |
|
||||
git submodule sync --recursive
|
||||
git submodule update --init --depth 1
|
||||
|
||||
- name: Clean Build Environment
|
||||
run: rm -Rf ${{runner.workspace}}/build
|
||||
@@ -49,7 +51,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DENABLE_INTERPRETER=True
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
@@ -140,6 +142,17 @@ jobs:
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_StructVerifier.log || true
|
||||
|
||||
- name: APITest tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target api_tests
|
||||
|
||||
- name: APITest Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_APITests.log || true
|
||||
|
||||
- name: Truncate test results
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
|
||||
+4
-5
@@ -1,7 +1,7 @@
|
||||
[submodule "External/vixl"]
|
||||
shallow = true
|
||||
path = External/vixl
|
||||
url = https://github.com/Sonicadvance1/vixl.git
|
||||
url = https://github.com/FEX-Emu/vixl.git
|
||||
[submodule "External/cpp-optparse"]
|
||||
path = External/cpp-optparse
|
||||
url = https://github.com/Sonicadvance1/cpp-optparse
|
||||
@@ -42,7 +42,6 @@
|
||||
[submodule "External/xxhash"]
|
||||
path = External/xxhash
|
||||
url = https://github.com/FEX-Emu/xxHash.git
|
||||
[submodule "External/Vulkan-Docs"]
|
||||
shallow = true
|
||||
path = External/Vulkan-Docs
|
||||
url = https://github.com/KhronosGroup/Vulkan-Docs.git
|
||||
[submodule "External/Catch2"]
|
||||
path = External/Catch2
|
||||
url = https://github.com/catchorg/Catch2.git
|
||||
+114
-59
@@ -18,11 +18,18 @@ option(ENABLE_STATIC_PIE "Enables static-pie build" FALSE)
|
||||
option(ENABLE_JEMALLOC "Enables jemalloc allocator" TRUE)
|
||||
option(ENABLE_OFFLINE_TELEMETRY "Enables FEX offline telemetry" TRUE)
|
||||
option(ENABLE_COMPILE_TIME_TRACE "Enables time trace compile option" FALSE)
|
||||
option(ENABLE_LIBCXX "Enables LLVM libc++" FALSE)
|
||||
option(ENABLE_INTERPRETER "Enables FEX's Interpreter" FALSE)
|
||||
|
||||
set (X86_C_COMPILER "x86_64-linux-gnu-gcc" CACHE STRING "c compiler for compiling x86 guest libs")
|
||||
set (X86_CXX_COMPILER "x86_64-linux-gnu-g++" CACHE STRING "c++ compiler for compiling x86 guest libs")
|
||||
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu" CACHE PATH "global data directory")
|
||||
|
||||
# These options are meant for package management
|
||||
set (TUNE_CPU "native" CACHE STRING "Override the CPU the build is tuned for")
|
||||
set (TUNE_ARCH "generic" CACHE STRING "Override the Arch the build is tuned for")
|
||||
set (OVERRIDE_VERSION "detect" CACHE STRING "Override the FEX version in the format of <MMYY>{.<REV>}")
|
||||
|
||||
string(TOUPPER "${CMAKE_BUILD_TYPE}" CMAKE_BUILD_TYPE)
|
||||
if (CMAKE_BUILD_TYPE MATCHES "DEBUG")
|
||||
set(ENABLE_ASSERTIONS TRUE)
|
||||
@@ -33,6 +40,11 @@ if (ENABLE_ASSERTIONS)
|
||||
add_definitions(-DASSERTIONS_ENABLED=1)
|
||||
endif()
|
||||
|
||||
if (ENABLE_INTERPRETER)
|
||||
message(STATUS "Interpreter enabled")
|
||||
add_definitions(-DINTERPRETER_ENABLED=1)
|
||||
endif()
|
||||
|
||||
set(CMAKE_CXX_STANDARD 20)
|
||||
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
|
||||
set(CMAKE_RUNTIME_OUTPUT_DIRECTORY ${CMAKE_BINARY_DIR}/Bin)
|
||||
@@ -89,6 +101,12 @@ if (ENABLE_LLD)
|
||||
link_libraries(${LD_OVERRIDE})
|
||||
endif()
|
||||
|
||||
if (ENABLE_LIBCXX)
|
||||
message(WARNING "This is an unsupported configuration and should only be used for testing")
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++11 -stdlib=libc++")
|
||||
set(CMAKE_EXE_LINKER_FLAGS "${CMAKE_EXE_LINKER_FLAGS} -stdlib=libc++ -lc++abi")
|
||||
endif()
|
||||
|
||||
if (NOT ENABLE_OFFLINE_TELEMETRY)
|
||||
# Disable FEX offline telemetry entirely if asked
|
||||
add_definitions(-DFEX_DISABLE_TELEMETRY=1)
|
||||
@@ -257,17 +275,23 @@ endif()
|
||||
|
||||
find_package(PkgConfig REQUIRED)
|
||||
find_package(Python 3.0 REQUIRED COMPONENTS Interpreter)
|
||||
pkg_check_modules(XXHASH libxxhash>=0.8.0 QUIET)
|
||||
|
||||
if (NOT XXHASH_FOUND)
|
||||
message(STATUS "xxHash not found. Using Externals")
|
||||
add_subdirectory(External/xxhash/)
|
||||
include_directories(External/xxhash/)
|
||||
endif()
|
||||
add_subdirectory(External/xxhash/)
|
||||
include_directories(External/xxhash/)
|
||||
|
||||
add_definitions(-Wno-trigraphs)
|
||||
add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
|
||||
|
||||
if (BUILD_TESTS)
|
||||
option(CATCH_BUILD_STATIC_LIBRARY "" ON)
|
||||
set(CATCH_BUILD_STATIC_LIBRARY ON)
|
||||
add_subdirectory(External/Catch2/)
|
||||
|
||||
# Pull in catch_discover_tests definition
|
||||
list(APPEND CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/External/Catch2/contrib/")
|
||||
include(Catch)
|
||||
endif()
|
||||
|
||||
add_subdirectory(External/cpp-optparse/)
|
||||
include_directories(External/cpp-optparse/)
|
||||
|
||||
@@ -314,33 +338,51 @@ if(ENABLE_WERROR OR ENABLE_STRICT_WERROR)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if(_M_ARM_64)
|
||||
if (CMAKE_CXX_COMPILER_VERSION VERSION_GREATER_EQUAL 999999.0)
|
||||
# Clang 12.0 fixed the -mcpu=native bug with mixed big.little implementers
|
||||
# Clang can not currently check for native Apple M1 type in hypervisor. Currently disabled
|
||||
check_cxx_compiler_flag("-mcpu=native" COMPILER_SUPPORTS_CPU_TYPE)
|
||||
if(COMPILER_SUPPORTS_CPU_TYPE)
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcpu=native")
|
||||
if (NOT TUNE_ARCH STREQUAL "generic")
|
||||
check_cxx_compiler_flag("-march=${TUNE_ARCH}" COMPILER_SUPPORTS_ARCH_TYPE)
|
||||
if(COMPILER_SUPPORTS_ARCH_TYPE)
|
||||
add_compile_options("-march=${TUNE_ARCH}")
|
||||
else()
|
||||
message(FATAL_ERROR "Trying to compile arch type '${TUNE_ARCH}' but the compiler doesn't support this")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (TUNE_CPU STREQUAL "native")
|
||||
if(_M_ARM_64)
|
||||
if (CMAKE_CXX_COMPILER_VERSION VERSION_GREATER_EQUAL 999999.0)
|
||||
# Clang 12.0 fixed the -mcpu=native bug with mixed big.little implementers
|
||||
# Clang can not currently check for native Apple M1 type in hypervisor. Currently disabled
|
||||
check_cxx_compiler_flag("-mcpu=native" COMPILER_SUPPORTS_CPU_TYPE)
|
||||
if(COMPILER_SUPPORTS_CPU_TYPE)
|
||||
add_compile_options("-mcpu=native")
|
||||
endif()
|
||||
else()
|
||||
# Due to an oversight in llvm, it declares any reasonably new Kryo CPU to only be ARMv8.0
|
||||
# Manually detect newer CPU revisions until clang and llvm fixes their bug
|
||||
# This script will either provide a supported CPU or 'native'
|
||||
# Additionally -march doesn't work under AArch64+Clang, so you have to use -mcpu or -mtune
|
||||
execute_process(COMMAND python3 "${PROJECT_SOURCE_DIR}/Scripts/aarch64_fit_native.py" "/proc/cpuinfo" "${CMAKE_CXX_COMPILER_VERSION}"
|
||||
OUTPUT_VARIABLE AARCH64_CPU)
|
||||
|
||||
string(STRIP ${AARCH64_CPU} AARCH64_CPU)
|
||||
|
||||
check_cxx_compiler_flag("-mcpu=${AARCH64_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
|
||||
if(COMPILER_SUPPORTS_CPU_TYPE)
|
||||
add_compile_options("-mcpu=${AARCH64_CPU}")
|
||||
endif()
|
||||
endif()
|
||||
else()
|
||||
# Due to an oversight in llvm, it declares any reasonably new Kryo CPU to only be ARMv8.0
|
||||
# Manually detect newer CPU revisions until clang and llvm fixes their bug
|
||||
# This script will either provide a supported CPU or 'native'
|
||||
# Additionally -march doesn't work under AArch64+Clang, so you have to use -mcpu or -mtune
|
||||
execute_process(COMMAND python3 "${PROJECT_SOURCE_DIR}/Scripts/aarch64_fit_native.py" "/proc/cpuinfo" "${CMAKE_CXX_COMPILER_VERSION}"
|
||||
OUTPUT_VARIABLE AARCH64_CPU)
|
||||
|
||||
string(STRIP ${AARCH64_CPU} AARCH64_CPU)
|
||||
|
||||
check_cxx_compiler_flag("-mcpu=${AARCH64_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
|
||||
if(COMPILER_SUPPORTS_CPU_TYPE)
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcpu=${AARCH64_CPU}")
|
||||
check_cxx_compiler_flag("-march=native" COMPILER_SUPPORTS_MARCH_NATIVE)
|
||||
if(COMPILER_SUPPORTS_MARCH_NATIVE)
|
||||
add_compile_options("-march=native")
|
||||
endif()
|
||||
endif()
|
||||
else()
|
||||
check_cxx_compiler_flag("-march=native" COMPILER_SUPPORTS_MARCH_NATIVE)
|
||||
if(COMPILER_SUPPORTS_MARCH_NATIVE)
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -march=native")
|
||||
check_cxx_compiler_flag("-mcpu=${TUNE_CPU}" COMPILER_SUPPORTS_CPU_TYPE)
|
||||
if(COMPILER_SUPPORTS_CPU_TYPE)
|
||||
add_compile_options("-mcpu=${TUNE_CPU}")
|
||||
else()
|
||||
message(FATAL_ERROR "Trying to compile cpu type '${TUNE_CPU}' but the compiler doesn't support this")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
@@ -414,6 +456,10 @@ if (BUILD_TESTS)
|
||||
enable_testing()
|
||||
message(STATUS "Unit tests are enabled")
|
||||
endif()
|
||||
|
||||
add_subdirectory(FEXHeaderUtils/)
|
||||
include_directories(FEXHeaderUtils/)
|
||||
|
||||
add_subdirectory(External/FEXCore)
|
||||
|
||||
# Binfmt_misc files must be installed prior to Source/ installs
|
||||
@@ -432,6 +478,8 @@ if (BUILD_TESTS)
|
||||
endif()
|
||||
|
||||
if (BUILD_THUNKS)
|
||||
add_subdirectory(ThunkLibs/Generator)
|
||||
|
||||
include(ExternalProject)
|
||||
|
||||
ExternalProject_Add(host-libs
|
||||
@@ -441,9 +489,10 @@ if (BUILD_THUNKS)
|
||||
CMAKE_ARGS
|
||||
"-DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}"
|
||||
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
|
||||
"-DVULKAN_XML=${CMAKE_SOURCE_DIR}/External/Vulkan-Docs/xml/vk.xml"
|
||||
"-DGENERATOR_EXE=$<TARGET_FILE:thunkgen>"
|
||||
INSTALL_COMMAND ""
|
||||
BUILD_ALWAYS ON
|
||||
DEPENDS thunkgen
|
||||
)
|
||||
|
||||
install(
|
||||
@@ -465,9 +514,10 @@ if (BUILD_THUNKS)
|
||||
"-DX86_CXX_COMPILER:STRING=${X86_CXX_COMPILER}"
|
||||
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
|
||||
"-DSTRUCT_VERIFIER=${CMAKE_SOURCE_DIR}/Scripts/StructPackVerifier.py"
|
||||
"-DVULKAN_XML=${CMAKE_SOURCE_DIR}/External/Vulkan-Docs/xml/vk.xml"
|
||||
"-DGENERATOR_EXE=$<TARGET_FILE:thunkgen>"
|
||||
INSTALL_COMMAND ""
|
||||
BUILD_ALWAYS ON
|
||||
DEPENDS thunkgen
|
||||
)
|
||||
|
||||
install(
|
||||
@@ -484,43 +534,48 @@ set(FEX_VERSION_MAJOR "0")
|
||||
set(FEX_VERSION_MINOR "0")
|
||||
set(FEX_VERSION_PATCH "0")
|
||||
|
||||
find_package(Git)
|
||||
if (GIT_FOUND)
|
||||
execute_process(
|
||||
COMMAND ${GIT_EXECUTABLE} describe --abbrev=0
|
||||
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
|
||||
RESULT_VARIABLE GIT_ERROR
|
||||
ERROR_QUIET
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
if (OVERRIDE_VERSION STREQUAL "detect")
|
||||
find_package(Git)
|
||||
if (GIT_FOUND)
|
||||
execute_process(
|
||||
COMMAND ${GIT_EXECUTABLE} describe --abbrev=0
|
||||
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
|
||||
RESULT_VARIABLE GIT_ERROR
|
||||
ERROR_QUIET
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
|
||||
if (NOT ${GIT_ERROR} EQUAL 0)
|
||||
# Likely built in a way that doesn't have tags
|
||||
# Setup a version tag that is unknown
|
||||
set(GIT_DESCRIBE_STRING "FEX-0000")
|
||||
if (NOT ${GIT_ERROR} EQUAL 0)
|
||||
# Likely built in a way that doesn't have tags
|
||||
# Setup a version tag that is unknown
|
||||
set(GIT_DESCRIBE_STRING "FEX-0000")
|
||||
endif()
|
||||
endif()
|
||||
else()
|
||||
set(GIT_DESCRIBE_STRING "FEX-${OVERRIDE_VERSION}")
|
||||
endif()
|
||||
|
||||
# Change something like `FEX-2106.1-76-<hash>` in to a list
|
||||
string(REPLACE "-" ";" DESCRIBE_LIST ${GIT_DESCRIBE_STRING})
|
||||
# Parse the version here
|
||||
# Change something like `FEX-2106.1-76-<hash>` in to a list
|
||||
string(REPLACE "-" ";" DESCRIBE_LIST ${GIT_DESCRIBE_STRING})
|
||||
|
||||
# Extract the `2106.1` element
|
||||
list(GET DESCRIBE_LIST 1 DESCRIBE_LIST)
|
||||
# Extract the `2106.1` element
|
||||
list(GET DESCRIBE_LIST 1 DESCRIBE_LIST)
|
||||
|
||||
# Change `2106.1` in to a list
|
||||
string(REPLACE "." ";" DESCRIBE_LIST ${DESCRIBE_LIST})
|
||||
# Change `2106.1` in to a list
|
||||
string(REPLACE "." ";" DESCRIBE_LIST ${DESCRIBE_LIST})
|
||||
|
||||
# Calculate list size
|
||||
list(LENGTH DESCRIBE_LIST LIST_SIZE)
|
||||
# Calculate list size
|
||||
list(LENGTH DESCRIBE_LIST LIST_SIZE)
|
||||
|
||||
# Pull out the major version
|
||||
list(GET DESCRIBE_LIST 0 FEX_VERSION_MAJOR)
|
||||
# Pull out the major version
|
||||
list(GET DESCRIBE_LIST 0 FEX_VERSION_MAJOR)
|
||||
|
||||
# Minor version only exists if there is a .1 at the end
|
||||
# eg: 2106 versus 2106.1
|
||||
if (LIST_SIZE GREATER 1)
|
||||
list(GET DESCRIBE_LIST 1 FEX_VERSION_MINOR)
|
||||
endif()
|
||||
# Minor version only exists if there is a .1 at the end
|
||||
# eg: 2106 versus 2106.1
|
||||
if (LIST_SIZE GREATER 1)
|
||||
list(GET DESCRIBE_LIST 1 FEX_VERSION_MINOR)
|
||||
endif()
|
||||
|
||||
# Package creation
|
||||
|
||||
+1
Submodule External/Catch2 added at c4e3767e26.
Vendored
+23
-18
@@ -37,27 +37,32 @@ endif()
|
||||
set(CMAKE_CXX_STANDARD 20)
|
||||
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
|
||||
|
||||
# Find our git hash
|
||||
find_package(Git)
|
||||
|
||||
set(GIT_SHORT_HASH "Unknown")
|
||||
set(GIT_DESCRIBE_STRING "FEX-Unknown")
|
||||
|
||||
if (GIT_FOUND)
|
||||
execute_process(
|
||||
COMMAND ${GIT_EXECUTABLE} rev-parse --short HEAD
|
||||
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE GIT_SHORT_HASH
|
||||
ERROR_QUIET
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
execute_process(
|
||||
COMMAND ${GIT_EXECUTABLE} describe
|
||||
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
|
||||
ERROR_QUIET
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
if (OVERRIDE_VERSION STREQUAL "detect")
|
||||
# Find our git hash
|
||||
find_package(Git)
|
||||
|
||||
if (GIT_FOUND)
|
||||
execute_process(
|
||||
COMMAND ${GIT_EXECUTABLE} rev-parse --short HEAD
|
||||
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE GIT_SHORT_HASH
|
||||
ERROR_QUIET
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
execute_process(
|
||||
COMMAND ${GIT_EXECUTABLE} describe
|
||||
WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}"
|
||||
OUTPUT_VARIABLE GIT_DESCRIBE_STRING
|
||||
ERROR_QUIET
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
endif()
|
||||
else()
|
||||
set(GIT_SHORT_HASH "${OVERRIDE_VERSION}")
|
||||
set(GIT_DESCRIBE_STRING "FEX-${OVERRIDE_VERSION}")
|
||||
endif()
|
||||
|
||||
configure_file(
|
||||
|
||||
+53
-41
@@ -1,7 +1,14 @@
|
||||
set (MAN_DIR ${CMAKE_INSTALL_PREFIX}/share/man CACHE PATH "MAN_DIR")
|
||||
|
||||
set (SRCS
|
||||
set (FEXCORE_BASE_SRCS
|
||||
Common/Paths.cpp
|
||||
Interface/Config/Config.cpp
|
||||
Utils/FileLoading.cpp
|
||||
Utils/ForcedAssert.cpp
|
||||
Utils/LogManager.cpp
|
||||
)
|
||||
|
||||
set (SRCS
|
||||
Common/JitSymbols.cpp
|
||||
Common/SoftFloat-3e/extF80_add.c
|
||||
Common/SoftFloat-3e/extF80_div.c
|
||||
@@ -70,7 +77,6 @@ set (SRCS
|
||||
Common/SoftFloat-3e/f32_to_extF80.c
|
||||
Common/SoftFloat-3e/s_normSubnormalF32Sig.c
|
||||
Common/SoftFloat-3e/s_f32UIToCommonNaN.c
|
||||
Interface/Config/Config.cpp
|
||||
Interface/Context/Context.cpp
|
||||
Interface/Core/LookupCache.cpp
|
||||
Interface/Core/BlockSamplingData.cpp
|
||||
@@ -94,19 +100,7 @@ set (SRCS
|
||||
Interface/Core/Dispatcher/Dispatcher.cpp
|
||||
Interface/Core/Dispatcher/X86Dispatcher.cpp
|
||||
Interface/Core/Dispatcher/Arm64Dispatcher.cpp
|
||||
Interface/Core/Interpreter/InterpreterCore.cpp
|
||||
Interface/Core/Interpreter/InterpreterOps.cpp
|
||||
Interface/Core/Interpreter/ALUOps.cpp
|
||||
Interface/Core/Interpreter/AtomicOps.cpp
|
||||
Interface/Core/Interpreter/BranchOps.cpp
|
||||
Interface/Core/Interpreter/ConversionOps.cpp
|
||||
Interface/Core/Interpreter/EncryptionOps.cpp
|
||||
Interface/Core/Interpreter/F80Ops.cpp
|
||||
Interface/Core/Interpreter/FlagOps.cpp
|
||||
Interface/Core/Interpreter/MemoryOps.cpp
|
||||
Interface/Core/Interpreter/MiscOps.cpp
|
||||
Interface/Core/Interpreter/MoveOps.cpp
|
||||
Interface/Core/Interpreter/VectorOps.cpp
|
||||
Interface/Core/Interpreter/InterpreterFallbacks.cpp
|
||||
Interface/Core/X86Tables/BaseTables.cpp
|
||||
Interface/Core/X86Tables/DDDTables.cpp
|
||||
Interface/Core/X86Tables/EVEXTables.cpp
|
||||
@@ -120,6 +114,7 @@ set (SRCS
|
||||
Interface/Core/X86Tables/X87Tables.cpp
|
||||
Interface/Core/X86Tables/XOPTables.cpp
|
||||
Interface/HLE/Thunks/Thunks.cpp
|
||||
Interface/IR/AOTIR.cpp
|
||||
Interface/IR/IRDumper.cpp
|
||||
Interface/IR/IRParser.cpp
|
||||
Interface/IR/IREmitter.cpp
|
||||
@@ -140,14 +135,28 @@ set (SRCS
|
||||
Interface/IR/Passes/SyscallOptimization.cpp
|
||||
Utils/Allocator.cpp
|
||||
Utils/Allocator/64BitAllocator.cpp
|
||||
Utils/FileLoading.cpp
|
||||
Utils/ForcedAssert.cpp
|
||||
Utils/LogManager.cpp
|
||||
Utils/NetStream.cpp
|
||||
Utils/Telemetry.cpp
|
||||
Utils/Threads.cpp
|
||||
)
|
||||
|
||||
if (ENABLE_INTERPRETER)
|
||||
list(APPEND SRCS
|
||||
Interface/Core/Interpreter/InterpreterCore.cpp
|
||||
Interface/Core/Interpreter/InterpreterOps.cpp
|
||||
Interface/Core/Interpreter/ALUOps.cpp
|
||||
Interface/Core/Interpreter/AtomicOps.cpp
|
||||
Interface/Core/Interpreter/BranchOps.cpp
|
||||
Interface/Core/Interpreter/ConversionOps.cpp
|
||||
Interface/Core/Interpreter/EncryptionOps.cpp
|
||||
Interface/Core/Interpreter/F80Ops.cpp
|
||||
Interface/Core/Interpreter/FlagOps.cpp
|
||||
Interface/Core/Interpreter/MemoryOps.cpp
|
||||
Interface/Core/Interpreter/MiscOps.cpp
|
||||
Interface/Core/Interpreter/MoveOps.cpp
|
||||
Interface/Core/Interpreter/VectorOps.cpp)
|
||||
endif()
|
||||
|
||||
if(_M_ARM_64)
|
||||
list(APPEND SRCS
|
||||
Interface/Core/ArchHelpers/Arm64.cpp)
|
||||
@@ -195,7 +204,7 @@ if (ENABLE_JIT_ARM64)
|
||||
Interface/Core/JIT/Arm64/VectorOps.cpp)
|
||||
endif()
|
||||
|
||||
set (LIBS vixl dl fmt::fmt xxhash tiny-json)
|
||||
set (LIBS vixl dl xxhash tiny-json)
|
||||
if (ENABLE_JEMALLOC)
|
||||
list (APPEND LIBS FEX_jemalloc)
|
||||
endif()
|
||||
@@ -296,17 +305,10 @@ install(FILES ${OUTPUT_MAN_NAME_COMPRESS} DESTINATION ${MAN_DIR}/man1)
|
||||
check_cxx_compiler_flag(-fdiagnostics-color=always GCC_COLOR)
|
||||
check_cxx_compiler_flag(-fcolor-diagnostics CLANG_COLOR)
|
||||
|
||||
function(AddObject Name Type)
|
||||
add_library(${Name} ${Type} ${SRCS})
|
||||
add_dependencies(${Name} IR_INC)
|
||||
add_dependencies(${Name} CONFIG_INC)
|
||||
|
||||
target_link_libraries(${Name} ${LIBS})
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
function(AddDefaultOptionsToTarget Name)
|
||||
set_target_properties(${Name} PROPERTIES C_VISIBILITY_PRESET hidden)
|
||||
set_target_properties(${Name} PROPERTIES CXX_VISIBILITY_PRESET hidden)
|
||||
set_target_properties(${Name} PROPERTIES VISIBILITY_INLINES_HIDDEN TRUE)
|
||||
|
||||
target_include_directories(${Name} PUBLIC "${CMAKE_CURRENT_BINARY_DIR}")
|
||||
|
||||
target_include_directories(${Name} PRIVATE IncludePrivate/)
|
||||
@@ -316,6 +318,7 @@ function(AddObject Name Type)
|
||||
target_include_directories(${Name} PUBLIC "${CMAKE_BINARY_DIR}/include/")
|
||||
|
||||
target_compile_definitions(${Name} PRIVATE ${DEFINES})
|
||||
add_dependencies(${Name} CONFIG_INC)
|
||||
|
||||
target_compile_options(${Name}
|
||||
PRIVATE
|
||||
@@ -338,20 +341,6 @@ function(AddObject Name Type)
|
||||
PRIVATE
|
||||
"-fcolor-diagnostics")
|
||||
endif()
|
||||
endfunction()
|
||||
|
||||
function(AddLibrary Name Type)
|
||||
add_library(${Name} ${Type} $<TARGET_OBJECTS:${PROJECT_NAME}_object>)
|
||||
target_link_libraries(${Name} ${LIBS})
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
set_target_properties(${Name} PROPERTIES C_VISIBILITY_PRESET hidden)
|
||||
set_target_properties(${Name} PROPERTIES CXX_VISIBILITY_PRESET hidden)
|
||||
set_target_properties(${Name} PROPERTIES VISIBILITY_INLINES_HIDDEN TRUE)
|
||||
set_target_properties(${Name} PROPERTIES VERSION ${FEXCore_VERSION} SOVERSION ${FEXCore_VERSION})
|
||||
|
||||
target_include_directories(${Name} PUBLIC "${CMAKE_CURRENT_BINARY_DIR}")
|
||||
target_include_directories(${Name} PUBLIC "${PROJECT_SOURCE_DIR}/include/")
|
||||
target_include_directories(${Name} PUBLIC "${CMAKE_BINARY_DIR}/include/")
|
||||
|
||||
if (CMAKE_BUILD_TYPE MATCHES "RELEASE")
|
||||
target_link_options(${Name}
|
||||
@@ -363,6 +352,29 @@ function(AddLibrary Name Type)
|
||||
endif()
|
||||
endfunction()
|
||||
|
||||
# Build FEXCore_Config static library
|
||||
add_library(FEXCore_Base STATIC ${FEXCORE_BASE_SRCS})
|
||||
target_link_libraries(FEXCore_Base fmt::fmt tiny-json)
|
||||
AddDefaultOptionsToTarget(FEXCore_Base)
|
||||
|
||||
function(AddObject Name Type)
|
||||
add_library(${Name} ${Type} ${SRCS})
|
||||
add_dependencies(${Name} IR_INC)
|
||||
|
||||
target_link_libraries(${Name} FEXCore_Base ${LIBS})
|
||||
AddDefaultOptionsToTarget(${Name})
|
||||
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
endfunction()
|
||||
|
||||
function(AddLibrary Name Type)
|
||||
add_library(${Name} ${Type} $<TARGET_OBJECTS:${PROJECT_NAME}_object>)
|
||||
target_link_libraries(${Name} FEXCore_Base ${LIBS})
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
|
||||
AddDefaultOptionsToTarget(${Name})
|
||||
endfunction()
|
||||
|
||||
AddObject(${PROJECT_NAME}_object OBJECT)
|
||||
AddLibrary(${PROJECT_NAME} STATIC)
|
||||
AddLibrary(${PROJECT_NAME}_shared SHARED)
|
||||
|
||||
+4
-4
@@ -23,7 +23,7 @@ namespace FEXCore {
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
fmt::print(fp.get(), "{:x} {:x} JIT_0x{:x}_{:x}\n", HostAddr, CodeSize, GuestAddr, HostAddr);
|
||||
fmt::print(fp.get(), "{} {:x} JIT_0x{:x}_{}\n", HostAddr, CodeSize, GuestAddr, HostAddr);
|
||||
}
|
||||
|
||||
void JITSymbols::Register(const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
|
||||
@@ -31,7 +31,7 @@ namespace FEXCore {
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
fmt::print(fp.get(), "{:x} {:x} {}_{:x}\n", HostAddr, CodeSize, Name, HostAddr);
|
||||
fmt::print(fp.get(), "{} {:x} {}_{}\n", HostAddr, CodeSize, Name, HostAddr);
|
||||
}
|
||||
|
||||
void JITSymbols::RegisterNamedRegion(const void *HostAddr, uint32_t CodeSize, std::string_view Name) {
|
||||
@@ -39,7 +39,7 @@ namespace FEXCore {
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
fmt::print(fp.get(), "{:x} {:x} {}\n", HostAddr, CodeSize, Name);
|
||||
fmt::print(fp.get(), "{} {:x} {}\n", HostAddr, CodeSize, Name);
|
||||
}
|
||||
|
||||
void JITSymbols::RegisterJITSpace(const void *HostAddr, uint32_t CodeSize) {
|
||||
@@ -47,7 +47,7 @@ namespace FEXCore {
|
||||
|
||||
// Linux perf format is very straightforward
|
||||
// `<HostPtr> <Size> <Name>\n`
|
||||
fmt::print(fp.get(), "{:x} {:x} FEXJIT\n", HostAddr, CodeSize);
|
||||
fmt::print(fp.get(), "{} {:x} FEXJIT\n", HostAddr, CodeSize);
|
||||
}
|
||||
|
||||
} // namespace FEXCore
|
||||
+264
-8
@@ -57,35 +57,131 @@ struct X80SoftFloat {
|
||||
|
||||
// Ops
|
||||
static X80SoftFloat FADD(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
faddp;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_add(lhs, rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FSUB(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
fsubp;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_sub(lhs, rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FMUL(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
fmulp;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_mul(lhs, rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FDIV(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
fdivp;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_div(lhs, rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FREM(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
X80SoftFloat Rem = extF80_rem(lhs, rhs);
|
||||
if (SignBit(Rem)) {
|
||||
Rem = extF80_add(Rem, rhs);
|
||||
}
|
||||
else {
|
||||
Rem.Sign = SignBit(lhs);
|
||||
}
|
||||
#if defined(DEBUG_X86_FLOAT)
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
fprem;
|
||||
fstpt %[result];
|
||||
ffreep %%st(0);
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Rem;
|
||||
return Result;
|
||||
#else
|
||||
return extF80_rem(lhs, rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FREM1(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
#if defined(DEBUG_X86_FLOAT)
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
fprem1;
|
||||
fstpt %[result];
|
||||
ffreep %%st(0);
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_rem(lhs, rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FRNDINT(X80SoftFloat const &lhs) {
|
||||
@@ -93,15 +189,47 @@ struct X80SoftFloat {
|
||||
}
|
||||
|
||||
static X80SoftFloat FXTRACT_SIG(X80SoftFloat const &lhs) {
|
||||
#if defined(DEBUG_X86_FLOAT)
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fxtract;
|
||||
fstpt %[result];
|
||||
ffreep %%st(0);
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
X80SoftFloat Tmp = lhs;
|
||||
Tmp.Exponent = 0x3FFF;
|
||||
Tmp.Sign = lhs.Sign;
|
||||
return Tmp;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FXTRACT_EXP(X80SoftFloat const &lhs) {
|
||||
#if defined(DEBUG_X86_FLOAT)
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fxtract;
|
||||
ffreep %%st(0);
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
int32_t TrueExp = lhs.Exponent - ExponentBias;
|
||||
return i32_to_extF80(TrueExp);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void FCMP(X80SoftFloat const &lhs, X80SoftFloat const &rhs, bool *eq, bool *lt, bool *nan) {
|
||||
@@ -112,61 +240,189 @@ struct X80SoftFloat {
|
||||
|
||||
static X80SoftFloat FSCALE(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FSCALE which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st1
|
||||
fldt %[lhs]; # st0
|
||||
fscale; # st0 = st0 * 2^(rdint(st1))
|
||||
fstpt %[result];
|
||||
ffreep %%st(0);
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
X80SoftFloat Int = FRNDINT(rhs);
|
||||
BIGFLOAT Src2_d = Int;
|
||||
Src2_d = exp2l(Src2_d);
|
||||
X80SoftFloat Src2_X80 = Src2_d;
|
||||
X80SoftFloat Result = extF80_mul(lhs, Src2_X80);
|
||||
return Result;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat F2XM1(X80SoftFloat const &lhs) {
|
||||
WARN_ONCE_FMT("x87: Application used F2XM1 which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
f2xm1; # st0 = 2^st(0) - 1
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
BIGFLOAT Src1_d = lhs;
|
||||
BIGFLOAT Result = exp2l(Src1_d);
|
||||
Result -= 1.0;
|
||||
return Result;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FYL2X(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FYL2X which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[rhs]; # st(1)
|
||||
fldt %[lhs]; # st(0)
|
||||
fyl2x; # st(1) * log2l(st(0))
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
BIGFLOAT Src1_d = lhs;
|
||||
BIGFLOAT Src2_d = rhs;
|
||||
BIGFLOAT Tmp = Src2_d * log2l(Src1_d);
|
||||
return Tmp;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FATAN(X80SoftFloat const &lhs, X80SoftFloat const &rhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FATAN which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs];
|
||||
fldt %[rhs];
|
||||
fpatan;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
, [rhs] "m" (rhs)
|
||||
: "st", "st(1)");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
BIGFLOAT Src1_d = lhs;
|
||||
BIGFLOAT Src2_d = rhs;
|
||||
BIGFLOAT Tmp = atan2l(Src1_d, Src2_d);
|
||||
return Tmp;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FTAN(X80SoftFloat const &lhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FTAN which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fptan;
|
||||
ffreep %%st(0);
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
BIGFLOAT Src_d = lhs;
|
||||
Src_d = tanl(Src_d);
|
||||
return Src_d;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FSIN(X80SoftFloat const &lhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FSIN which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fsin;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
BIGFLOAT Src_d = lhs;
|
||||
Src_d = sinl(Src_d);
|
||||
return Src_d;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FCOS(X80SoftFloat const &lhs) {
|
||||
WARN_ONCE_FMT("x87: Application used FCOS which may have accuracy problems");
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fcos;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
BIGFLOAT Src_d = lhs;
|
||||
Src_d = cosl(Src_d);
|
||||
return Src_d;
|
||||
#endif
|
||||
}
|
||||
|
||||
static X80SoftFloat FSQRT(X80SoftFloat const &lhs) {
|
||||
#ifdef DEBUG_X86_FLOAT
|
||||
BIGFLOAT Result;
|
||||
asm (R"(
|
||||
fninit;
|
||||
fldt %[lhs]; # st0
|
||||
fsqrt;
|
||||
fstpt %[result];
|
||||
)"
|
||||
: [result] "=m" (Result)
|
||||
: [lhs] "m" (lhs)
|
||||
: "st");
|
||||
|
||||
return Result;
|
||||
#else
|
||||
return extF80_sqrt(lhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
operator float() const {
|
||||
|
||||
+22
-1
@@ -416,7 +416,9 @@ namespace JSON {
|
||||
Meta->Load();
|
||||
|
||||
// Do configuration option fix ups after everything is reloaded
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_THREADS)) {
|
||||
{
|
||||
// Always fix up the number of threads and create the configuration
|
||||
// Otherwise the application could receive zero as the number of threads
|
||||
FEX_CONFIG_OPT(Cores, THREADS);
|
||||
if (Cores == 0) {
|
||||
// When the number of emulated CPU cores is zero then auto detect
|
||||
@@ -424,6 +426,25 @@ namespace JSON {
|
||||
}
|
||||
}
|
||||
|
||||
if (FEXCore::Config::Exists(FEXCore::Config::CONFIG_CORE)) {
|
||||
// Sanitize Core option
|
||||
FEX_CONFIG_OPT(Core, CORE);
|
||||
#if (_M_X86_64)
|
||||
constexpr uint32_t MaxCoreNumber = 2;
|
||||
#else
|
||||
constexpr uint32_t MaxCoreNumber = 1;
|
||||
#endif
|
||||
#ifdef INTERPRETER_ENABLED
|
||||
constexpr uint32_t MinCoreNumber = 0;
|
||||
#else
|
||||
constexpr uint32_t MinCoreNumber = 1;
|
||||
#endif
|
||||
if (Core > MaxCoreNumber || Core < MinCoreNumber) {
|
||||
// Sanitize the core option by setting the core to the JIT if invalid
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_CORE, std::to_string(FEXCore::Config::CONFIG_IRJIT));
|
||||
}
|
||||
}
|
||||
|
||||
std::string ContainerPrefix { FindContainerPrefix() };
|
||||
auto ExpandPathIfExists = [&ContainerPrefix](FEXCore::Config::ConfigOption Config, std::string PathName) {
|
||||
auto NewPath = ExpandPath(ContainerPrefix, PathName);
|
||||
|
||||
+1
-1
@@ -32,7 +32,7 @@
|
||||
},
|
||||
"Threads": {
|
||||
"Type": "uint32",
|
||||
"Default": "1",
|
||||
"Default": "0",
|
||||
"ShortArg": "T",
|
||||
"Desc": [
|
||||
"Number of physical hardware threads to tell the process we have.",
|
||||
|
||||
+20
-12
@@ -51,7 +51,7 @@ namespace FEXCore::Context {
|
||||
CTX->CustomExitHandler = std::move(handler);
|
||||
}
|
||||
|
||||
ExitHandler GetExitHandler(FEXCore::Context::Context *CTX) {
|
||||
ExitHandler GetExitHandler(const FEXCore::Context::Context *CTX) {
|
||||
return CTX->CustomExitHandler;
|
||||
}
|
||||
|
||||
@@ -71,23 +71,23 @@ namespace FEXCore::Context {
|
||||
return CTX->RunUntilExit();
|
||||
}
|
||||
|
||||
int GetProgramStatus(FEXCore::Context::Context *CTX) {
|
||||
int GetProgramStatus(const FEXCore::Context::Context *CTX) {
|
||||
return CTX->GetProgramStatus();
|
||||
}
|
||||
|
||||
FEXCore::Context::ExitReason GetExitReason(FEXCore::Context::Context *CTX) {
|
||||
FEXCore::Context::ExitReason GetExitReason(const FEXCore::Context::Context *CTX) {
|
||||
return CTX->ParentThread->ExitReason;
|
||||
}
|
||||
|
||||
bool IsDone(FEXCore::Context::Context *CTX) {
|
||||
bool IsDone(const FEXCore::Context::Context *CTX) {
|
||||
return CTX->IsPaused();
|
||||
}
|
||||
|
||||
void GetCPUState(FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *State) {
|
||||
void GetCPUState(const FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *State) {
|
||||
memcpy(State, CTX->ParentThread->CurrentFrame, sizeof(FEXCore::Core::CPUState));
|
||||
}
|
||||
|
||||
void SetCPUState(FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *State) {
|
||||
void SetCPUState(FEXCore::Context::Context *CTX, const FEXCore::Core::CPUState *State) {
|
||||
memcpy(CTX->ParentThread->CurrentFrame, State, sizeof(FEXCore::Core::CPUState));
|
||||
}
|
||||
|
||||
@@ -115,11 +115,11 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
void RegisterHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
CTX->RegisterHostSignalHandler(Signal, Func, Required);
|
||||
CTX->RegisterHostSignalHandler(Signal, std::move(Func), Required);
|
||||
}
|
||||
|
||||
void RegisterFrontendHostSignalHandler(FEXCore::Context::Context *CTX, int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
CTX->RegisterFrontendHostSignalHandler(Signal, Func, Required);
|
||||
CTX->RegisterFrontendHostSignalHandler(Signal, std::move(Func), Required);
|
||||
}
|
||||
|
||||
FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID) {
|
||||
@@ -162,12 +162,20 @@ namespace FEXCore::Context {
|
||||
return CTX->CPUID.RunFunction(Function, Leaf);
|
||||
}
|
||||
|
||||
void SetAOTIRLoader(FEXCore::Context::Context *CTX, std::function<int(const std::string&)> CacheReader) {
|
||||
CTX->AOTIRLoader = CacheReader;
|
||||
FEX_DEFAULT_VISIBILITY FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(FEXCore::Context::Context *CTX, uint32_t Function, uint32_t Leaf, uint32_t CPU) {
|
||||
return CTX->CPUID.RunFunctionName(Function, Leaf, CPU);
|
||||
}
|
||||
|
||||
void SetAOTIRWriter(FEXCore::Context::Context *CTX, std::function<std::unique_ptr<std::ostream>(const std::string&)> CacheWriter) {
|
||||
CTX->AOTIRWriter = CacheWriter;
|
||||
void SetAOTIRLoader(FEXCore::Context::Context *CTX, std::function<int(const std::string&)> CacheReader) {
|
||||
CTX->SetAOTIRLoader(CacheReader);
|
||||
}
|
||||
|
||||
void SetAOTIRWriter(FEXCore::Context::Context *CTX, std::function<std::unique_ptr<std::ofstream>(const std::string&)> CacheWriter) {
|
||||
CTX->SetAOTIRWriter(CacheWriter);
|
||||
}
|
||||
|
||||
void SetAOTIRRenamer(FEXCore::Context::Context *CTX, std::function<void(const std::string&)> CacheRenamer) {
|
||||
CTX->SetAOTIRRenamer(CacheRenamer);
|
||||
}
|
||||
|
||||
void FinalizeAOTIRCache(FEXCore::Context::Context *CTX) {
|
||||
|
||||
+22
-68
@@ -4,6 +4,7 @@
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/HostFeatures.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "Interface/IR/AOTIR.h"
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
@@ -57,38 +58,6 @@ namespace FEXCore::Context {
|
||||
MODE_SINGLESTEP = 1,
|
||||
};
|
||||
|
||||
struct AOTIRInlineEntry {
|
||||
uint64_t GuestHash;
|
||||
uint64_t GuestLength;
|
||||
|
||||
/* RAData followed by IRData */
|
||||
uint8_t InlineData[0];
|
||||
|
||||
IR::RegisterAllocationData *GetRAData();
|
||||
IR::IRListView *GetIRData();
|
||||
};
|
||||
|
||||
struct AOTIRInlineIndexEntry {
|
||||
uint64_t GuestStart;
|
||||
uint64_t DataOffset;
|
||||
};
|
||||
|
||||
struct AOTIRInlineIndex {
|
||||
uint64_t Count;
|
||||
uint64_t DataBase;
|
||||
AOTIRInlineIndexEntry Entries[0];
|
||||
|
||||
AOTIRInlineEntry *Find(uint64_t GuestStart);
|
||||
AOTIRInlineEntry *GetInlineEntry(uint64_t DataOffset);
|
||||
};
|
||||
|
||||
struct AOTIRCaptureCacheEntry {
|
||||
std::unique_ptr<std::ostream> Stream;
|
||||
std::map<uint64_t, uint64_t> Index;
|
||||
|
||||
void AppendAOTIRCaptureCache(uint64_t GuestRIP, uint64_t Start, uint64_t Length, uint64_t Hash, FEXCore::IR::IRListView *IRList, FEXCore::IR::RegisterAllocationData *RAData);
|
||||
};
|
||||
|
||||
struct Context {
|
||||
friend class FEXCore::HLE::SyscallHandler;
|
||||
#ifdef JIT_ARM64
|
||||
@@ -156,32 +125,6 @@ namespace FEXCore::Context {
|
||||
CustomCPUFactoryType CustomCPUFactory;
|
||||
FEXCore::Context::ExitHandler CustomExitHandler;
|
||||
|
||||
struct AOTIRCacheEntry {
|
||||
AOTIRInlineIndex *Array;
|
||||
void *mapping;
|
||||
size_t size;
|
||||
};
|
||||
|
||||
std::unordered_map<std::string, AOTIRCacheEntry> AOTIRCache;
|
||||
std::function<int(const std::string&)> AOTIRLoader;
|
||||
std::function<std::unique_ptr<std::ostream>(const std::string&)> AOTIRWriter;
|
||||
std::unordered_map<std::string, AOTIRCaptureCacheEntry> AOTIRCaptureCache;
|
||||
|
||||
struct AddrToFileEntry {
|
||||
uint64_t Start;
|
||||
uint64_t Len;
|
||||
uint64_t Offset;
|
||||
std::string fileid;
|
||||
std::string filename;
|
||||
void *CachedFileEntry;
|
||||
bool ContainsCode;
|
||||
};
|
||||
|
||||
using AddrToFileMapType = std::map<uint64_t, AddrToFileEntry>;
|
||||
AddrToFileMapType AddrToFile;
|
||||
std::map<std::string, std::string> FilesWithCode;
|
||||
|
||||
AddrToFileMapType::iterator FindAddrForFile(uint64_t Entry, uint64_t Length);
|
||||
#ifdef BLOCKSTATS
|
||||
std::unique_ptr<FEXCore::BlockSamplingData> BlockData;
|
||||
#endif
|
||||
@@ -253,9 +196,6 @@ namespace FEXCore::Context {
|
||||
// same as CompileBlock, but aborts on failure
|
||||
void CompileBlockJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP);
|
||||
|
||||
bool LoadAOTIRCache(int streamfd);
|
||||
void FinalizeAOTIRCache();
|
||||
void WriteFilesWithCode(std::function<void(const std::string& fileid, const std::string& filename)> Writer);
|
||||
/**
|
||||
* @brief Initializes the JIT compilers for the thread
|
||||
*
|
||||
@@ -337,6 +277,26 @@ namespace FEXCore::Context {
|
||||
// Public for threading
|
||||
void ExecutionThread(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
void FinalizeAOTIRCache() {
|
||||
IRCaptureCache.FinalizeAOTIRCache();
|
||||
}
|
||||
|
||||
void WriteFilesWithCode(std::function<void(const std::string& fileid, const std::string& filename)> Writer) {
|
||||
IRCaptureCache.WriteFilesWithCode(Writer);
|
||||
}
|
||||
|
||||
void SetAOTIRLoader(std::function<int(const std::string&)> CacheReader) {
|
||||
IRCaptureCache.SetAOTIRLoader(CacheReader);
|
||||
}
|
||||
|
||||
void SetAOTIRWriter(std::function<std::unique_ptr<std::ofstream>(const std::string&)> CacheWriter) {
|
||||
IRCaptureCache.SetAOTIRWriter(CacheWriter);
|
||||
}
|
||||
|
||||
void SetAOTIRRenamer(std::function<void(const std::string&)> CacheRenamer) {
|
||||
IRCaptureCache.SetAOTIRRenamer(CacheRenamer);
|
||||
}
|
||||
|
||||
protected:
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread, bool AlsoClearIRCache);
|
||||
|
||||
@@ -362,13 +322,7 @@ namespace FEXCore::Context {
|
||||
std::mutex ExitMutex;
|
||||
std::unique_ptr<GdbServer> DebugServer;
|
||||
|
||||
std::shared_mutex AOTIRCacheLock;
|
||||
std::shared_mutex AOTIRCaptureCacheWriteoutLock;
|
||||
std::atomic<bool> AOTIRCaptureCacheWriteoutFlusing;
|
||||
|
||||
std::queue<std::function<void()>> AOTIRCaptureCacheWriteoutQueue;
|
||||
void AOTIRCaptureCacheWriteoutQueue_Flush();
|
||||
void AOTIRCaptureCacheWriteoutQueue_Append(const std::function<void()> &fn);
|
||||
IR::AOTIRCaptureCache IRCaptureCache;
|
||||
|
||||
bool StartPaused = false;
|
||||
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
|
||||
|
||||
@@ -535,7 +535,6 @@ uint64_t HandleCASPAL_ARMv8(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
}
|
||||
|
||||
bool HandleAtomicVectorStore(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
mcontext_t* mcontext = &reinterpret_cast<ucontext_t*>(_ucontext)->uc_mcontext;
|
||||
siginfo_t* info = reinterpret_cast<siginfo_t*>(_info);
|
||||
|
||||
if (info->si_code != BUS_ADRALN) {
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
@@ -15,37 +16,16 @@ namespace FEXCore::CPU {
|
||||
#define STATE x28
|
||||
|
||||
// We want vixl to not allocate a default buffer. Jit and dispatcher will manually create one.
|
||||
Arm64Emitter::Arm64Emitter(size_t size) : vixl::aarch64::Assembler(size, vixl::aarch64::PositionDependentCode) {
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::Context *ctx, size_t size) : vixl::aarch64::Assembler(size, vixl::aarch64::PositionDependentCode) {
|
||||
CPU.SetUp();
|
||||
|
||||
auto Features = vixl::CPUFeatures::InferFromOS();
|
||||
SupportsAtomics = Features.Has(vixl::CPUFeatures::Feature::kAtomics);
|
||||
// RCPC is bugged on Snapdragon 865
|
||||
// Causes glibc cond16 test to immediately throw assert
|
||||
// __pthread_mutex_cond_lock: Assertion `mutex->__data.__owner == 0'
|
||||
SupportsRCPC = false; //Features.Has(vixl::CPUFeatures::Feature::kRCpc);
|
||||
|
||||
if (SupportsAtomics) {
|
||||
if (ctx->HostFeatures.SupportsAtomics) {
|
||||
// Hypervisor can hide this on the c630?
|
||||
Features.Combine(vixl::CPUFeatures::Feature::kLORegions);
|
||||
}
|
||||
|
||||
SetCPUFeatures(Features);
|
||||
|
||||
if (!SupportsAtomics) {
|
||||
WARN_ONCE_FMT("Host CPU doesn't support atomics. Expect bad performance");
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
// We need to get the CPU's cache line size
|
||||
// We expect sane targets that have correct cacheline sizes across clusters
|
||||
uint64_t CTR;
|
||||
__asm volatile ("mrs %[ctr], ctr_el0"
|
||||
: [ctr] "=r"(CTR));
|
||||
|
||||
DCacheLineSize = 4 << ((CTR >> 16) & 0xF);
|
||||
ICacheLineSize = 4 << (CTR & 0xF);
|
||||
#endif
|
||||
}
|
||||
|
||||
void Arm64Emitter::LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant) {
|
||||
@@ -70,7 +50,7 @@ void Arm64Emitter::PushCalleeSavedRegisters() {
|
||||
// We need to save pairs of registers
|
||||
// We save r19-r30
|
||||
MemOperand PairOffset(sp, -16, PreIndex);
|
||||
const std::array<std::pair<vixl::aarch64::XRegister, vixl::aarch64::XRegister>, 6> CalleeSaved = {{
|
||||
const std::array<std::pair<vixl::aarch64::Register, vixl::aarch64::Register>, 6> CalleeSaved = {{
|
||||
{x19, x20},
|
||||
{x21, x22},
|
||||
{x23, x24},
|
||||
@@ -133,7 +113,7 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
}
|
||||
|
||||
MemOperand PairOffset(sp, 16, PostIndex);
|
||||
const std::array<std::pair<vixl::aarch64::XRegister, vixl::aarch64::XRegister>, 6> CalleeSaved = {{
|
||||
const std::array<std::pair<vixl::aarch64::Register, vixl::aarch64::Register>, 6> CalleeSaved = {{
|
||||
{x29, x30},
|
||||
{x27, x28},
|
||||
{x25, x26},
|
||||
|
||||
@@ -58,12 +58,9 @@ const std::array<aarch64::VRegister, 12> RAFPR = {
|
||||
// be used by both Arm64 JIT and ARM64 Dispatcher
|
||||
class Arm64Emitter : public vixl::aarch64::Assembler {
|
||||
protected:
|
||||
Arm64Emitter(size_t size);
|
||||
Arm64Emitter(FEXCore::Context::Context *ctx, size_t size);
|
||||
|
||||
vixl::aarch64::CPU CPU;
|
||||
bool SupportsAtomics{};
|
||||
bool SupportsRCPC{};
|
||||
|
||||
void LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant);
|
||||
void SpillStaticRegs(bool FPRs = true, uint32_t SpillMask = ~0U);
|
||||
void FillStaticRegs(bool FPRs = true, uint32_t FillMask = ~0U);
|
||||
@@ -79,9 +76,6 @@ protected:
|
||||
|
||||
uint32_t SpillSlots{};
|
||||
|
||||
uint32_t DCacheLineSize{};
|
||||
uint32_t ICacheLineSize{};
|
||||
|
||||
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
|
||||
};
|
||||
|
||||
|
||||
@@ -24,6 +24,7 @@ struct X86ContextBackup {
|
||||
uint64_t GPRs[23];
|
||||
FEXCore::x86_64::_libc_fpstate FPRState;
|
||||
uint64_t sa_mask;
|
||||
bool FaultToTopAndGeneratedException;
|
||||
|
||||
// Guest state
|
||||
int Signal;
|
||||
@@ -46,6 +47,7 @@ struct ArmContextBackup {
|
||||
uint32_t FPCR;
|
||||
__uint128_t FPRs[32];
|
||||
uint64_t sa_mask;
|
||||
bool FaultToTopAndGeneratedException;
|
||||
|
||||
// Guest state
|
||||
int Signal;
|
||||
|
||||
+349
-65
@@ -5,14 +5,16 @@ desc: Handles presented capability bits for guest cpu
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include "Common/StringConv.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/HostFeatures.h"
|
||||
#include "Utils/FileLoading.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include "git_version.h"
|
||||
|
||||
#include <cstring>
|
||||
@@ -21,6 +23,66 @@ $end_info$
|
||||
#endif
|
||||
|
||||
namespace FEXCore {
|
||||
namespace ProductNames {
|
||||
#ifdef _M_ARM_64
|
||||
static const char ARM_UNKNOWN[] = "Unknown ARM CPU";
|
||||
static const char ARM_A57[] = "Cortex-A57";
|
||||
static const char ARM_A72[] = "Cortex-A72";
|
||||
static const char ARM_A73[] = "Cortex-A73";
|
||||
static const char ARM_A75[] = "Cortex-A75";
|
||||
static const char ARM_A76[] = "Cortex-A76";
|
||||
static const char ARM_A76AE[] = "Cortex-A76AE";
|
||||
static const char ARM_V1[] = "Neoverse V1";
|
||||
static const char ARM_A77[] = "Cortex-A77";
|
||||
static const char ARM_A78[] = "Cortex-A78";
|
||||
static const char ARM_A78AE[] = "Cortex-A78AE";
|
||||
static const char ARM_A78C[] = "Cortex-A78C";
|
||||
static const char ARM_A710[] = "Cortex-A710";
|
||||
static const char ARM_X1[] = "Cortex-X1";
|
||||
static const char ARM_X2[] = "Cortex-X2";
|
||||
static const char ARM_N1[] = "Neoverse N1";
|
||||
static const char ARM_N2[] = "Neoverse N2";
|
||||
static const char ARM_E1[] = "Neoverse E1";
|
||||
static const char ARM_A35[] = "Cortex-A35";
|
||||
static const char ARM_A53[] = "Cortex-A53";
|
||||
static const char ARM_A55[] = "Cortex-A55";
|
||||
static const char ARM_A65[] = "Cortex-A65";
|
||||
static const char ARM_A510[] = "Cortex-A510";
|
||||
|
||||
static const char ARM_Kryo200[] = "Kryo 2xx";
|
||||
static const char ARM_Kryo300[] = "Kryo 3xx";
|
||||
static const char ARM_Kryo400[] = "Kryo 4xx/5xx";
|
||||
|
||||
static const char ARM_Kryo200S[] = "Kryo 2xx S";
|
||||
static const char ARM_Kryo300S[] = "Kryo 3xx S";
|
||||
static const char ARM_Kryo400S[] = "Kryo 4xx/5xx S";
|
||||
|
||||
static const char ARM_Denver[] = "Nvidia Denver";
|
||||
static const char ARM_Carmel[] = "Nvidia Carmel";
|
||||
|
||||
static const char ARM_Firestorm[] = "Apple Firestorm";
|
||||
static const char ARM_Icestorm[] = "Apple Icestorm";
|
||||
#else
|
||||
static const char UNKNOWN[] = "Unknown CPU";
|
||||
#endif
|
||||
}
|
||||
|
||||
static uint32_t GetCPUID() {
|
||||
uint32_t CPU{};
|
||||
FHU::Syscalls::getcpu(&CPU, nullptr);
|
||||
return CPU;
|
||||
}
|
||||
|
||||
static uint32_t CalculateNumberOfCPUs() {
|
||||
size_t CPUs = 1;
|
||||
|
||||
while(std::filesystem::exists("/sys/devices/system/cpu/cpu" + std::to_string(CPUs))) {
|
||||
CPUs++;
|
||||
}
|
||||
|
||||
return CPUs;
|
||||
}
|
||||
|
||||
constexpr uint32_t SUPPORTS_AVX = 0;
|
||||
// #define CPUID_AMD
|
||||
#ifdef CPUID_AMD
|
||||
@@ -49,60 +111,242 @@ static uint32_t GetCycleCounterFrequency() {
|
||||
return Result;
|
||||
}
|
||||
|
||||
static bool GetHostHybridFlag() {
|
||||
int MaxCPUs = 64;
|
||||
size_t AllocSize = CPU_ALLOC_SIZE(MaxCPUs);
|
||||
cpu_set_t *Set = CPU_ALLOC(MaxCPUs);
|
||||
CPU_ZERO_S(AllocSize, Set);
|
||||
void CPUIDEmu::SetupHostHybridFlag() {
|
||||
size_t CPUs = CalculateNumberOfCPUs();
|
||||
PerCPUData.resize(CPUs);
|
||||
|
||||
int Result{};
|
||||
for (;;) {
|
||||
Result = sched_getaffinity(0, AllocSize, Set);
|
||||
if (Result == 0 ||
|
||||
(Result == -1 && errno != EINVAL)) {
|
||||
break;
|
||||
}
|
||||
|
||||
MaxCPUs <<= 1;
|
||||
CPU_FREE(Set);
|
||||
Set = CPU_ALLOC(MaxCPUs);
|
||||
AllocSize = CPU_ALLOC_SIZE(MaxCPUs);
|
||||
CPU_ZERO_S(AllocSize, Set);
|
||||
}
|
||||
|
||||
if (Result != 0) {
|
||||
return false;
|
||||
}
|
||||
|
||||
int CPUs = CPU_COUNT_S(AllocSize, Set);
|
||||
|
||||
bool Hybrid = false;
|
||||
uint64_t MIDR{};
|
||||
for (int i = 0; i < CPUs; ++i) {
|
||||
if (CPU_ISSET_S(i, AllocSize, Set)) {
|
||||
std::error_code ec{};
|
||||
std::string MIDRPath = "/sys/devices/system/cpu/cpu" + std::to_string(i) + "/regs/identification/midr_el1";
|
||||
if (std::filesystem::exists(MIDRPath, ec)) {
|
||||
std::vector<char> Data{};
|
||||
// Needs to be a fixed size since depending on kernel it will try to read a full page of data and fail
|
||||
// Only read 18 bytes for a 64bit value prefixed with 0x
|
||||
if (FEXCore::FileLoading::LoadFile(Data, MIDRPath, 18)) {
|
||||
uint64_t NewMIDR{};
|
||||
if (FEXCore::StrConv::Conv(&Data.at(0), &NewMIDR)) {
|
||||
if (MIDR != 0 && MIDR != NewMIDR) {
|
||||
// CPU mismatch, claim hybrid
|
||||
Hybrid = true;
|
||||
break;
|
||||
}
|
||||
MIDR = NewMIDR;
|
||||
for (size_t i = 0; i < CPUs; ++i) {
|
||||
std::error_code ec{};
|
||||
std::string MIDRPath = "/sys/devices/system/cpu/cpu" + std::to_string(i) + "/regs/identification/midr_el1";
|
||||
if (std::filesystem::exists(MIDRPath, ec)) {
|
||||
std::vector<char> Data{};
|
||||
// Needs to be a fixed size since depending on kernel it will try to read a full page of data and fail
|
||||
// Only read 18 bytes for a 64bit value prefixed with 0x
|
||||
if (FEXCore::FileLoading::LoadFile(Data, MIDRPath, 18)) {
|
||||
uint64_t NewMIDR{};
|
||||
std::string_view MIDRView(&Data.at(0), 18);
|
||||
if (FEXCore::StrConv::Conv(MIDRView, &NewMIDR)) {
|
||||
if (MIDR != 0 && MIDR != NewMIDR) {
|
||||
// CPU mismatch, claim hybrid
|
||||
Hybrid = true;
|
||||
}
|
||||
|
||||
// Truncate to 32-bits, top 32-bits are all reserved in MIDR
|
||||
PerCPUData[i].ProductName = ProductNames::ARM_UNKNOWN;
|
||||
PerCPUData[i].MIDR = NewMIDR;
|
||||
MIDR = NewMIDR;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
CPU_FREE(Set);
|
||||
return Hybrid;
|
||||
struct CPUMIDR {
|
||||
uint8_t Implementer;
|
||||
uint16_t Part;
|
||||
bool DefaultBig; // Defaults to a big core
|
||||
const char *ProductName{};
|
||||
};
|
||||
|
||||
// CPU priority order
|
||||
// This is mostly arbitrary but will sort by some sort of CPU priority by performance
|
||||
// Relative list so things they will commonly end up in big.little configurations sort of relate
|
||||
static constexpr std::array<CPUMIDR, 35> CPUMIDRs = {{
|
||||
// Typically big CPU cores
|
||||
{0x61, 0x023, 1, ProductNames::ARM_Firestorm}, // Apple M1 Firestorm
|
||||
|
||||
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
|
||||
{0x41, 0xd4b, 1, ProductNames::ARM_A78C}, // A78C
|
||||
{0x41, 0xd4a, 1, ProductNames::ARM_E1}, // E1
|
||||
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
|
||||
{0x41, 0xd48, 1, ProductNames::ARM_X2}, // X2
|
||||
{0x41, 0xd47, 1, ProductNames::ARM_A710}, // A710
|
||||
{0x41, 0xd44, 1, ProductNames::ARM_X1}, // X1
|
||||
{0x41, 0xd42, 1, ProductNames::ARM_A78AE}, // A78AE
|
||||
{0x41, 0xd41, 1, ProductNames::ARM_A78}, // A78
|
||||
{0x41, 0xd40, 1, ProductNames::ARM_V1}, // V1
|
||||
{0x41, 0xd0e, 1, ProductNames::ARM_A76AE}, // A76AE
|
||||
{0x41, 0xd0d, 1, ProductNames::ARM_A77}, // A77
|
||||
{0x41, 0xd0c, 1, ProductNames::ARM_N1}, // N1
|
||||
{0x41, 0xd0b, 1, ProductNames::ARM_A76}, // A76
|
||||
{0x51, 0x804, 1, ProductNames::ARM_Kryo400}, // Kryo 4xx Gold (A76 based)
|
||||
{0x41, 0xd0a, 1, ProductNames::ARM_A75}, // A75
|
||||
{0x51, 0x802, 1, ProductNames::ARM_Kryo300}, // Kryo 3xx Gold (A75 based)
|
||||
{0x41, 0xd09, 1, ProductNames::ARM_A73}, // A73
|
||||
{0x51, 0x800, 1, ProductNames::ARM_Kryo200}, // Kryo 2xx Gold (A73 based)
|
||||
{0x41, 0xd08, 1, ProductNames::ARM_A72}, // A72
|
||||
|
||||
{0x4e, 0x004, 1, ProductNames::ARM_Carmel}, // Carmel
|
||||
|
||||
// Denver rated above A57 to match TX2 weirdness
|
||||
{0x4e, 0x003, 1, ProductNames::ARM_Denver}, // Denver
|
||||
|
||||
{0x41, 0xd07, 1, ProductNames::ARM_A57}, // A57
|
||||
|
||||
// Typically Little CPU cores
|
||||
{0x61, 0x022, 0, ProductNames::ARM_Icestorm}, // Apple M1 Icestorm
|
||||
{0x41, 0xd46, 0, ProductNames::ARM_A510}, // A510
|
||||
{0x41, 0xd06, 0, ProductNames::ARM_A65}, // A65
|
||||
{0x41, 0xd05, 0, ProductNames::ARM_A55}, // A55
|
||||
{0x51, 0x805, 0, ProductNames::ARM_Kryo400S}, // Kryo 4xx/5xx Silver (A55 based)
|
||||
{0x51, 0x803, 0, ProductNames::ARM_Kryo300S}, // Kryo 3xx Silver (A55 based)
|
||||
{0x41, 0xd03, 0, ProductNames::ARM_A53}, // A53
|
||||
{0x51, 0x801, 0, ProductNames::ARM_Kryo200S}, // Kryo 2xx Silver (A53 based)
|
||||
{0x41, 0xd04, 0, ProductNames::ARM_A35}, // A35
|
||||
|
||||
{0x41, 0, 0, ProductNames::ARM_UNKNOWN}, // Invalid CPU or Apple CPU inside Parallels VM
|
||||
{0x0, 0, 0, ProductNames::ARM_UNKNOWN}, // Invalid starting point is lowest ranked
|
||||
}};
|
||||
|
||||
auto FindDefinedMIDR = [](uint32_t MIDR) -> const CPUMIDR* {
|
||||
uint8_t Implementer = MIDR >> 24;
|
||||
uint16_t Part = (MIDR >> 4) & 0xFFF;
|
||||
|
||||
for (auto &MIDROption : CPUMIDRs) {
|
||||
if (MIDROption.Implementer == Implementer &&
|
||||
MIDROption.Part == Part) {
|
||||
return &MIDROption;
|
||||
}
|
||||
}
|
||||
|
||||
return nullptr;
|
||||
};
|
||||
|
||||
if (Hybrid) {
|
||||
// Walk the MIDRs and calculate big little designs
|
||||
std::vector<const CPUMIDR*> BigCores;
|
||||
std::vector<const CPUMIDR*> LittleCores;
|
||||
|
||||
// Separate CPU cores out to big or little selected
|
||||
for (size_t i = 0; i < CPUs; ++i) {
|
||||
uint32_t MIDR = PerCPUData[i].MIDR;
|
||||
auto MIDROption = FindDefinedMIDR(MIDR);
|
||||
if (MIDROption) {
|
||||
// Found one
|
||||
if (MIDROption->DefaultBig) {
|
||||
BigCores.emplace_back(MIDROption);
|
||||
}
|
||||
else {
|
||||
LittleCores.emplace_back(MIDROption);
|
||||
}
|
||||
}
|
||||
else {
|
||||
// If we didn't insert this MIDR then claim it is a little core.
|
||||
LittleCores.emplace_back(&CPUMIDRs.back());
|
||||
}
|
||||
}
|
||||
|
||||
if (LittleCores.empty()) {
|
||||
// If we only ended up with big cores then we need to move some to be little cores
|
||||
uint32_t LowestMIDR = ~0U;
|
||||
uint32_t LowestMIDRIdx = 0;
|
||||
// Walk all the big cores
|
||||
for (size_t i = 0; i < BigCores.size(); ++i) {
|
||||
uint8_t Implementer = BigCores[i]->Implementer;
|
||||
uint16_t Part = BigCores[i]->Part;
|
||||
|
||||
// Walk our list of CPUMIDRs to find the most little core
|
||||
for (size_t j = LowestMIDRIdx; j < CPUMIDRs.size(); ++j) {
|
||||
auto &MIDROption = CPUMIDRs[i];
|
||||
if ((MIDROption.Implementer == Implementer &&
|
||||
MIDROption.Part == Part) ||
|
||||
(MIDROption.Implementer == 0 &&
|
||||
MIDROption.Part == 0)) {
|
||||
|
||||
LowestMIDRIdx = j;
|
||||
LowestMIDR = MIDR;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Now we WILL have found a big core to demote to little status
|
||||
// Demote them
|
||||
std::erase_if(BigCores, [&LittleCores, LowestMIDR](auto *Entry) {
|
||||
// Demote by erase copy to little array
|
||||
uint8_t Implementer = LowestMIDR >> 24;
|
||||
uint16_t Part = (LowestMIDR >> 4) & 0xFFF;
|
||||
|
||||
if (Entry->Implementer == Implementer &&
|
||||
Entry->Part == Part) {
|
||||
// Add it to the BigCore list
|
||||
LittleCores.emplace_back(Entry);
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
});
|
||||
}
|
||||
|
||||
if (BigCores.empty()) {
|
||||
// We never found a CPU core we understand
|
||||
// Grab the first core, consider it as little, move everything else to Big
|
||||
uint32_t LittleMIDR = PerCPUData[0].MIDR;
|
||||
// Now walk the little cores and move them to Big if they don't match
|
||||
std::erase_if(LittleCores, [&BigCores, LittleMIDR](auto *Entry) {
|
||||
// You're promoted now
|
||||
uint8_t Implementer = LittleMIDR >> 24;
|
||||
uint16_t Part = (LittleMIDR >> 4) & 0xFFF;
|
||||
|
||||
if (Entry->Implementer != Implementer ||
|
||||
Entry->Part != Part) {
|
||||
// Add it to the BigCore list
|
||||
BigCores.emplace_back(Entry);
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
});
|
||||
}
|
||||
|
||||
// Now walk the per CPU data one more time and set if it is big or little
|
||||
for (auto &Data : PerCPUData) {
|
||||
uint8_t Implementer = Data.MIDR >> 24;
|
||||
uint16_t Part = (Data.MIDR >> 4) & 0xFFF;
|
||||
|
||||
bool FoundBig{};
|
||||
const CPUMIDR *MIDR{};
|
||||
for (auto Big : BigCores) {
|
||||
if (Big->Implementer == Implementer &&
|
||||
Big->Part == Part) {
|
||||
FoundBig = true;
|
||||
MIDR = Big;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (!FoundBig) {
|
||||
for (auto Little : LittleCores) {
|
||||
if (Little->Implementer == Implementer &&
|
||||
Little->Part == Part) {
|
||||
MIDR = Little;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Data.IsBig = FoundBig;
|
||||
if (MIDR) {
|
||||
Data.ProductName = MIDR->ProductName ?: ProductNames::ARM_UNKNOWN;
|
||||
}
|
||||
else {
|
||||
Data.ProductName = ProductNames::ARM_UNKNOWN;
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
// If we aren't hybrid then just claim everything is big
|
||||
for (size_t i = 0; i < CPUs; ++i) {
|
||||
uint32_t MIDR = PerCPUData[i].MIDR;
|
||||
auto MIDROption = FindDefinedMIDR(MIDR);
|
||||
|
||||
PerCPUData[i].IsBig = true;
|
||||
if (MIDROption) {
|
||||
PerCPUData[i].ProductName = MIDROption->ProductName ?: ProductNames::ARM_UNKNOWN;
|
||||
}
|
||||
else {
|
||||
PerCPUData[i].ProductName = ProductNames::ARM_UNKNOWN;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#else
|
||||
@@ -119,16 +363,21 @@ static uint32_t GetCycleCounterFrequency() {
|
||||
return 0;
|
||||
}
|
||||
|
||||
static bool GetHostHybridFlag() {
|
||||
void CPUIDEmu::SetupHostHybridFlag() {
|
||||
uint32_t eax, ebx, ecx, edx;
|
||||
__cpuid(0, eax, ebx, ecx, edx);
|
||||
if (eax >= 0x7) {
|
||||
__cpuid(0x7, eax, ebx, ecx, edx);
|
||||
// Bit 15 of edx claims hybrid CPU
|
||||
return (edx & (1U << 15)) != 0;
|
||||
Hybrid = (edx & (1U << 15)) != 0;
|
||||
}
|
||||
|
||||
return false;
|
||||
size_t CPUs = CalculateNumberOfCPUs();
|
||||
PerCPUData.resize(CPUs);
|
||||
for (size_t i = 0; i < CPUs; ++i) {
|
||||
PerCPUData[i].IsBig = true;
|
||||
PerCPUData[i].ProductName = ProductNames::UNKNOWN;
|
||||
}
|
||||
}
|
||||
|
||||
#endif
|
||||
@@ -155,6 +404,8 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
uint32_t CoreCount = Cores();
|
||||
// XXX: Enable once the rest of the SSE4.2 instructions are emulated
|
||||
uint32_t SupportsSSE42 = CTX->HostFeatures.SupportsCRC && false ? 1 : 0;
|
||||
|
||||
Res.eax = FAMILY_IDENTIFIER;
|
||||
|
||||
@@ -184,7 +435,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
(0 << 17) | // Process-context identifiers
|
||||
(0 << 18) | // Prefetching from memory mapped device
|
||||
(1 << 19) | // SSE4.1
|
||||
(0 << 20) | // SSE4.2
|
||||
(SupportsSSE42 << 20) | // SSE4.2
|
||||
(0 << 21) | // X2APIC
|
||||
(1 << 22) | // MOVBE
|
||||
(1 << 23) | // POPCNT
|
||||
@@ -386,7 +637,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) {
|
||||
(0 << 5) | // AVX2 support
|
||||
(1 << 6) | // FPU data pointer updated only on exception
|
||||
(1 << 7) | // SMEP support
|
||||
(0 << 8) | // BMI2
|
||||
(1 << 8) | // BMI2
|
||||
(0 << 9) | // Enhanced REP MOVSB/STOSB
|
||||
(1 << 10) | // INVPCID for system software control of process-context
|
||||
(0 << 11) | // Restricted transactional memory
|
||||
@@ -549,6 +800,18 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_15h(uint32_t Leaf) {
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_1Ah(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
if (Hybrid) {
|
||||
uint32_t CPU = GetCPUID();
|
||||
auto &Data = PerCPUData[CPU];
|
||||
// 0x40 is a big CPU
|
||||
// 0x20 is a little CPU
|
||||
Res.eax |= (Data.IsBig ? 0x40 : 0x20) << 24;
|
||||
}
|
||||
return Res;
|
||||
}
|
||||
|
||||
// Highest extended function implemented
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0000h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
@@ -636,7 +899,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) {
|
||||
(1 << 24) | // FXSAVE/FXRSTOR
|
||||
(1 << 25) | // FXSAVE/FXRSTOR Optimizations
|
||||
(0 << 26) | // 1 gigabit pages
|
||||
(0 << 27) | // RDTSCP
|
||||
(1 << 27) | // RDTSCP
|
||||
(0 << 28) | // Reserved
|
||||
(1 << 29) | // Long Mode
|
||||
(0 << 30) | // 3DNow! Extensions
|
||||
@@ -644,27 +907,44 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) {
|
||||
return Res;
|
||||
}
|
||||
|
||||
constexpr char ProcessorBrand[48] = {
|
||||
constexpr char ProcessorBrand[32] = {
|
||||
GIT_DESCRIBE_STRING
|
||||
"\0"
|
||||
};
|
||||
|
||||
constexpr ssize_t DESCRIBE_STR_SIZE = std::char_traits<char>::length(GIT_DESCRIBE_STRING);
|
||||
static_assert(DESCRIBE_STR_SIZE < 32);
|
||||
|
||||
//Processor brand string
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0002h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
memcpy(&Res, &ProcessorBrand[0], sizeof(FEXCore::CPUID::FunctionResults));
|
||||
return Res;
|
||||
return Function_8000_0002h(Leaf, GetCPUID());
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0003h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
memcpy(&Res, &ProcessorBrand[16], sizeof(FEXCore::CPUID::FunctionResults));
|
||||
return Res;
|
||||
return Function_8000_0003h(Leaf, GetCPUID());
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0004h(uint32_t Leaf) {
|
||||
return Function_8000_0004h(Leaf, GetCPUID());
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0002h(uint32_t Leaf, uint32_t CPU) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
memcpy(&Res, &ProcessorBrand[32], sizeof(FEXCore::CPUID::FunctionResults));
|
||||
memset(&Res, ' ', sizeof(FEXCore::CPUID::FunctionResults));
|
||||
memcpy(&Res, &ProcessorBrand[0], std::min(16L, DESCRIBE_STR_SIZE));
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0003h(uint32_t Leaf, uint32_t CPU) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
memset(&Res, ' ', sizeof(FEXCore::CPUID::FunctionResults));
|
||||
memcpy(&Res, &ProcessorBrand[16], std::max(0L, DESCRIBE_STR_SIZE - 16));
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0004h(uint32_t Leaf, uint32_t CPU) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
auto &Data = PerCPUData[CPU];
|
||||
memcpy(&Res, Data.ProductName, std::min(strlen(Data.ProductName), sizeof(FEXCore::CPUID::FunctionResults)));
|
||||
return Res;
|
||||
}
|
||||
|
||||
@@ -757,7 +1037,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0008h(uint32_t Leaf) {
|
||||
Res.ebx =
|
||||
(0 << 2) | // XSaveErPtr: Saving and restoring error pointers
|
||||
(0 << 1) | // IRPerf: Instructions retired count support
|
||||
(0 << 0); // CLZERO support
|
||||
(CTX->HostFeatures.SupportsCLZERO << 0); // CLZERO support
|
||||
|
||||
uint32_t CoreCount = Cores() - 1;
|
||||
Res.ecx =
|
||||
@@ -920,6 +1200,10 @@ void CPUIDEmu::Init(FEXCore::Context::Context *ctx) {
|
||||
// 0x16: Processor frequency information
|
||||
// 0x17: SoC vendor attribute enumeration
|
||||
|
||||
// 0x1A: Hybrid Information Sub-leaf
|
||||
#ifndef CPUID_AMD
|
||||
RegisterFunction(0x1A, &CPUIDEmu::Function_1Ah);
|
||||
#endif
|
||||
// Largest extended function number
|
||||
RegisterFunction(0x8000'0000, &CPUIDEmu::Function_8000_0000h);
|
||||
// Processor vendor
|
||||
@@ -960,7 +1244,7 @@ void CPUIDEmu::Init(FEXCore::Context::Context *ctx) {
|
||||
// 0x8000'001F: AMD Secure Encryption
|
||||
|
||||
// Setup some state tracking
|
||||
Hybrid = GetHostHybridFlag();
|
||||
SetupHostHybridFlag();
|
||||
}
|
||||
}
|
||||
|
||||
+31
@@ -23,6 +23,10 @@ private:
|
||||
constexpr static uint32_t CPUID_VENDOR_AMD3 = 0x444D4163; // "cAMD"
|
||||
|
||||
public:
|
||||
// X86 cacheline size effectively has to be hardcoded to 64
|
||||
// if we report anything differently then applications are likely to break
|
||||
constexpr static uint64_t CACHELINE_SIZE = 64;
|
||||
|
||||
void Init(FEXCore::Context::Context *ctx);
|
||||
|
||||
FEXCore::CPUID::FunctionResults RunFunction(uint32_t Function, uint32_t Leaf) {
|
||||
@@ -34,6 +38,16 @@ public:
|
||||
|
||||
return (this->*Handler->second)(Leaf);
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults RunFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) {
|
||||
if (Function == 0x8000'0002U)
|
||||
return Function_8000_0002h(Leaf, CPU % PerCPUData.size());
|
||||
else if (Function == 0x8000'0003U)
|
||||
return Function_8000_0003h(Leaf, CPU % PerCPUData.size());
|
||||
else
|
||||
return Function_8000_0004h(Leaf, CPU % PerCPUData.size());
|
||||
}
|
||||
|
||||
private:
|
||||
FEXCore::Context::Context *CTX;
|
||||
bool Hybrid{};
|
||||
@@ -45,6 +59,14 @@ private:
|
||||
}
|
||||
|
||||
std::unordered_map<uint32_t, FunctionHandler> FunctionHandlers;
|
||||
struct CPUData {
|
||||
const char *ProductName{};
|
||||
#ifdef _M_ARM_64
|
||||
uint32_t MIDR{};
|
||||
#endif
|
||||
bool IsBig{};
|
||||
};
|
||||
std::vector<CPUData> PerCPUData{};
|
||||
|
||||
// Functions
|
||||
FEXCore::CPUID::FunctionResults Function_0h(uint32_t Leaf);
|
||||
@@ -55,11 +77,17 @@ private:
|
||||
FEXCore::CPUID::FunctionResults Function_07h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_0Dh(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_15h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_1Ah(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0000h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0001h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0002h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0003h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0004h(uint32_t Leaf);
|
||||
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0002h(uint32_t Leaf, uint32_t CPU);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0003h(uint32_t Leaf, uint32_t CPU);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0004h(uint32_t Leaf, uint32_t CPU);
|
||||
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0005h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0006h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0007h(uint32_t Leaf);
|
||||
@@ -68,5 +96,8 @@ private:
|
||||
FEXCore::CPUID::FunctionResults Function_8000_0019h(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_8000_001Dh(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_Reserved(uint32_t Leaf);
|
||||
|
||||
void SetupHostHybridFlag();
|
||||
|
||||
};
|
||||
}
|
||||
+114
-426
@@ -41,6 +41,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
@@ -50,6 +51,7 @@ $end_info$
|
||||
#include <cstdint>
|
||||
#include <filesystem>
|
||||
#include <functional>
|
||||
#include <fstream>
|
||||
#include <map>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
@@ -61,7 +63,6 @@ $end_info$
|
||||
#include <string.h>
|
||||
#include <string>
|
||||
#include <string_view>
|
||||
#include <sstream>
|
||||
#include <sys/mman.h>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/syscall.h>
|
||||
@@ -141,54 +142,8 @@ std::string_view const& GetGRegName(unsigned Reg) {
|
||||
} // namespace FEXCore::Core
|
||||
|
||||
namespace FEXCore::Context {
|
||||
void Context::AOTIRCaptureCacheWriteoutQueue_Flush() {
|
||||
{
|
||||
std::shared_lock lk{AOTIRCaptureCacheWriteoutLock};
|
||||
if (AOTIRCaptureCacheWriteoutQueue.size() == 0) {
|
||||
AOTIRCaptureCacheWriteoutFlusing.store(false);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
for (;;) {
|
||||
AOTIRCaptureCacheWriteoutLock.lock();
|
||||
std::function<void()> fn = std::move(AOTIRCaptureCacheWriteoutQueue.front());
|
||||
bool MaybeEmpty = false;
|
||||
AOTIRCaptureCacheWriteoutQueue.pop();
|
||||
MaybeEmpty = AOTIRCaptureCacheWriteoutQueue.size() == 0;
|
||||
AOTIRCaptureCacheWriteoutLock.unlock();
|
||||
|
||||
fn();
|
||||
if (MaybeEmpty) {
|
||||
std::shared_lock lk{AOTIRCaptureCacheWriteoutLock};
|
||||
if (AOTIRCaptureCacheWriteoutQueue.size() == 0) {
|
||||
AOTIRCaptureCacheWriteoutFlusing.store(false);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
LOGMAN_MSG_A_FMT("Must never get here");
|
||||
}
|
||||
|
||||
void Context::AOTIRCaptureCacheWriteoutQueue_Append(const std::function<void()> &fn) {
|
||||
bool Flush = false;
|
||||
|
||||
{
|
||||
std::unique_lock lk{AOTIRCaptureCacheWriteoutLock};
|
||||
AOTIRCaptureCacheWriteoutQueue.push(fn);
|
||||
if (AOTIRCaptureCacheWriteoutQueue.size() > 10000) {
|
||||
Flush = true;
|
||||
}
|
||||
}
|
||||
|
||||
bool test_val = false;
|
||||
if (Flush && AOTIRCaptureCacheWriteoutFlusing.compare_exchange_strong(test_val, true)) {
|
||||
AOTIRCaptureCacheWriteoutQueue_Flush();
|
||||
}
|
||||
}
|
||||
|
||||
Context::Context() {
|
||||
Context::Context()
|
||||
: IRCaptureCache {this} {
|
||||
#ifdef BLOCKSTATS
|
||||
BlockData = std::make_unique<FEXCore::BlockSamplingData>();
|
||||
#endif
|
||||
@@ -211,10 +166,6 @@ namespace FEXCore::Context {
|
||||
}
|
||||
Threads.clear();
|
||||
}
|
||||
|
||||
for (auto &Mod: AOTIRCache) {
|
||||
FEXCore::Allocator::munmap(Mod.second.mapping, Mod.second.size);
|
||||
}
|
||||
}
|
||||
|
||||
static FEXCore::Core::CPUState CreateDefaultCPUState() {
|
||||
@@ -323,7 +274,7 @@ namespace FEXCore::Context {
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Pause);
|
||||
if (Thread->RunningEvents.Running.load()) {
|
||||
// Only attempt to stop this thread if it is running
|
||||
tgkill(Thread->ThreadManager.PID, Thread->ThreadManager.TID, SignalDelegator::SIGNAL_FOR_PAUSE);
|
||||
FHU::Syscalls::tgkill(Thread->ThreadManager.PID, Thread->ThreadManager.TID, SignalDelegator::SIGNAL_FOR_PAUSE);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -386,7 +337,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
void Context::Stop(bool IgnoreCurrentThread) {
|
||||
pid_t tid = gettid();
|
||||
pid_t tid = FHU::Syscalls::gettid();
|
||||
FEXCore::Core::InternalThreadState* CurrentThread{};
|
||||
|
||||
// Tell all the threads that they should stop
|
||||
@@ -426,20 +377,25 @@ namespace FEXCore::Context {
|
||||
void Context::StopThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
if (Thread->RunningEvents.Running.exchange(false)) {
|
||||
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Stop);
|
||||
tgkill(Thread->ThreadManager.PID, Thread->ThreadManager.TID, SignalDelegator::SIGNAL_FOR_PAUSE);
|
||||
FHU::Syscalls::tgkill(Thread->ThreadManager.PID, Thread->ThreadManager.TID, SignalDelegator::SIGNAL_FOR_PAUSE);
|
||||
}
|
||||
}
|
||||
|
||||
void Context::SignalThread(FEXCore::Core::InternalThreadState *Thread, FEXCore::Core::SignalEvent Event) {
|
||||
if (Thread->RunningEvents.Running.load()) {
|
||||
Thread->SignalReason.store(Event);
|
||||
tgkill(Thread->ThreadManager.PID, Thread->ThreadManager.TID, SignalDelegator::SIGNAL_FOR_PAUSE);
|
||||
FHU::Syscalls::tgkill(Thread->ThreadManager.PID, Thread->ThreadManager.TID, SignalDelegator::SIGNAL_FOR_PAUSE);
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::Context::ExitReason Context::RunUntilExit() {
|
||||
if(!StartPaused)
|
||||
Run();
|
||||
if(!StartPaused) {
|
||||
// We will only have one thread at this point, but just in case run notify everything
|
||||
std::lock_guard lk(ThreadCreationMutex);
|
||||
for (auto &Thread : Threads) {
|
||||
Thread->StartRunning.NotifyAll();
|
||||
}
|
||||
}
|
||||
|
||||
ExecutionThread(ParentThread);
|
||||
while(true) {
|
||||
@@ -474,7 +430,6 @@ namespace FEXCore::Context {
|
||||
};
|
||||
|
||||
LocalLoader->AddIR(IRHandler);
|
||||
|
||||
}
|
||||
|
||||
struct ExecutionThreadHandler {
|
||||
@@ -502,7 +457,7 @@ namespace FEXCore::Context {
|
||||
|
||||
void Context::InitializeThreadTLSData(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// Let's do some initial bookkeeping here
|
||||
Thread->ThreadManager.TID = ::gettid();
|
||||
Thread->ThreadManager.TID = FHU::Syscalls::gettid();
|
||||
Thread->ThreadManager.PID = ::getpid();
|
||||
SignalDelegation->RegisterTLSState(Thread);
|
||||
ThunkHandler->RegisterTLSState(Thread);
|
||||
@@ -538,9 +493,11 @@ namespace FEXCore::Context {
|
||||
|
||||
// Create CPU backend
|
||||
switch (Config.Core) {
|
||||
#ifdef INTERPRETER_ENABLED
|
||||
case FEXCore::Config::CONFIG_INTERPRETER:
|
||||
State->CPUBackend = FEXCore::CPU::CreateInterpreterCore(this, State, CompileThread);
|
||||
break;
|
||||
#endif
|
||||
case FEXCore::Config::CONFIG_IRJIT:
|
||||
State->PassManager->InsertRegisterAllocationPass(DoSRA);
|
||||
|
||||
@@ -665,6 +622,58 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
static void IRDumper(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, IR::RegisterAllocationData* RA) {
|
||||
FILE* f = nullptr;
|
||||
bool CloseAfter = false;
|
||||
const auto DumpIRStr = Thread->CTX->Config.DumpIR();
|
||||
|
||||
if (DumpIRStr =="stderr") {
|
||||
f = stderr;
|
||||
}
|
||||
else if (DumpIRStr =="stdout") {
|
||||
f = stdout;
|
||||
}
|
||||
else {
|
||||
const auto fileName = fmt::format("{}/{:x}{}", DumpIRStr, GuestRIP, RA ? "-post.ir" : "-pre.ir");
|
||||
f = fopen(fileName.c_str(), "w");
|
||||
CloseAfter = true;
|
||||
}
|
||||
|
||||
if (f) {
|
||||
std::stringstream out;
|
||||
auto NewIR = Thread->OpDispatcher->ViewIR();
|
||||
FEXCore::IR::Dump(&out, &NewIR, RA);
|
||||
fmt::print(f,"IR-{} 0x{:x}:\n{}\n@@@@@\n", RA ? "post" : "pre", GuestRIP, out.str());
|
||||
|
||||
if (CloseAfter) {
|
||||
fclose(f);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
static void ValidateIR(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// Convert to text, Parse, Convert to text again and make sure the texts match
|
||||
std::stringstream out;
|
||||
static auto compaction = IR::CreateIRCompaction();
|
||||
compaction->Run(Thread->OpDispatcher.get());
|
||||
auto NewIR = Thread->OpDispatcher->ViewIR();
|
||||
Dump(&out, &NewIR, nullptr);
|
||||
out.seekg(0);
|
||||
auto reparsed = IR::Parse(&out);
|
||||
if (reparsed == nullptr) {
|
||||
LOGMAN_MSG_A_FMT("Failed to parse IR\n");
|
||||
} else {
|
||||
std::stringstream out2;
|
||||
auto NewIR2 = reparsed->ViewIR();
|
||||
Dump(&out2, &NewIR2, nullptr);
|
||||
if (out.str() != out2.str()) {
|
||||
LogMan::Msg::IFmt("one:\n {}", out.str());
|
||||
LogMan::Msg::IFmt("two:\n {}", out2.str());
|
||||
LOGMAN_MSG_A_FMT("Parsed IR doesn't match\n");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Context::GenerateIRResult Context::GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
uint8_t const *GuestCode{};
|
||||
GuestCode = reinterpret_cast<uint8_t const*>(GuestRIP);
|
||||
@@ -734,7 +743,7 @@ namespace FEXCore::Context {
|
||||
else {
|
||||
if (Thread->OpDispatcher->HandledLock != IsLocked) {
|
||||
HadDispatchError = true;
|
||||
LogMan::Msg::EFmt("Missing LOCK HANDLER at 0x{:x}{{'{}'}}", Block.Entry + BlockInstructionsLength, TableInfo->Name);
|
||||
LogMan::Msg::EFmt("Missing LOCK HANDLER at 0x{:x}{{'{}'}}", Block.Entry + BlockInstructionsLength, TableInfo->Name ?: "UND");
|
||||
}
|
||||
BlockInstructionsLength += DecodedInfo->InstSize;
|
||||
TotalInstructionsLength += DecodedInfo->InstSize;
|
||||
@@ -771,73 +780,32 @@ namespace FEXCore::Context {
|
||||
|
||||
Thread->OpDispatcher->Finalize();
|
||||
|
||||
const auto IRDumper = [Thread, GuestRIP](IR::RegisterAllocationData* RA) {
|
||||
FILE* f = nullptr;
|
||||
bool CloseAfter = false;
|
||||
const auto DumpIRStr = Thread->CTX->Config.DumpIR();
|
||||
|
||||
if (DumpIRStr =="stderr") {
|
||||
f = stderr;
|
||||
}
|
||||
else if (DumpIRStr =="stdout") {
|
||||
f = stdout;
|
||||
}
|
||||
else {
|
||||
const auto fileName = fmt::format("{}/{:x}{}", DumpIRStr, GuestRIP, RA ? "-post.ir" : "-pre.ir");
|
||||
f = fopen(fileName.c_str(), "w");
|
||||
CloseAfter = true;
|
||||
// Debug
|
||||
{
|
||||
if (Thread->CTX->Config.DumpIR() != "no") {
|
||||
IRDumper(Thread, GuestRIP, nullptr);
|
||||
}
|
||||
|
||||
if (f) {
|
||||
std::stringstream out;
|
||||
auto NewIR = Thread->OpDispatcher->ViewIR();
|
||||
FEXCore::IR::Dump(&out, &NewIR, RA);
|
||||
fmt::print(f,"IR-{} 0x{:x}:\n{}\n@@@@@\n", RA ? "post" : "pre", GuestRIP, out.str());
|
||||
|
||||
if (CloseAfter) {
|
||||
fclose(f);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
if (Thread->CTX->Config.DumpIR() != "no") {
|
||||
IRDumper(nullptr);
|
||||
}
|
||||
|
||||
if (Thread->CTX->Config.ValidateIRarser) {
|
||||
// Convert to text, Parse, Convert to text again and make sure the texts match
|
||||
std::stringstream out;
|
||||
static auto compaction = IR::CreateIRCompaction();
|
||||
compaction->Run(Thread->OpDispatcher.get());
|
||||
auto NewIR = Thread->OpDispatcher->ViewIR();
|
||||
Dump(&out, &NewIR, nullptr);
|
||||
out.seekg(0);
|
||||
auto reparsed = IR::Parse(&out);
|
||||
if (reparsed == nullptr) {
|
||||
LOGMAN_MSG_A_FMT("Failed to parse IR\n");
|
||||
} else {
|
||||
std::stringstream out2;
|
||||
auto NewIR2 = reparsed->ViewIR();
|
||||
Dump(&out2, &NewIR2, nullptr);
|
||||
if (out.str() != out2.str()) {
|
||||
LogMan::Msg::IFmt("one:\n {}", out.str());
|
||||
LogMan::Msg::IFmt("two:\n {}", out2.str());
|
||||
LOGMAN_MSG_A_FMT("Parsed IR doesn't match\n");
|
||||
}
|
||||
if (Thread->CTX->Config.ValidateIRarser) {
|
||||
ValidateIR(Thread);
|
||||
}
|
||||
}
|
||||
|
||||
// Run the passmanager over the IR from the dispatcher
|
||||
Thread->PassManager->Run(Thread->OpDispatcher.get());
|
||||
|
||||
if (Thread->CTX->Config.DumpIR() != "no") {
|
||||
IRDumper(Thread->PassManager->HasPass("RA") ? Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA")->GetAllocationData() : nullptr);
|
||||
}
|
||||
// Debug
|
||||
{
|
||||
if (Thread->CTX->Config.DumpIR() != "no") {
|
||||
IRDumper(Thread, GuestRIP, Thread->PassManager->HasPass("RA") ? Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA")->GetAllocationData() : nullptr);
|
||||
}
|
||||
|
||||
if (Thread->OpDispatcher->ShouldDump) {
|
||||
std::stringstream out;
|
||||
auto NewIR = Thread->OpDispatcher->ViewIR();
|
||||
FEXCore::IR::Dump(&out, &NewIR, Thread->PassManager->HasPass("RA") ? Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA")->GetAllocationData() : nullptr);
|
||||
LogMan::Msg::IFmt("IR 0x{:x}:\n{}\n@@@@@\n", GuestRIP, out.str());
|
||||
if (Thread->OpDispatcher->ShouldDump) {
|
||||
std::stringstream out;
|
||||
auto NewIR = Thread->OpDispatcher->ViewIR();
|
||||
FEXCore::IR::Dump(&out, &NewIR, Thread->PassManager->HasPass("RA") ? Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA")->GetAllocationData() : nullptr);
|
||||
LogMan::Msg::IFmt("IR 0x{:x}:\n{}\n@@@@@\n", GuestRIP, out.str());
|
||||
}
|
||||
}
|
||||
|
||||
auto RAData = Thread->PassManager->HasPass("RA") ? Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA")->PullAllocationData() : nullptr;
|
||||
@@ -855,63 +823,6 @@ namespace FEXCore::Context {
|
||||
};
|
||||
}
|
||||
|
||||
AOTIRInlineEntry *AOTIRInlineIndex::GetInlineEntry(uint64_t DataOffset) {
|
||||
uintptr_t This = (uintptr_t)this;
|
||||
|
||||
return (AOTIRInlineEntry*)(This + DataBase + DataOffset);
|
||||
}
|
||||
|
||||
AOTIRInlineEntry *AOTIRInlineIndex::Find(uint64_t GuestStart) {
|
||||
ssize_t l = 0;
|
||||
ssize_t r = Count - 1;
|
||||
|
||||
while (l <= r) {
|
||||
size_t m = l + (r - l) / 2;
|
||||
|
||||
if (Entries[m].GuestStart == GuestStart)
|
||||
return GetInlineEntry(Entries[m].DataOffset);
|
||||
else if (Entries[m].GuestStart < GuestStart)
|
||||
l = m + 1;
|
||||
else
|
||||
r = m - 1;
|
||||
}
|
||||
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
IR::RegisterAllocationData *AOTIRInlineEntry::GetRAData() {
|
||||
return (IR::RegisterAllocationData *)InlineData;
|
||||
}
|
||||
|
||||
IR::IRListView *AOTIRInlineEntry::GetIRData() {
|
||||
auto RAData = GetRAData();
|
||||
auto Offset = RAData->Size(RAData->MapCount);
|
||||
|
||||
return (IR::IRListView *)&InlineData[Offset];
|
||||
}
|
||||
|
||||
void AOTIRCaptureCacheEntry::AppendAOTIRCaptureCache(uint64_t GuestRIP, uint64_t Start, uint64_t Length, uint64_t Hash, FEXCore::IR::IRListView *IRList, FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
auto Inserted = Index.emplace(GuestRIP, Stream->tellp());
|
||||
|
||||
if (Inserted.second) {
|
||||
//GuestHash
|
||||
Stream->write((const char*)&Hash, sizeof(Hash));
|
||||
|
||||
//GuestLength
|
||||
Stream->write((const char*)&Length, sizeof(Length));
|
||||
|
||||
// RAData (inline)
|
||||
// In file, IsShared is always set
|
||||
auto Shared = RAData->IsShared;
|
||||
RAData->IsShared = true;
|
||||
Stream->write((const char*)RAData, RAData->Size(RAData->MapCount));
|
||||
RAData->IsShared = Shared;
|
||||
|
||||
// IRData (inline)
|
||||
IRList->Serialize(*Stream);
|
||||
}
|
||||
}
|
||||
|
||||
Context::CompileCodeResult Context::CompileCode(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
FEXCore::IR::IRListView *IRList {};
|
||||
FEXCore::Core::DebugData *DebugData {};
|
||||
@@ -935,55 +846,17 @@ namespace FEXCore::Context {
|
||||
GeneratedIR = false;
|
||||
}
|
||||
|
||||
// AOT IR bookkeeping and cache
|
||||
{
|
||||
std::shared_lock lk(AOTIRCacheLock);
|
||||
auto file = AddrToFile.lower_bound(GuestRIP);
|
||||
if (file != AddrToFile.begin()) {
|
||||
--file;
|
||||
if (!file->second.ContainsCode) {
|
||||
file->second.ContainsCode = true;
|
||||
FilesWithCode[file->second.fileid] = file->second.filename;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (IRList == nullptr && Config.AOTIRLoad) {
|
||||
std::shared_lock lk(AOTIRCacheLock);
|
||||
auto file = AddrToFile.lower_bound(GuestRIP);
|
||||
if (file != AddrToFile.begin()) {
|
||||
--file;
|
||||
auto Mod = (AOTIRInlineIndex*)file->second.CachedFileEntry;
|
||||
|
||||
if (Mod == nullptr) {
|
||||
file->second.CachedFileEntry = Mod = AOTIRCache[file->second.fileid].Array;
|
||||
}
|
||||
|
||||
if (Mod != nullptr)
|
||||
{
|
||||
auto AOTEntry = Mod->Find(GuestRIP - file->second.Start + file->second.Offset);
|
||||
|
||||
if (AOTEntry) {
|
||||
// verify hash
|
||||
auto MappedStart = GuestRIP;
|
||||
auto hash = XXH3_64bits((void*)MappedStart, AOTEntry->GuestLength);
|
||||
if (hash == AOTEntry->GuestHash) {
|
||||
IRList = AOTEntry->GetIRData();
|
||||
//LogMan::Msg::DFmt("using {} + {:x} -> {:x}\n", file->second.fileid, AOTEntry->first, GuestRIP);
|
||||
|
||||
|
||||
RAData = AOTEntry->GetRAData();;
|
||||
DebugData = new FEXCore::Core::DebugData();
|
||||
StartAddr = MappedStart;
|
||||
Length = AOTEntry->GuestLength;
|
||||
|
||||
GeneratedIR = true;
|
||||
} else {
|
||||
LogMan::Msg::IFmt("AOTIR: hash check failed {:x}\n", MappedStart);
|
||||
}
|
||||
} else {
|
||||
//LogMan::Msg::IFmt("AOTIR: Failed to find {:x}, {:x}, {}\n", GuestRIP, GuestRIP - file->second.Start + file->second.Offset, file->second.fileid);
|
||||
}
|
||||
}
|
||||
auto [IRCopy, RACopy, DebugDataCopy, _StartAddr, _Length, _GeneratedIR] = IRCaptureCache.PreGenerateIRFetch(GuestRIP, IRList);
|
||||
if (_GeneratedIR) {
|
||||
// Setup pointers to internal structures
|
||||
IRList = IRCopy;
|
||||
RAData = RACopy;
|
||||
DebugData = DebugDataCopy;
|
||||
StartAddr = _StartAddr;
|
||||
Length = _Length;
|
||||
GeneratedIR = _GeneratedIR;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1024,114 +897,6 @@ namespace FEXCore::Context {
|
||||
};
|
||||
}
|
||||
|
||||
static bool readAll(int fd, void *data, size_t size) {
|
||||
int rv = read(fd, data, size);
|
||||
|
||||
if (rv != size)
|
||||
return false;
|
||||
else
|
||||
return true;
|
||||
}
|
||||
|
||||
bool Context::LoadAOTIRCache(int streamfd) {
|
||||
uint64_t tag;
|
||||
|
||||
if (!readAll(streamfd, (char*)&tag, sizeof(tag)) || tag != 0xDEADBEEFC0D30004)
|
||||
return false;
|
||||
|
||||
std::string Module;
|
||||
uint64_t ModSize;
|
||||
uint64_t IndexSize;
|
||||
|
||||
lseek(streamfd, -sizeof(ModSize), SEEK_END);
|
||||
|
||||
if (!readAll(streamfd, (char*)&ModSize, sizeof(ModSize)))
|
||||
return false;
|
||||
|
||||
Module.resize(ModSize);
|
||||
|
||||
lseek(streamfd, -sizeof(ModSize) - ModSize, SEEK_END);
|
||||
|
||||
if (!readAll(streamfd, (char*)&Module[0], Module.size()))
|
||||
return false;
|
||||
|
||||
lseek(streamfd, -sizeof(ModSize) - ModSize - sizeof(IndexSize), SEEK_END);
|
||||
|
||||
if (!readAll(streamfd, (char*)&IndexSize, sizeof(IndexSize)))
|
||||
return false;
|
||||
|
||||
struct stat fileinfo;
|
||||
if (fstat(streamfd, &fileinfo) < 0)
|
||||
return false;
|
||||
size_t Size = (fileinfo.st_size + 4095) & ~4095;
|
||||
|
||||
size_t IndexOffset = fileinfo.st_size - IndexSize -sizeof(ModSize) - ModSize - sizeof(IndexSize);
|
||||
|
||||
void *FilePtr = FEXCore::Allocator::mmap(nullptr, Size, PROT_READ, MAP_SHARED, streamfd, 0);
|
||||
|
||||
if (FilePtr == MAP_FAILED)
|
||||
return false;
|
||||
|
||||
auto Array = (AOTIRInlineIndex *)((char*)FilePtr + IndexOffset);
|
||||
|
||||
AOTIRCache.insert({Module, {Array, FilePtr, Size}});
|
||||
|
||||
LogMan::Msg::DFmt("AOTIR: Module {} has {} functions", Module, Array->Count);
|
||||
|
||||
return true;
|
||||
|
||||
}
|
||||
|
||||
void Context::WriteFilesWithCode(std::function<void(const std::string& fileid, const std::string& filename)> Writer) {
|
||||
std::shared_lock lk(AOTIRCacheLock);
|
||||
for( const auto &File: FilesWithCode) {
|
||||
Writer(File.first, File.second);
|
||||
}
|
||||
}
|
||||
|
||||
void Context::FinalizeAOTIRCache() {
|
||||
AOTIRCaptureCacheWriteoutQueue_Flush();
|
||||
|
||||
std::unique_lock lk(AOTIRCacheLock);
|
||||
|
||||
for (auto& [String, Entry] : AOTIRCaptureCache) {
|
||||
if (!Entry.Stream) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const auto ModSize = String.size();
|
||||
auto &stream = Entry.Stream;
|
||||
|
||||
// pad to 32 bytes
|
||||
constexpr char Zero = 0;
|
||||
while(stream->tellp() & 31)
|
||||
stream->write(&Zero, 1);
|
||||
|
||||
// AOTIRInlineIndex
|
||||
const auto FnCount = Entry.Index.size();
|
||||
const size_t DataBase = -stream->tellp();
|
||||
|
||||
stream->write((const char*)&FnCount, sizeof(FnCount));
|
||||
stream->write((const char*)&DataBase, sizeof(DataBase));
|
||||
|
||||
for (const auto& [GuestStart, DataOffset] : Entry.Index) {
|
||||
//AOTIRInlineIndexEntry
|
||||
|
||||
// GuestStart
|
||||
stream->write((const char*)&GuestStart, sizeof(GuestStart));
|
||||
|
||||
// DataOffset
|
||||
stream->write((const char*)&DataOffset, sizeof(DataOffset));
|
||||
}
|
||||
|
||||
// End of file header
|
||||
const auto IndexSize = FnCount * sizeof(AOTIRInlineIndexEntry) + sizeof(DataBase) + sizeof(FnCount);
|
||||
stream->write((const char*)&IndexSize, sizeof(IndexSize));
|
||||
stream->write(String.c_str(), ModSize);
|
||||
stream->write((const char*)&ModSize, sizeof(ModSize));
|
||||
}
|
||||
}
|
||||
|
||||
void Context::CompileBlockJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
auto NewBlock = CompileBlock(Frame, GuestRIP);
|
||||
|
||||
@@ -1143,19 +908,6 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
Context::AddrToFileMapType::iterator Context::FindAddrForFile(uint64_t Entry, uint64_t Length) {
|
||||
// Thread safety here! We are returning an iterator to the map object
|
||||
// This needs the AOTIRCacheLock locked prior to coming in to the function
|
||||
auto file = AddrToFile.lower_bound(Entry);
|
||||
if (file != AddrToFile.begin()) {
|
||||
--file;
|
||||
if (file->second.Start <= Entry && (file->second.Start + file->second.Len) >= (Entry + Length)) {
|
||||
return file;
|
||||
}
|
||||
}
|
||||
return AddrToFile.end();
|
||||
}
|
||||
|
||||
uintptr_t Context::CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
@@ -1227,56 +979,19 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
// Both generated ir and LibraryJITName need a named region lookup
|
||||
if (GeneratedIR || Config.LibraryJITNaming()) {
|
||||
std::shared_lock lk(AOTIRCacheLock);
|
||||
|
||||
auto file = FindAddrForFile(StartAddr, Length);
|
||||
|
||||
// Only go down this path if we actually found a library region
|
||||
if (file != AddrToFile.end()) {
|
||||
if (DebugData && Config.LibraryJITNaming()) {
|
||||
Symbols.RegisterNamedRegion(CodePtr, DebugData->HostCodeSize, file->second.filename);
|
||||
}
|
||||
|
||||
// Add to AOT cache if aot generation is enabled
|
||||
if (GeneratedIR && RAData &&
|
||||
(Config.AOTIRCapture() || Config.AOTIRGenerate())) {
|
||||
auto hash = XXH3_64bits((void*)StartAddr, Length);
|
||||
|
||||
auto LocalRIP = GuestRIP - file->second.Start + file->second.Offset;
|
||||
auto LocalStartAddr = StartAddr - file->second.Start + file->second.Offset;
|
||||
auto fileid = file->second.fileid;
|
||||
AOTIRCaptureCacheWriteoutQueue_Append([this, LocalRIP, LocalStartAddr, Length, hash, IRList, RAData, fileid]() {
|
||||
auto *AotFile = &AOTIRCaptureCache[fileid];
|
||||
|
||||
if (!AotFile->Stream) {
|
||||
AotFile->Stream = AOTIRWriter(fileid);
|
||||
uint64_t tag = 0xDEADBEEFC0D30004;
|
||||
AotFile->Stream->write((char*)&tag, sizeof(tag));
|
||||
}
|
||||
AotFile->AppendAOTIRCaptureCache(LocalRIP, LocalStartAddr, Length, hash, IRList, RAData);
|
||||
});
|
||||
|
||||
if (Config.AOTIRGenerate()) {
|
||||
// cleanup memory and early exit here -- we're not running the application
|
||||
|
||||
if (DecrementRefCount)
|
||||
--Thread->CompileBlockReentrantRefCount;
|
||||
|
||||
Thread->CPUBackend->ClearCache();
|
||||
|
||||
return (uintptr_t)CodePtr;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Insert to caches if we generated IR
|
||||
if (GeneratedIR) {
|
||||
// Add to thread local ir cache
|
||||
Core::LocalIREntry Entry = {StartAddr, Length, decltype(Entry.IR)(IRList), decltype(Entry.RAData)(RAData), decltype(Entry.DebugData)(DebugData)};
|
||||
Thread->LocalIRCache.insert({GuestRIP, std::move(Entry)});
|
||||
}
|
||||
if (IRCaptureCache.PostCompileCode(
|
||||
Thread,
|
||||
CodePtr,
|
||||
GuestRIP,
|
||||
StartAddr,
|
||||
Length,
|
||||
RAData,
|
||||
IRList,
|
||||
DebugData,
|
||||
GeneratedIR,
|
||||
DecrementRefCount)) {
|
||||
// Early exit
|
||||
return (uintptr_t)CodePtr;
|
||||
}
|
||||
|
||||
if (DecrementRefCount)
|
||||
@@ -1406,38 +1121,11 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
void Context::AddNamedRegion(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename) {
|
||||
// TODO: Support overlapping maps and region splitting
|
||||
auto base_filename = std::filesystem::path(filename).filename().string();
|
||||
|
||||
if (!base_filename.empty()) {
|
||||
auto filename_hash = XXH3_64bits(filename.c_str(), filename.size());
|
||||
|
||||
auto fileid = base_filename + "-" + std::to_string(filename_hash) + "-";
|
||||
|
||||
// append optimization flags to the fileid
|
||||
fileid += (Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL) ? "S" : "s";
|
||||
fileid += Config.TSOEnabled ? "T" : "t";
|
||||
fileid += Config.ABILocalFlags ? "L" : "l";
|
||||
fileid += Config.ABINoPF ? "p" : "P";
|
||||
|
||||
std::unique_lock lk(AOTIRCacheLock);
|
||||
|
||||
AddrToFile.insert({ Base, { Base, Size, Offset, fileid, filename, nullptr, false} });
|
||||
|
||||
if (Config.AOTIRLoad && !AOTIRCache.contains(fileid) && AOTIRLoader) {
|
||||
auto streamfd = AOTIRLoader(fileid);
|
||||
if (streamfd != -1) {
|
||||
LoadAOTIRCache(streamfd);
|
||||
close(streamfd);
|
||||
}
|
||||
}
|
||||
}
|
||||
IRCaptureCache.AddNamedRegion(Base, Size, Offset, filename);
|
||||
}
|
||||
|
||||
void Context::RemoveNamedRegion(uintptr_t Base, uintptr_t Size) {
|
||||
std::unique_lock lk(AOTIRCacheLock);
|
||||
// TODO: Support partial removing
|
||||
AddrToFile.erase(Base);
|
||||
IRCaptureCache.RemoveNamedRegion(Base, Size);
|
||||
}
|
||||
|
||||
void ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, std::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) {
|
||||
|
||||
@@ -11,6 +11,8 @@
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include <array>
|
||||
#include <bit>
|
||||
#include <cmath>
|
||||
@@ -37,7 +39,7 @@ static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
|
||||
#define STATE x28
|
||||
|
||||
Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config)
|
||||
: Dispatcher(ctx, Thread), Arm64Emitter(MAX_DISPATCHER_CODE_SIZE) {
|
||||
: Dispatcher(ctx, Thread), Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE) {
|
||||
SRAEnabled = config.StaticRegisterAssignment;
|
||||
SetAllowAssembler(true);
|
||||
|
||||
@@ -337,6 +339,31 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
hlt(0);
|
||||
}
|
||||
|
||||
{
|
||||
// Guest Overflow handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
OverflowExceptionInstructionAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (SRAEnabled)
|
||||
SpillStaticRegs();
|
||||
|
||||
LoadConstant(x0, reinterpret_cast<uint64_t>(&SynchronousFaultData));
|
||||
LoadConstant(w1, 1);
|
||||
strb(w1, MemOperand(x0, offsetof(Dispatcher::SynchronousFaultDataStruct, FaultToTopAndGeneratedException)));
|
||||
LoadConstant(w1, X86State::X86_TRAPNO_OF);
|
||||
str(w1, MemOperand(x0, offsetof(Dispatcher::SynchronousFaultDataStruct, TrapNo)));
|
||||
LoadConstant(w1, 0x80);
|
||||
str(w1, MemOperand(x0, offsetof(Dispatcher::SynchronousFaultDataStruct, si_code)));
|
||||
LoadConstant(x1, 0);
|
||||
str(w1, MemOperand(x0, offsetof(Dispatcher::SynchronousFaultDataStruct, err_code)));
|
||||
|
||||
// hlt/udf = SIGILL
|
||||
// brk = SIGTRAP
|
||||
// ??? = SIGSEGV
|
||||
// Force a SIGSEGV by loading zero
|
||||
ldr(x1, MemOperand(x1));
|
||||
}
|
||||
|
||||
{
|
||||
ThreadPauseHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
|
||||
if (SRAEnabled)
|
||||
@@ -428,7 +455,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
|
||||
GetBuffer()->SetExecutable();
|
||||
|
||||
if (CTX->Config.BlockJITNaming()) {
|
||||
std::string Name = "Dispatch_" + std::to_string(::gettid());
|
||||
std::string Name = "Dispatch_" + std::to_string(FHU::Syscalls::gettid());
|
||||
CTX->Symbols.Register(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr), Name);
|
||||
}
|
||||
if (CTX->Config.GlobalJITNaming()) {
|
||||
|
||||
+106
-20
@@ -81,6 +81,10 @@ ArchHelpers::Context::ContextBackup* Dispatcher::StoreThreadState(int Signal, vo
|
||||
Context->UContextLocation = 0;
|
||||
Context->SigInfoLocation = 0;
|
||||
|
||||
// Store fault to top status and then reset it
|
||||
Context->FaultToTopAndGeneratedException = SynchronousFaultData.FaultToTopAndGeneratedException;
|
||||
SynchronousFaultData.FaultToTopAndGeneratedException = false;
|
||||
|
||||
return Context;
|
||||
}
|
||||
|
||||
@@ -118,7 +122,8 @@ void Dispatcher::RestoreThreadState(void *ucontext) {
|
||||
[[maybe_unused]] auto *guest_siginfo = reinterpret_cast<siginfo_t*>(Context->SigInfoLocation);
|
||||
|
||||
// If the guest modified the RIP then we need to take special precautions here
|
||||
if (Context->OriginalRIP != guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_RIP]) {
|
||||
if (Context->OriginalRIP != guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_RIP] ||
|
||||
Context->FaultToTopAndGeneratedException) {
|
||||
// Hack! Go back to the top of the dispatcher top
|
||||
// This is only safe inside the JIT rather than anything outside of it
|
||||
ArchHelpers::Context::SetPc(ucontext, AbsoluteLoopTopAddressFillSRA);
|
||||
@@ -127,6 +132,14 @@ void Dispatcher::RestoreThreadState(void *ucontext) {
|
||||
|
||||
Frame->State.rip = guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_RIP];
|
||||
// XXX: Full context setting
|
||||
uint32_t eflags = guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_EFL];
|
||||
for (size_t i = 0; i < 32; ++i) {
|
||||
Frame->State.flags[i] = (eflags & (1U << i)) ? 1 : 0;
|
||||
}
|
||||
|
||||
Frame->State.flags[1] = 1;
|
||||
Frame->State.flags[9] = 1;
|
||||
|
||||
#define COPY_REG(x) \
|
||||
Frame->State.gregs[X86State::REG_##x] = guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_##x];
|
||||
COPY_REG(R8);
|
||||
@@ -146,13 +159,29 @@ void Dispatcher::RestoreThreadState(void *ucontext) {
|
||||
COPY_REG(RCX);
|
||||
COPY_REG(RSP);
|
||||
#undef COPY_REG
|
||||
FEXCore::x86_64::_libc_fpstate *fpstate = reinterpret_cast<FEXCore::x86_64::_libc_fpstate*>(guest_uctx->uc_mcontext.fpregs);
|
||||
// Copy float registers
|
||||
memcpy(Frame->State.mm, fpstate->_st, sizeof(Frame->State.mm));
|
||||
memcpy(Frame->State.xmm, fpstate->_xmm, sizeof(Frame->State.xmm));
|
||||
|
||||
// FCW store default
|
||||
Frame->State.FCW = fpstate->fcw;
|
||||
Frame->State.FTW = fpstate->ftw;
|
||||
|
||||
// Deconstruct FSW
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C0_LOC] = (fpstate->fsw >> 8) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C1_LOC] = (fpstate->fsw >> 9) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C2_LOC] = (fpstate->fsw >> 10) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C3_LOC] = (fpstate->fsw >> 14) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_TOP_LOC] = (fpstate->fsw >> 11) & 0b111;
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto *guest_uctx = reinterpret_cast<FEXCore::x86::ucontext_t*>(Context->UContextLocation);
|
||||
[[maybe_unused]] auto *guest_siginfo = reinterpret_cast<FEXCore::x86::siginfo_t*>(Context->SigInfoLocation);
|
||||
// If the guest modified the RIP then we need to take special precautions here
|
||||
if (Context->OriginalRIP != guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP]) {
|
||||
if (Context->OriginalRIP != guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP] ||
|
||||
Context->FaultToTopAndGeneratedException) {
|
||||
// Hack! Go back to the top of the dispatcher top
|
||||
// This is only safe inside the JIT rather than anything outside of it
|
||||
ArchHelpers::Context::SetPc(ucontext, AbsoluteLoopTopAddressFillSRA);
|
||||
@@ -160,17 +189,54 @@ void Dispatcher::RestoreThreadState(void *ucontext) {
|
||||
ArchHelpers::Context::SetState(ucontext, reinterpret_cast<uint64_t>(Frame));
|
||||
|
||||
// XXX: Full context setting
|
||||
// First 32-bytes of flags is EFLAGS broken out
|
||||
uint32_t eflags = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EFL];
|
||||
for (size_t i = 0; i < 32; ++i) {
|
||||
Frame->State.flags[i] = (eflags & (1U << i)) ? 1 : 0;
|
||||
}
|
||||
|
||||
Frame->State.flags[1] = 1;
|
||||
Frame->State.flags[9] = 1;
|
||||
|
||||
Frame->State.rip = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP];
|
||||
Frame->State.cs = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS];
|
||||
Frame->State.ds = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS];
|
||||
Frame->State.es = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES];
|
||||
Frame->State.fs = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS];
|
||||
Frame->State.gs = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS];
|
||||
Frame->State.ss = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS];
|
||||
#define COPY_REG(x) \
|
||||
Frame->State.gregs[X86State::REG_##x] = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_##x];
|
||||
COPY_REG(RDI);
|
||||
COPY_REG(RSI);
|
||||
COPY_REG(RBP);
|
||||
COPY_REG(RBX);
|
||||
COPY_REG(RDX);
|
||||
COPY_REG(RAX);
|
||||
COPY_REG(RCX);
|
||||
COPY_REG(RSP);
|
||||
COPY_REG(RDI);
|
||||
COPY_REG(RSI);
|
||||
COPY_REG(RBP);
|
||||
COPY_REG(RBX);
|
||||
COPY_REG(RDX);
|
||||
COPY_REG(RAX);
|
||||
COPY_REG(RCX);
|
||||
COPY_REG(RSP);
|
||||
#undef COPY_REG
|
||||
FEXCore::x86::_libc_fpstate *fpstate = reinterpret_cast<FEXCore::x86::_libc_fpstate*>(guest_uctx->uc_mcontext.fpregs);
|
||||
|
||||
// Copy float registers
|
||||
for (size_t i = 0; i < 8; ++i) {
|
||||
// 32-bit st register size is only 10 bytes. Not padded to 16byte like x86-64
|
||||
memcpy(&Frame->State.mm[i], &fpstate->_st[i], 10);
|
||||
}
|
||||
|
||||
// Extended XMM state
|
||||
memcpy(fpstate->_xmm, Frame->State.xmm, sizeof(Frame->State.xmm));
|
||||
|
||||
// FCW store default
|
||||
Frame->State.FCW = fpstate->fcw;
|
||||
Frame->State.FTW = fpstate->ftw;
|
||||
|
||||
// Deconstruct FSW
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C0_LOC] = (fpstate->fsw >> 8) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C1_LOC] = (fpstate->fsw >> 9) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C2_LOC] = (fpstate->fsw >> 10) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_C3_LOC] = (fpstate->fsw >> 14) & 1;
|
||||
Frame->State.flags[FEXCore::X86State::X87FLAG_TOP_LOC] = (fpstate->fsw >> 11) & 0b111;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -267,7 +333,10 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
// This can also happen if we have put a signal on hold, then we just reenabled the signal
|
||||
// So we are in the syscall handler
|
||||
// Only throw a log message in this case
|
||||
LogMan::Msg::EFmt("Signals in dispatcher have unsynchronized context");
|
||||
if constexpr (false) {
|
||||
// XXX: Messages in the signal handler can cause us to crash
|
||||
LogMan::Msg::EFmt("Signals in dispatcher have unsynchronized context");
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -334,8 +403,23 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_RIP] = Frame->State.rip;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_EFL] = 0;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_CSGSFS] = 0;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_ERR] = ConvertSignalToError(Signal, HostSigInfo);
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_TRAPNO] = ConvertSignalToTrapNo(Signal, HostSigInfo);
|
||||
|
||||
// aarch64 and x86_64 siginfo_t matches. We can just copy this over
|
||||
// SI_USER could also potentially have random data in it, needs to be bit perfect
|
||||
// For guest faults we don't have a real way to reconstruct state to a real guest RIP
|
||||
*guest_siginfo = *HostSigInfo;
|
||||
|
||||
if (ContextBackup->FaultToTopAndGeneratedException) {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_TRAPNO] = SynchronousFaultData.TrapNo;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_ERR] = SynchronousFaultData.err_code;
|
||||
|
||||
// Overwrite si_code
|
||||
guest_siginfo->si_code = SynchronousFaultData.si_code;
|
||||
}
|
||||
else {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_TRAPNO] = ConvertSignalToTrapNo(Signal, HostSigInfo);
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_ERR] = ConvertSignalToError(Signal, HostSigInfo);
|
||||
}
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_OLDMASK] = 0;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_CR2] = 0;
|
||||
|
||||
@@ -380,11 +464,6 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
guest_uctx->uc_stack.ss_sp = GuestStack->ss_sp;
|
||||
guest_uctx->uc_stack.ss_size = GuestStack->ss_size;
|
||||
|
||||
// aarch64 and x86_64 siginfo_t matches. We can just copy this over
|
||||
// SI_USER could also potentially have random data in it, needs to be bit perfect
|
||||
// For guest faults we don't have a real way to reconstruct state to a real guest RIP
|
||||
*guest_siginfo = *HostSigInfo;
|
||||
|
||||
Frame->State.gregs[X86State::REG_RSI] = SigInfoLocation;
|
||||
Frame->State.gregs[X86State::REG_RDX] = UContextLocation;
|
||||
}
|
||||
@@ -421,8 +500,16 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS] = Frame->State.fs;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES] = Frame->State.es;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS] = Frame->State.ds;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = ConvertSignalToTrapNo(Signal, HostSigInfo);
|
||||
if (ContextBackup->FaultToTopAndGeneratedException) {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = SynchronousFaultData.TrapNo;
|
||||
guest_siginfo->si_code = SynchronousFaultData.si_code;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = SynchronousFaultData.err_code;
|
||||
}
|
||||
else {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = ConvertSignalToTrapNo(Signal, HostSigInfo);
|
||||
guest_siginfo->si_code = HostSigInfo->si_code;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = ConvertSignalToError(Signal, HostSigInfo);
|
||||
}
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP] = Frame->State.rip;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS] = Frame->State.cs;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EFL] = 0;
|
||||
@@ -470,7 +557,6 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
|
||||
// These three elements are in every siginfo
|
||||
guest_siginfo->si_signo = HostSigInfo->si_signo;
|
||||
guest_siginfo->si_errno = HostSigInfo->si_errno;
|
||||
guest_siginfo->si_code = HostSigInfo->si_code;
|
||||
|
||||
switch (Signal) {
|
||||
case SIGSEGV:
|
||||
|
||||
@@ -49,12 +49,19 @@ public:
|
||||
uint64_t ExitFunctionLinkerAddress{};
|
||||
uint64_t SignalHandlerReturnAddress{};
|
||||
uint64_t UnimplementedInstructionAddress{};
|
||||
uint64_t OverflowExceptionInstructionAddress{};
|
||||
|
||||
uint64_t PauseReturnInstruction{};
|
||||
|
||||
/** @} */
|
||||
|
||||
uint32_t SignalHandlerRefCounter{};
|
||||
struct SynchronousFaultDataStruct {
|
||||
bool FaultToTopAndGeneratedException{};
|
||||
uint32_t TrapNo;
|
||||
uint32_t err_code;
|
||||
uint32_t si_code;
|
||||
} SynchronousFaultData;
|
||||
|
||||
uint64_t Start{};
|
||||
uint64_t End{};
|
||||
|
||||
@@ -11,6 +11,7 @@
|
||||
#include <FEXCore/Core/CPUBackend.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include <cmath>
|
||||
#include <memory>
|
||||
@@ -284,6 +285,24 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
ud2();
|
||||
}
|
||||
|
||||
{
|
||||
// Guest Overflow handler
|
||||
// Needs to be distinct from the SignalHandlerReturnAddress
|
||||
OverflowExceptionInstructionAddress = getCurr<uint64_t>();
|
||||
|
||||
// ud2 = SIGILL
|
||||
// int3 = SIGTRAP
|
||||
// hlt = SIGSEGV
|
||||
|
||||
mov(rax, reinterpret_cast<uint64_t>(&SynchronousFaultData));
|
||||
add(byte [rax + offsetof(Dispatcher::SynchronousFaultDataStruct, FaultToTopAndGeneratedException)], 1);
|
||||
mov(dword [rax + offsetof(Dispatcher::SynchronousFaultDataStruct, TrapNo)], X86State::X86_TRAPNO_OF);
|
||||
mov(dword [rax + offsetof(Dispatcher::SynchronousFaultDataStruct, err_code)], 0);
|
||||
mov(dword [rax + offsetof(Dispatcher::SynchronousFaultDataStruct, si_code)], 0x80);
|
||||
|
||||
hlt();
|
||||
}
|
||||
|
||||
{
|
||||
ReturnPtr = getCurr<FEXCore::Context::Context::IntCallbackReturn>();
|
||||
// using CallbackReturn = FEX_NAKED void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
|
||||
@@ -312,7 +331,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
|
||||
End = Start + getSize();
|
||||
|
||||
if (CTX->Config.BlockJITNaming()) {
|
||||
std::string Name = "Dispatch_" + std::to_string(::gettid());
|
||||
std::string Name = "Dispatch_" + std::to_string(FHU::Syscalls::gettid());
|
||||
CTX->Symbols.Register(reinterpret_cast<void*>(Start), End-Start, Name);
|
||||
}
|
||||
if (CTX->Config.GlobalJITNaming()) {
|
||||
|
||||
+16
-17
@@ -399,12 +399,12 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
|
||||
LogMan::Msg::DFmt("Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name, Op, DecodeInst->PC);
|
||||
LogMan::Msg::DFmt("Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
|
||||
return false;
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_INVALID) {
|
||||
LogMan::Msg::DFmt("Invalid or Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name, Op, DecodeInst->PC);
|
||||
LogMan::Msg::DFmt("Invalid or Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -664,7 +664,7 @@ bool Decoder::NormalOp(FEXCore::X86Tables::X86InstInfo const *Info, uint16_t Op,
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Bytes == 0, "Inst at 0x{:x}: 0x{:04x} '{}' Had an instruction of size {} with {} remaining",
|
||||
DecodeInst->PC, DecodeInst->OP, DecodeInst->TableInfo->Name, InstructionSize, Bytes);
|
||||
DecodeInst->PC, DecodeInst->OP, DecodeInst->TableInfo->Name ?: "UND", InstructionSize, Bytes);
|
||||
DecodeInst->InstSize = InstructionSize;
|
||||
return true;
|
||||
}
|
||||
@@ -680,12 +680,12 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
|
||||
LogMan::Msg::DFmt("Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name, Op, DecodeInst->PC);
|
||||
LogMan::Msg::DFmt("Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
|
||||
return false;
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_INVALID) {
|
||||
LogMan::Msg::DFmt("Invalid or Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name, Op, DecodeInst->PC);
|
||||
LogMan::Msg::DFmt("Invalid or Unknown instruction: {} 0x{:04x} 0x{:x}", Info->Name ?: "UND", Op, DecodeInst->PC);
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -880,20 +880,19 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
|
||||
}
|
||||
case 0x38: { // F38 Table!
|
||||
constexpr uint16_t PF_38_NONE = 0;
|
||||
constexpr uint16_t PF_38_66 = 1;
|
||||
constexpr uint16_t PF_38_F2 = 2;
|
||||
constexpr uint16_t PF_38_F3 = 3;
|
||||
constexpr uint16_t PF_38_66 = (1U << 0);
|
||||
constexpr uint16_t PF_38_F2 = (1U << 1);
|
||||
constexpr uint16_t PF_38_F3 = (1U << 2);
|
||||
|
||||
uint16_t Prefix = PF_38_NONE;
|
||||
if (DecodeInst->LastEscapePrefix == 0xF2) {
|
||||
// Repeat prefix or instruction-specific
|
||||
Prefix = PF_38_F2;
|
||||
} else if (DecodeInst->LastEscapePrefix == 0xF3) {
|
||||
// Repeat prefix or instruction-specific
|
||||
Prefix = PF_38_F3;
|
||||
} else if (DecodeInst->LastEscapePrefix == 0x66) {
|
||||
// Operand size
|
||||
Prefix = PF_38_66;
|
||||
if (DecodeInst->Flags & DecodeFlags::FLAG_OPERAND_SIZE) {
|
||||
Prefix |= PF_38_66;
|
||||
}
|
||||
if (DecodeInst->Flags & DecodeFlags::FLAG_REPNE_PREFIX) {
|
||||
Prefix |= PF_38_F2;
|
||||
}
|
||||
if (DecodeInst->Flags & DecodeFlags::FLAG_REP_PREFIX) {
|
||||
Prefix |= PF_38_F3;
|
||||
}
|
||||
|
||||
uint16_t LocalOp = (Prefix << 8) | ReadByte();
|
||||
|
||||
+1
-1
@@ -35,7 +35,7 @@ public:
|
||||
|
||||
uint64_t DecodedMinAddress {};
|
||||
uint64_t DecodedMaxAddress {~0ULL};
|
||||
|
||||
|
||||
void SetSectionMaxAddress(uint64_t v) { SectionMaxAddress = v; }
|
||||
void SetExternalBranches(std::set<uint64_t> *v) { ExternalBranches = v; }
|
||||
private:
|
||||
|
||||
+7
-6
@@ -480,7 +480,7 @@ std::string buildMemoryMap() {
|
||||
xml << "<?xml version='1.0'?>\n";
|
||||
|
||||
xml << "<!DOCTYPE memory-map>\n";
|
||||
xml << "<memory-map>";
|
||||
xml << "<memory-map>\n";
|
||||
|
||||
std::fstream fs("/proc/self/maps", std::fstream::in | std::fstream::binary);
|
||||
std::string Line;
|
||||
@@ -494,7 +494,7 @@ std::string buildMemoryMap() {
|
||||
}
|
||||
}
|
||||
|
||||
xml << "</memory-map>";
|
||||
xml << "</memory-map>\n";
|
||||
|
||||
xml << std::flush;
|
||||
|
||||
@@ -595,20 +595,20 @@ GdbServer::HandledPacketType GdbServer::handleXfer(const std::string &packet) {
|
||||
ThreadString = ss.str();
|
||||
}
|
||||
|
||||
return {encode(ThreadString.substr(offset, length)), HandledPacketType::TYPE_ACK};
|
||||
return {encode(ThreadString), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
|
||||
if (object == "memory-map") {
|
||||
if (offset == 0) {
|
||||
MemoryMapString = buildMemoryMap();
|
||||
}
|
||||
return {encode(MemoryMapString.substr(offset, length)), HandledPacketType::TYPE_ACK};
|
||||
return {encode(MemoryMapString), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
if (object == "osdata") {
|
||||
if (offset == 0) {
|
||||
OSDataString = buildOSData();
|
||||
}
|
||||
return {encode(OSDataString.substr(offset, length)), HandledPacketType::TYPE_ACK};
|
||||
return {encode(OSDataString), HandledPacketType::TYPE_ACK};
|
||||
}
|
||||
|
||||
return {"", HandledPacketType::TYPE_UNKNOWN};
|
||||
@@ -970,7 +970,8 @@ GdbServer::HandledPacketType GdbServer::handleThreadOp(const std::string &packet
|
||||
GdbServer::HandledPacketType GdbServer::handleBreakpoint(const std::string &packet) {
|
||||
auto ss = std::istringstream(packet);
|
||||
|
||||
bool Set{};
|
||||
// Don't do anything with set breakpoints yet
|
||||
[[maybe_unused]] bool Set{};
|
||||
uint64_t Addr;
|
||||
uint64_t Type;
|
||||
Set = ss.get() == 'Z';
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/HostFeatures.h"
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
@@ -13,14 +14,103 @@
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
// Data Zero Prohibited flag
|
||||
// 0b0 = ZVA/GVA/GZVA permitted
|
||||
// 0b1 = ZVA/GVA/GZVA prohibited
|
||||
constexpr uint32_t DCZID_DZP_MASK = 0b1'0000;
|
||||
// Log2 of the blocksize in 32-bit words
|
||||
constexpr uint32_t DCZID_BS_MASK = 0b0'1111;
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
static uint32_t GetDCZID() {
|
||||
uint64_t Result{};
|
||||
__asm("mrs %[Res], DCZID_EL0"
|
||||
: [Res] "=r" (Result));
|
||||
return Result;
|
||||
}
|
||||
|
||||
static uint32_t GetFPCR() {
|
||||
uint64_t Result{};
|
||||
__asm ("mrs %[Res], FPCR"
|
||||
: [Res] "=r" (Result));
|
||||
return Result;
|
||||
}
|
||||
|
||||
static void SetFPCR(uint64_t Value) {
|
||||
__asm ("msr FPCR, %[Value]"
|
||||
:: [Value] "r" (Value));
|
||||
}
|
||||
|
||||
#else
|
||||
static uint32_t GetDCZID() {
|
||||
// Return unsupported
|
||||
return DCZID_DZP_MASK;
|
||||
}
|
||||
#endif
|
||||
|
||||
|
||||
HostFeatures::HostFeatures() {
|
||||
#ifdef _M_ARM_64
|
||||
auto Features = vixl::CPUFeatures::InferFromOS();
|
||||
SupportsAES = Features.Has(vixl::CPUFeatures::Feature::kAES);
|
||||
SupportsCRC = Features.Has(vixl::CPUFeatures::Feature::kCRC32);
|
||||
SupportsAtomics = Features.Has(vixl::CPUFeatures::Feature::kAtomics);
|
||||
// Only supported when FEAT_AFP is supported
|
||||
SupportsFlushInputsToZero = Features.Has(vixl::CPUFeatures::Feature::kAFP);
|
||||
|
||||
// RCPC is bugged on Snapdragon 865
|
||||
// Causes glibc cond16 test to immediately throw assert
|
||||
// __pthread_mutex_cond_lock: Assertion `mutex->__data.__owner == 0'
|
||||
SupportsRCPC = false; //Features.Has(vixl::CPUFeatures::Feature::kRCpc);
|
||||
|
||||
// We need to get the CPU's cache line size
|
||||
// We expect sane targets that have correct cacheline sizes across clusters
|
||||
uint64_t CTR;
|
||||
__asm volatile ("mrs %[ctr], ctr_el0"
|
||||
: [ctr] "=r"(CTR));
|
||||
|
||||
DCacheLineSize = 4 << ((CTR >> 16) & 0xF);
|
||||
ICacheLineSize = 4 << (CTR & 0xF);
|
||||
|
||||
if (!SupportsAtomics) {
|
||||
WARN_ONCE_FMT("Host CPU doesn't support atomics. Expect bad performance");
|
||||
}
|
||||
#endif
|
||||
#ifdef _M_X86_64
|
||||
Xbyak::util::Cpu Features{};
|
||||
SupportsAES = Features.has(Xbyak::util::Cpu::tAESNI);
|
||||
SupportsCRC = Features.has(Xbyak::util::Cpu::tSSE42);
|
||||
SupportsFlushInputsToZero = true;
|
||||
SupportsFloatExceptions = true;
|
||||
#else
|
||||
// Test if this CPU supports float exception trapping by attempting to enable
|
||||
// On unsupported these bits are architecturally defined as RAZ/WI
|
||||
constexpr uint32_t ExceptionEnableTraps =
|
||||
(1U << 8) | // Invalid Operation float exception trap enable
|
||||
(1U << 9) | // Divide by zero float exception trap enable
|
||||
(1U << 10) | // Overflow float exception trap enable
|
||||
(1U << 11) | // Underflow float exception trap enable
|
||||
(1U << 12) | // Inexact float exception trap enable
|
||||
(1U << 15); // Input Denormal float exception trap enable
|
||||
|
||||
uint32_t OriginalFPCR = GetFPCR();
|
||||
uint32_t FPCR = OriginalFPCR | ExceptionEnableTraps;
|
||||
SetFPCR(FPCR);
|
||||
FPCR = GetFPCR();
|
||||
SupportsFloatExceptions = (FPCR & ExceptionEnableTraps) == ExceptionEnableTraps;
|
||||
|
||||
// Set FPCR back to original just in case anything changed
|
||||
SetFPCR(OriginalFPCR);
|
||||
#endif
|
||||
|
||||
// Check if we can support cacheline clears
|
||||
uint32_t DCZID = GetDCZID();
|
||||
if ((DCZID & DCZID_DZP_MASK) == 0) {
|
||||
uint32_t DCZID_Log2 = DCZID & DCZID_BS_MASK;
|
||||
uint32_t DCZID_Bytes = (1 << DCZID_Log2) * sizeof(uint32_t);
|
||||
// If the DC ZVA size matches the emulated cache line size
|
||||
// This means we can use the instruction
|
||||
SupportsCLZERO = DCZID_Bytes == CPUIDEmu::CACHELINE_SIZE;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,9 +1,27 @@
|
||||
#pragma once
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore {
|
||||
class HostFeatures final {
|
||||
public:
|
||||
HostFeatures();
|
||||
|
||||
/**
|
||||
* @brief Backend features that change how codegen is generated from IR
|
||||
*
|
||||
* Specifically things that affect the IR->Codegen process
|
||||
* Not the x86->IR process
|
||||
*/
|
||||
uint32_t DCacheLineSize{};
|
||||
uint32_t ICacheLineSize{};
|
||||
bool SupportsAES{};
|
||||
bool SupportsCRC{};
|
||||
bool SupportsCLZERO{};
|
||||
bool SupportsAtomics{};
|
||||
bool SupportsRCPC{};
|
||||
|
||||
// Float exception behaviour
|
||||
bool SupportsFlushInputsToZero{};
|
||||
bool SupportsFloatExceptions{};
|
||||
};
|
||||
}
|
||||
@@ -38,7 +38,13 @@ DEF_OP(Constant) {
|
||||
|
||||
DEF_OP(EntrypointOffset) {
|
||||
auto Op = IROp->C<IR::IROp_EntrypointOffset>();
|
||||
GD = Data->CurrentEntry + Op->Offset;
|
||||
uint64_t Mask = ~0ULL;
|
||||
uint8_t OpSize = IROp->Size;
|
||||
if (OpSize == 4) {
|
||||
Mask = 0xFFFF'FFFFULL;
|
||||
}
|
||||
|
||||
GD = (Data->CurrentEntry + Op->Offset) & Mask;
|
||||
}
|
||||
|
||||
DEF_OP(InlineConstant) {
|
||||
@@ -493,6 +499,54 @@ DEF_OP(Extr) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(PDep) {
|
||||
const auto Op = IROp->C<IR::IROp_PExt>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (OpSize != 4 && OpSize != 8) {
|
||||
LOGMAN_MSG_A_FMT("Unknown PDep Size: {}\n", OpSize);
|
||||
return;
|
||||
}
|
||||
|
||||
const uint64_t Input = OpSize == 4 ? *GetSrc<uint32_t*>(Data->SSAData, Op->Args(0))
|
||||
: *GetSrc<uint64_t*>(Data->SSAData, Op->Args(0));
|
||||
uint64_t Mask = OpSize == 4 ? *GetSrc<uint32_t*>(Data->SSAData, Op->Args(1))
|
||||
: *GetSrc<uint64_t*>(Data->SSAData, Op->Args(1));
|
||||
|
||||
uint64_t Result = 0;
|
||||
for (uint64_t Index = 0; Mask > 0; Index++) {
|
||||
const uint64_t Offset = std::countr_zero(Mask);
|
||||
Mask &= Mask - 1;
|
||||
Result |= ((Input >> Index) & 1) << Offset;
|
||||
}
|
||||
|
||||
GD = Result;
|
||||
}
|
||||
|
||||
DEF_OP(PExt) {
|
||||
const auto Op = IROp->C<IR::IROp_PExt>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (OpSize != 4 && OpSize != 8) {
|
||||
LOGMAN_MSG_A_FMT("Unknown PExt Size: {}\n", OpSize);
|
||||
return;
|
||||
}
|
||||
|
||||
const uint64_t Input = OpSize == 4 ? *GetSrc<uint32_t*>(Data->SSAData, Op->Args(0))
|
||||
: *GetSrc<uint64_t*>(Data->SSAData, Op->Args(0));
|
||||
uint64_t Mask = OpSize == 4 ? *GetSrc<uint32_t*>(Data->SSAData, Op->Args(1))
|
||||
: *GetSrc<uint64_t*>(Data->SSAData, Op->Args(1));
|
||||
|
||||
uint64_t Result = 0;
|
||||
for (uint64_t Offset = 0; Mask > 0; Offset++) {
|
||||
const uint64_t Index = std::countr_zero(Mask);
|
||||
Mask &= Mask - 1;
|
||||
Result |= ((Input >> Index) & 1) << Offset;
|
||||
}
|
||||
|
||||
GD = Result;
|
||||
}
|
||||
|
||||
DEF_OP(LDiv) {
|
||||
auto Op = IROp->C<IR::IROp_LDiv>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
@@ -787,9 +841,8 @@ DEF_OP(Bfi) {
|
||||
|
||||
DEF_OP(Bfe) {
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_A_FMT(OpSize <= 8, "OpSize is too large for BFE: {}", OpSize);
|
||||
LOGMAN_THROW_A_FMT(IROp->Size <= 8, "OpSize is too large for BFE: {}", IROp->Size);
|
||||
uint64_t SourceMask = (1ULL << Op->Width) - 1;
|
||||
if (Op->Width == 64)
|
||||
SourceMask = ~0ULL;
|
||||
@@ -800,9 +853,8 @@ DEF_OP(Bfe) {
|
||||
|
||||
DEF_OP(Sbfe) {
|
||||
auto Op = IROp->C<IR::IROp_Sbfe>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_A_FMT(OpSize <= 8, "OpSize is too large for SBFE: {}", OpSize);
|
||||
LOGMAN_THROW_A_FMT(IROp->Size <= 8, "OpSize is too large for SBFE: {}", IROp->Size);
|
||||
int64_t Src = *GetSrc<int64_t*>(Data->SSAData, Op->Header.Args[0]);
|
||||
uint64_t ShiftLeftAmount = (64 - (Op->Width + Op->lsb));
|
||||
uint64_t ShiftRightAmount = ShiftLeftAmount + Op->lsb;
|
||||
@@ -841,11 +893,10 @@ DEF_OP(Select) {
|
||||
|
||||
DEF_OP(VExtractToGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint32_t SourceSize = GetOpSize(Data->CurrentIR, Op->Header.Args[0]);
|
||||
|
||||
LOGMAN_THROW_A_FMT(OpSize <= 16, "OpSize is too large for VExtractToGPR: {}", OpSize);
|
||||
LOGMAN_THROW_A_FMT(IROp->Size <= 16, "OpSize is too large for VExtractToGPR: {}", IROp->Size);
|
||||
|
||||
if (SourceSize == 16) {
|
||||
__uint128_t SourceMask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
|
||||
|
||||
@@ -298,6 +298,63 @@ namespace AES {
|
||||
}
|
||||
}
|
||||
|
||||
namespace CRC32 {
|
||||
// CRC32 per byte lookup table.
|
||||
constexpr std::array<uint32_t, 256> CRC32CTable = []() consteval {
|
||||
std::array<uint32_t, 256> Table{};
|
||||
|
||||
// Clang 11.x doesn't support bitreverse as a consteval
|
||||
// constexpr uint32_t Polynomial = 0x1EDC6F41;
|
||||
constexpr uint32_t PolynomialRev = 0x82F63B78; //__builtin_bitreverse32(Polynomial);
|
||||
|
||||
for (size_t Char = 0; Char < std::size(Table); ++Char) {
|
||||
uint32_t CurrentChar = Char;
|
||||
for (size_t i = 0; i < 8; ++i) {
|
||||
if (CurrentChar & 1) {
|
||||
CurrentChar = (CurrentChar >> 1) ^ PolynomialRev;
|
||||
}
|
||||
else {
|
||||
CurrentChar >>= 1;
|
||||
}
|
||||
}
|
||||
Table[Char] = CurrentChar;
|
||||
}
|
||||
|
||||
return Table;
|
||||
}();
|
||||
|
||||
uint32_t crc32cb(uint32_t Accumulator, uint8_t data) {
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ data] ^ Accumulator >> 8;
|
||||
return Accumulator;
|
||||
}
|
||||
|
||||
uint32_t crc32ch(uint32_t Accumulator, uint16_t data) {
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 0) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 8) & 0xFF)] ^ Accumulator >> 8;
|
||||
return Accumulator;
|
||||
}
|
||||
|
||||
uint32_t crc32cw(uint32_t Accumulator, uint32_t data) {
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 0) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 8) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 16) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 24) & 0xFF)] ^ Accumulator >> 8;
|
||||
return Accumulator;
|
||||
}
|
||||
|
||||
uint32_t crc32cx(uint32_t Accumulator, uint64_t data) {
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 0) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 8) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 16) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 24) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 32) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 40) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 48) & 0xFF)] ^ Accumulator >> 8;
|
||||
Accumulator = CRC32CTable[(uint8_t)Accumulator ^ ((data >> 56) & 0xFF)] ^ Accumulator >> 8;
|
||||
return Accumulator;
|
||||
}
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
|
||||
@@ -429,6 +486,33 @@ DEF_OP(AESKeyGenAssist) {
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
}
|
||||
|
||||
DEF_OP(CRC32) {
|
||||
auto Op = IROp->C<IR::IROp_CRC32>();
|
||||
uint32_t Src1 = *GetSrc<uint32_t*>(Data->SSAData, Op->Src1);
|
||||
uint8_t *Src2 = GetSrc<uint8_t*>(Data->SSAData, Op->Src2);
|
||||
uint32_t Tmp{};
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 1:
|
||||
Tmp = CRC32::crc32cb(Src1, *(uint8_t*)Src2);
|
||||
break;
|
||||
case 2:
|
||||
Tmp = CRC32::crc32ch(Src1, *(uint16_t*)Src2);
|
||||
break;
|
||||
case 4:
|
||||
Tmp = CRC32::crc32cw(Src1, *(uint32_t*)Src2);
|
||||
break;
|
||||
case 8:
|
||||
Tmp = CRC32::crc32cx(Src1, *(uint64_t*)Src2);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown CRC32C size: {}", Op->SrcSize);
|
||||
break;
|
||||
|
||||
}
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -323,7 +323,7 @@ DEF_OP(F80BCDLOAD) {
|
||||
|
||||
DEF_OP(F80BCDSTORE) {
|
||||
auto Op = IROp->C<IR::IROp_F80BCDStore>();
|
||||
X80SoftFloat Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]);
|
||||
X80SoftFloat Src1 = X80SoftFloat::FRNDINT(*GetSrc<X80SoftFloat*>(Data->SSAData, Op->Header.Args[0]));
|
||||
bool Negative = Src1.Sign;
|
||||
|
||||
// Clear the Sign bit
|
||||
|
||||
@@ -227,6 +227,8 @@ struct OpHandlers<IR::OP_F80BCDSTORE> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
bool Negative = Src1.Sign;
|
||||
|
||||
Src1 = X80SoftFloat::FRNDINT(Src1);
|
||||
|
||||
// Clear the Sign bit
|
||||
Src1.Sign = 0;
|
||||
|
||||
|
||||
@@ -0,0 +1,209 @@
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/F80Ops.h"
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
template<typename R, typename... Args>
|
||||
static FallbackInfo GetFallbackInfo(R(*fn)(Args...)) {
|
||||
return {FABI_UNKNOWN, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(float)) {
|
||||
return {FABI_F80_F32, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(double)) {
|
||||
return {FABI_F80_F64, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(int16_t)) {
|
||||
return {FABI_F80_I16, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(void(*fn)(uint16_t)) {
|
||||
return {FABI_VOID_U16, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(int32_t)) {
|
||||
return {FABI_F80_I32, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(float(*fn)(X80SoftFloat)) {
|
||||
return {FABI_F32_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(double(*fn)(X80SoftFloat)) {
|
||||
return {FABI_F64_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(int16_t(*fn)(X80SoftFloat)) {
|
||||
return {FABI_I16_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(int32_t(*fn)(X80SoftFloat)) {
|
||||
return {FABI_I32_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(int64_t(*fn)(X80SoftFloat)) {
|
||||
return {FABI_I64_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(uint64_t(*fn)(X80SoftFloat, X80SoftFloat)) {
|
||||
return {FABI_I64_F80_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(X80SoftFloat)) {
|
||||
return {FABI_F80_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(X80SoftFloat, X80SoftFloat)) {
|
||||
return {FABI_F80_F80_F80, (void*)fn};
|
||||
}
|
||||
|
||||
bool InterpreterOps::GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Info) {
|
||||
uint8_t OpSize = IROp->Size;
|
||||
switch(IROp->Op) {
|
||||
case IR::OP_F80LOADFCW: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80LOADFCW>::handle);
|
||||
return true;
|
||||
}
|
||||
|
||||
case IR::OP_F80CVTTO: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTTo>();
|
||||
|
||||
switch (Op->Size) {
|
||||
case 4: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4);
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8);
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CVT: {
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4);
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8);
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CVTINT: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTInt>();
|
||||
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
*Info = GetFallbackInfo(Op->Truncate ? &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t : &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2);
|
||||
return true;
|
||||
}
|
||||
case 4: {
|
||||
*Info = GetFallbackInfo(Op->Truncate ? &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t : &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4);
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
*Info = GetFallbackInfo(Op->Truncate ? &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t : &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8);
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CMP: {
|
||||
auto Op = IROp->C<IR::IROp_F80Cmp>();
|
||||
|
||||
static constexpr std::array handlers{
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<0>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<1>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<2>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<3>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<4>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<5>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<6>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<7>,
|
||||
};
|
||||
|
||||
*Info = GetFallbackInfo(handlers[Op->Flags]);
|
||||
return true;
|
||||
}
|
||||
|
||||
case IR::OP_F80CVTTOINT: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTToInt>();
|
||||
|
||||
switch (Op->Size) {
|
||||
case 2: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2);
|
||||
return true;
|
||||
}
|
||||
case 4: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4);
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
#define COMMON_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle); \
|
||||
return true; \
|
||||
}
|
||||
|
||||
// Unary
|
||||
COMMON_X87_OP(ROUND)
|
||||
COMMON_X87_OP(F2XM1)
|
||||
COMMON_X87_OP(TAN)
|
||||
COMMON_X87_OP(SQRT)
|
||||
COMMON_X87_OP(SIN)
|
||||
COMMON_X87_OP(COS)
|
||||
COMMON_X87_OP(XTRACT_EXP)
|
||||
COMMON_X87_OP(XTRACT_SIG)
|
||||
COMMON_X87_OP(BCDSTORE)
|
||||
COMMON_X87_OP(BCDLOAD)
|
||||
|
||||
// Binary
|
||||
COMMON_X87_OP(ADD)
|
||||
COMMON_X87_OP(SUB)
|
||||
COMMON_X87_OP(MUL)
|
||||
COMMON_X87_OP(DIV)
|
||||
COMMON_X87_OP(FYL2X)
|
||||
COMMON_X87_OP(ATAN)
|
||||
COMMON_X87_OP(FPREM1)
|
||||
COMMON_X87_OP(FPREM)
|
||||
COMMON_X87_OP(SCALE)
|
||||
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
@@ -73,6 +73,8 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(ASHR, Ashr);
|
||||
REGISTER_OP(ROR, Ror);
|
||||
REGISTER_OP(EXTR, Extr);
|
||||
REGISTER_OP(PDEP, PDep);
|
||||
REGISTER_OP(PEXT, PExt);
|
||||
REGISTER_OP(LDIV, LDiv);
|
||||
REGISTER_OP(LUDIV, LUDiv);
|
||||
REGISTER_OP(LREM, LRem);
|
||||
@@ -157,6 +159,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(VLOADMEMELEMENT, VLoadMemElement);
|
||||
REGISTER_OP(VSTOREMEMELEMENT, VStoreMemElement);
|
||||
REGISTER_OP(CACHELINECLEAR, CacheLineClear);
|
||||
REGISTER_OP(CACHELINEZERO, CacheLineZero);
|
||||
|
||||
// Misc ops
|
||||
REGISTER_OP(DUMMY, NoOp);
|
||||
@@ -172,6 +175,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(GETROUNDINGMODE, GetRoundingMode);
|
||||
REGISTER_OP(SETROUNDINGMODE, SetRoundingMode);
|
||||
REGISTER_OP(INVALIDATEFLAGS, NoOp);
|
||||
REGISTER_OP(PROCESSORID, ProcessorID);
|
||||
|
||||
// Move ops
|
||||
REGISTER_OP(EXTRACTELEMENTPAIR, ExtractElementPair);
|
||||
@@ -279,6 +283,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(VAESDEC, AESDec);
|
||||
REGISTER_OP(VAESDECLAST, AESDecLast);
|
||||
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
|
||||
REGISTER_OP(CRC32, CRC32);
|
||||
|
||||
// F80 ops
|
||||
REGISTER_OP(F80LOADFCW, F80LOADFCW);
|
||||
@@ -317,205 +322,6 @@ void InterpreterOps::Op_Unhandled(FEXCore::IR::IROp_Header *IROp, IROpData *Data
|
||||
void InterpreterOps::Op_NoOp(FEXCore::IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node) {
|
||||
}
|
||||
|
||||
template<typename R, typename... Args>
|
||||
static FallbackInfo GetFallbackInfo(R(*fn)(Args...)) {
|
||||
return {FABI_UNKNOWN, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(float)) {
|
||||
return {FABI_F80_F32, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(double)) {
|
||||
return {FABI_F80_F64, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(int16_t)) {
|
||||
return {FABI_F80_I16, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(void(*fn)(uint16_t)) {
|
||||
return {FABI_VOID_U16, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(int32_t)) {
|
||||
return {FABI_F80_I32, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(float(*fn)(X80SoftFloat)) {
|
||||
return {FABI_F32_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(double(*fn)(X80SoftFloat)) {
|
||||
return {FABI_F64_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(int16_t(*fn)(X80SoftFloat)) {
|
||||
return {FABI_I16_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(int32_t(*fn)(X80SoftFloat)) {
|
||||
return {FABI_I32_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(int64_t(*fn)(X80SoftFloat)) {
|
||||
return {FABI_I64_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(uint64_t(*fn)(X80SoftFloat, X80SoftFloat)) {
|
||||
return {FABI_I64_F80_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(X80SoftFloat)) {
|
||||
return {FABI_F80_F80, (void*)fn};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(X80SoftFloat, X80SoftFloat)) {
|
||||
return {FABI_F80_F80_F80, (void*)fn};
|
||||
}
|
||||
|
||||
bool InterpreterOps::GetFallbackHandler(IR::IROp_Header *IROp, FallbackInfo *Info) {
|
||||
uint8_t OpSize = IROp->Size;
|
||||
switch(IROp->Op) {
|
||||
case IR::OP_F80LOADFCW: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80LOADFCW>::handle);
|
||||
return true;
|
||||
}
|
||||
|
||||
case IR::OP_F80CVTTO: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTTo>();
|
||||
|
||||
switch (Op->Size) {
|
||||
case 4: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4);
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8);
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CVT: {
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4);
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle8);
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CVTINT: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTInt>();
|
||||
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
*Info = GetFallbackInfo(Op->Truncate ? &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t : &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2);
|
||||
return true;
|
||||
}
|
||||
case 4: {
|
||||
*Info = GetFallbackInfo(Op->Truncate ? &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t : &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4);
|
||||
return true;
|
||||
}
|
||||
case 8: {
|
||||
*Info = GetFallbackInfo(Op->Truncate ? &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t : &FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8);
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case IR::OP_F80CMP: {
|
||||
auto Op = IROp->C<IR::IROp_F80Cmp>();
|
||||
|
||||
static constexpr std::array handlers{
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<0>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<1>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<2>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<3>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<4>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<5>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<6>,
|
||||
&FEXCore::CPU::OpHandlers<IR::OP_F80CMP>::handle<7>,
|
||||
};
|
||||
|
||||
*Info = GetFallbackInfo(handlers[Op->Flags]);
|
||||
return true;
|
||||
}
|
||||
|
||||
case IR::OP_F80CVTTOINT: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTToInt>();
|
||||
|
||||
switch (Op->Size) {
|
||||
case 2: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2);
|
||||
return true;
|
||||
}
|
||||
case 4: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4);
|
||||
return true;
|
||||
}
|
||||
default: LogMan::Msg::DFmt("Unhandled size: {}", OpSize);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
#define COMMON_X87_OP(OP) \
|
||||
case IR::OP_F80##OP: { \
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80##OP>::handle); \
|
||||
return true; \
|
||||
}
|
||||
|
||||
// Unary
|
||||
COMMON_X87_OP(ROUND)
|
||||
COMMON_X87_OP(F2XM1)
|
||||
COMMON_X87_OP(TAN)
|
||||
COMMON_X87_OP(SQRT)
|
||||
COMMON_X87_OP(SIN)
|
||||
COMMON_X87_OP(COS)
|
||||
COMMON_X87_OP(XTRACT_EXP)
|
||||
COMMON_X87_OP(XTRACT_SIG)
|
||||
COMMON_X87_OP(BCDSTORE)
|
||||
COMMON_X87_OP(BCDLOAD)
|
||||
|
||||
// Binary
|
||||
COMMON_X87_OP(ADD)
|
||||
COMMON_X87_OP(SUB)
|
||||
COMMON_X87_OP(MUL)
|
||||
COMMON_X87_OP(DIV)
|
||||
COMMON_X87_OP(FYL2X)
|
||||
COMMON_X87_OP(ATAN)
|
||||
COMMON_X87_OP(FPREM1)
|
||||
COMMON_X87_OP(FPREM)
|
||||
COMMON_X87_OP(SCALE)
|
||||
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
void InterpreterOps::InterpretIR(FEXCore::Core::InternalThreadState *Thread, uint64_t Entry, FEXCore::IR::IRListView *CurrentIR, FEXCore::Core::DebugData *DebugData) {
|
||||
volatile void *StackEntry = alloca(0);
|
||||
|
||||
|
||||
@@ -96,6 +96,8 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(Rol);
|
||||
DEF_OP(Ror);
|
||||
DEF_OP(Extr);
|
||||
DEF_OP(PDep);
|
||||
DEF_OP(PExt);
|
||||
DEF_OP(LDiv);
|
||||
DEF_OP(LUDiv);
|
||||
DEF_OP(LRem);
|
||||
@@ -180,6 +182,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(VLoadMemElement);
|
||||
DEF_OP(VStoreMemElement);
|
||||
DEF_OP(CacheLineClear);
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
///< Misc ops
|
||||
DEF_OP(EndBlock);
|
||||
@@ -190,6 +193,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(Print);
|
||||
DEF_OP(GetRoundingMode);
|
||||
DEF_OP(SetRoundingMode);
|
||||
DEF_OP(ProcessorID);
|
||||
|
||||
///< Move ops
|
||||
DEF_OP(ExtractElementPair);
|
||||
@@ -294,6 +298,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(AESDec);
|
||||
DEF_OP(AESDecLast);
|
||||
DEF_OP(AESKeyGenAssist);
|
||||
DEF_OP(CRC32);
|
||||
|
||||
///< F80 ops
|
||||
DEF_OP(F80LOADFCW);
|
||||
|
||||
@@ -264,5 +264,22 @@ DEF_OP(CacheLineClear) {
|
||||
CacheLineFlush(MemData);
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineZero) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineZero>();
|
||||
|
||||
uintptr_t MemData = *GetSrc<uintptr_t*>(Data->SSAData, Op->Addr);
|
||||
|
||||
// Force cacheline alignment
|
||||
MemData = MemData & ~(CPUIDEmu::CACHELINE_SIZE - 1);
|
||||
|
||||
using DataType = uint64_t;
|
||||
DataType *MemData64 = reinterpret_cast<DataType*>(MemData);
|
||||
|
||||
// 64-byte cache line zero
|
||||
for (size_t i = 0; i < (CPUIDEmu::CACHELINE_SIZE / sizeof(DataType)); ++i) {
|
||||
MemData64[i] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -8,6 +8,8 @@ $end_info$
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include <cstdint>
|
||||
#ifdef _M_X86_64
|
||||
#include <xmmintrin.h>
|
||||
@@ -46,7 +48,7 @@ DEF_OP(Break) {
|
||||
StopThread(Data->State);
|
||||
break;
|
||||
case FEXCore::IR::Break_InvalidInstruction:
|
||||
tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGILL);
|
||||
FHU::Syscalls::tgkill(Data->State->ThreadManager.PID, Data->State->ThreadManager.TID, SIGILL);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Break Reason: {}", Op->Reason); break;
|
||||
}
|
||||
@@ -140,6 +142,12 @@ DEF_OP(Print) {
|
||||
LOGMAN_MSG_A_FMT("Unknown value size: {}", OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(ProcessorID) {
|
||||
uint32_t CPU, CPUNode;
|
||||
FHU::Syscalls::getcpu(&CPU, &CPUNode);
|
||||
GD = (CPUNode << 12) | CPU;
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -8,6 +8,7 @@ $end_info$
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterDefines.h"
|
||||
|
||||
#include <bit>
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
@@ -1401,10 +1402,9 @@ DEF_OP(VInsScalarElement) {
|
||||
|
||||
DEF_OP(VExtractElement) {
|
||||
auto Op = IROp->C<IR::IROp_VExtractElement>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
|
||||
uint32_t SourceSize = GetOpSize(Data->CurrentIR, Op->Header.Args[0]);
|
||||
LOGMAN_THROW_A_FMT(OpSize <= 16, "OpSize is too large for VExtractElement: {}", OpSize);
|
||||
LOGMAN_THROW_A_FMT(IROp->Size <= 16, "OpSize is too large for VExtractElement: {}", IROp->Size);
|
||||
if (SourceSize == 16) {
|
||||
__uint128_t SourceMask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
|
||||
uint64_t Shift = Op->Header.ElementSize * Op->Index * 8;
|
||||
|
||||
+133
-4
@@ -8,6 +8,10 @@ $end_info$
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
#define GRD(Node) (IROp->Size <= 4 ? GetDst<RA_32>(Node) : GetDst<RA_64>(Node))
|
||||
#define GRS(Node) (IROp->Size <= 4 ? GetReg<RA_32>(Node) : GetReg<RA_64>(Node))
|
||||
|
||||
static uint64_t LUDIV(uint64_t SrcHigh, uint64_t SrcLow, uint64_t Divisor) {
|
||||
__uint128_t Source = (static_cast<__uint128_t>(SrcHigh) << 64) | SrcLow;
|
||||
__uint128_t Res = Source / Divisor;
|
||||
@@ -61,7 +65,13 @@ DEF_OP(EntrypointOffset) {
|
||||
|
||||
auto Constant = Entry + Op->Offset;
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
LoadConstant(Dst, Constant);
|
||||
uint64_t Mask = ~0ULL;
|
||||
uint8_t OpSize = IROp->Size;
|
||||
if (OpSize == 4) {
|
||||
Mask = 0xFFFF'FFFFULL;
|
||||
}
|
||||
|
||||
LoadConstant(Dst, Constant & Mask);
|
||||
}
|
||||
|
||||
DEF_OP(InlineConstant) {
|
||||
@@ -80,8 +90,6 @@ DEF_OP(CycleCounter) {
|
||||
#endif
|
||||
}
|
||||
|
||||
#define GRS(Node) (IROp->Size <= 4 ? GetReg<RA_32>(Node) : GetReg<RA_64>(Node))
|
||||
|
||||
DEF_OP(Add) {
|
||||
auto Op = IROp->C<IR::IROp_Add>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
@@ -511,6 +519,125 @@ DEF_OP(Extr) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(PDep) {
|
||||
auto Op = IROp->C<IR::IROp_PExt>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const Register Input = GRS(Op->Args(0).ID());
|
||||
const Register Mask = GRS(Op->Args(1).ID());
|
||||
const Register Dest = GRS(Node);
|
||||
|
||||
const Register ShiftedBitReg = OpSize <= 4 ? TMP1.W() : TMP1;
|
||||
const Register BitReg = OpSize <= 4 ? TMP2.W() : TMP2;
|
||||
const Register SubMaskReg = OpSize <= 4 ? TMP3.W() : TMP3;
|
||||
const Register IndexReg = OpSize <= 4 ? TMP4.W() : TMP4;
|
||||
const Register SizedZero = OpSize <= 4 ? Register{wzr} : Register{xzr};
|
||||
|
||||
const Register InputReg = OpSize <= 4 ? SRA64[0].W() : SRA64[0];
|
||||
const Register MaskReg = OpSize <= 4 ? SRA64[1].W() : SRA64[1];
|
||||
const Register DestReg = OpSize <= 4 ? SRA64[2].W() : SRA64[2];
|
||||
const auto SpillCode = 1U << InputReg.GetCode() |
|
||||
1U << MaskReg.GetCode() |
|
||||
1U << DestReg.GetCode();
|
||||
|
||||
aarch64::Label EarlyExit;
|
||||
aarch64::Label NextBit;
|
||||
aarch64::Label Done;
|
||||
|
||||
cbz(Mask, &EarlyExit);
|
||||
mov(IndexReg, SizedZero);
|
||||
|
||||
// We sadly need to spill regs for this for the time being
|
||||
// TODO: Remove when scratch registers can be allocated
|
||||
// explicitly.
|
||||
SpillStaticRegs(false, SpillCode);
|
||||
mov(InputReg, Input);
|
||||
mov(MaskReg, Mask);
|
||||
mov(DestReg, SizedZero);
|
||||
|
||||
// Main loop
|
||||
bind(&NextBit);
|
||||
rbit(ShiftedBitReg, MaskReg);
|
||||
clz(ShiftedBitReg, ShiftedBitReg);
|
||||
lsrv(BitReg, InputReg, IndexReg);
|
||||
and_(BitReg, BitReg, 1);
|
||||
sub(SubMaskReg, MaskReg, 1);
|
||||
add(IndexReg, IndexReg, 1);
|
||||
ands(MaskReg, MaskReg, SubMaskReg);
|
||||
lslv(ShiftedBitReg, BitReg, ShiftedBitReg);
|
||||
orr(DestReg, DestReg, ShiftedBitReg);
|
||||
b(&NextBit, Condition::ne);
|
||||
// Store result in a temp so it doesn't get clobbered.
|
||||
// and restore it after the re-fill below.
|
||||
mov(IndexReg, DestReg);
|
||||
// Restore our registers before leaving
|
||||
// TODO: Also remove along with above TODO.
|
||||
FillStaticRegs(false, SpillCode);
|
||||
mov(Dest, IndexReg);
|
||||
b(&Done);
|
||||
|
||||
// Early exit
|
||||
bind(&EarlyExit);
|
||||
mov(Dest, SizedZero);
|
||||
|
||||
// All done with nothing to do.
|
||||
bind(&Done);
|
||||
}
|
||||
|
||||
DEF_OP(PExt) {
|
||||
auto Op = IROp->C<IR::IROp_PExt>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const Register Input = GRS(Op->Args(0).ID());
|
||||
const Register Mask = GRS(Op->Args(1).ID());
|
||||
const Register Dest = GRS(Node);
|
||||
|
||||
const Register MaskReg = OpSize <= 4 ? TMP1.W() : TMP1;
|
||||
const Register BitReg = OpSize <= 4 ? TMP2.W() : TMP2;
|
||||
const Register SubMaskReg = OpSize <= 4 ? TMP3.W() : TMP3;
|
||||
const Register Offset = OpSize <= 4 ? TMP4.W() : TMP4;
|
||||
const Register SizedZero = OpSize <= 4 ? Register{wzr} : Register{xzr};
|
||||
|
||||
aarch64::Label EarlyExit;
|
||||
aarch64::Label NextBit;
|
||||
aarch64::Label Done;
|
||||
|
||||
cbz(Mask, &EarlyExit);
|
||||
mov(MaskReg, Mask);
|
||||
mov(Offset, SizedZero);
|
||||
|
||||
// We sadly need to spill a reg for this for the time being
|
||||
// TODO: Remove when scratch registers can be allocated
|
||||
// explicitly.
|
||||
SpillStaticRegs(false, 1U << Mask.GetCode());
|
||||
mov(Mask, SizedZero);
|
||||
|
||||
// Main loop
|
||||
bind(&NextBit);
|
||||
rbit(BitReg, MaskReg);
|
||||
clz(BitReg, BitReg);
|
||||
sub(SubMaskReg, MaskReg, 1);
|
||||
ands(MaskReg, SubMaskReg, MaskReg);
|
||||
lsrv(BitReg, Input, BitReg);
|
||||
and_(BitReg, BitReg, 1);
|
||||
lslv(BitReg, BitReg, Offset);
|
||||
add(Offset, Offset, 1);
|
||||
orr(Mask, BitReg, Mask);
|
||||
b(&NextBit, Condition::ne);
|
||||
mov(Dest, Mask);
|
||||
// Restore our mask register before leaving
|
||||
// TODO: Also remove along with above TODO.
|
||||
FillStaticRegs(false, 1U << Mask.GetCode());
|
||||
b(&Done);
|
||||
|
||||
// Early exit
|
||||
bind(&EarlyExit);
|
||||
mov(Dest, SizedZero);
|
||||
|
||||
// All done with nothing to do.
|
||||
bind(&Done);
|
||||
}
|
||||
|
||||
DEF_OP(LDiv) {
|
||||
auto Op = IROp->C<IR::IROp_LDiv>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
@@ -916,7 +1043,7 @@ Condition MapSelectCC(IR::CondClassType Cond) {
|
||||
case FEXCore::IR::COND_FLU: return Condition::lt;
|
||||
case FEXCore::IR::COND_FGE: return Condition::ge;
|
||||
case FEXCore::IR::COND_FLEU:return Condition::le;
|
||||
case FEXCore::IR::COND_FGT: return Condition::hi;
|
||||
case FEXCore::IR::COND_FGT: return Condition::gt;
|
||||
case FEXCore::IR::COND_FU: return Condition::vs;
|
||||
case FEXCore::IR::COND_FNU: return Condition::vc;
|
||||
case FEXCore::IR::COND_VS:
|
||||
@@ -1099,6 +1226,8 @@ void Arm64JITCore::RegisterALUHandlers() {
|
||||
REGISTER_OP(ASHR, Ashr);
|
||||
REGISTER_OP(ROR, Ror);
|
||||
REGISTER_OP(EXTR, Extr);
|
||||
REGISTER_OP(PDEP, PDep);
|
||||
REGISTER_OP(PEXT, PExt);
|
||||
REGISTER_OP(LDIV, LDiv);
|
||||
REGISTER_OP(LUDIV, LUDiv);
|
||||
REGISTER_OP(LREM, LRem);
|
||||
|
||||
+13
-13
@@ -19,7 +19,7 @@ DEF_OP(CASPair) {
|
||||
auto Desired = GetSrcPair<RA_64>(Op->Header.Args[1].ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[2].ID());
|
||||
|
||||
if (SupportsAtomics) {
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mov(TMP3, Expected.first);
|
||||
mov(TMP4, Expected.second);
|
||||
|
||||
@@ -110,7 +110,7 @@ DEF_OP(CAS) {
|
||||
auto Desired = GetReg<RA_64>(Op->Header.Args[1].ID());
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[2].ID());
|
||||
|
||||
if (SupportsAtomics) {
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mov(TMP2, Expected);
|
||||
switch (OpSize) {
|
||||
case 1: casalb(TMP2.W(), Desired.W(), MemOperand(MemSrc)); break;
|
||||
@@ -218,7 +218,7 @@ DEF_OP(AtomicAdd) {
|
||||
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (SupportsAtomics) {
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
switch (IROp->Size) {
|
||||
case 1: staddlb(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 2: staddlh(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
@@ -276,7 +276,7 @@ DEF_OP(AtomicSub) {
|
||||
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (SupportsAtomics) {
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
neg(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
switch (IROp->Size) {
|
||||
case 1: staddlb(TMP2.W(), MemOperand(MemSrc)); break;
|
||||
@@ -335,7 +335,7 @@ DEF_OP(AtomicAnd) {
|
||||
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (SupportsAtomics) {
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mvn(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
switch (IROp->Size) {
|
||||
case 1: stclrlb(TMP2.W(), MemOperand(MemSrc)); break;
|
||||
@@ -394,7 +394,7 @@ DEF_OP(AtomicOr) {
|
||||
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (SupportsAtomics) {
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
switch (IROp->Size) {
|
||||
case 1: stsetlb(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 2: stsetlh(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
@@ -452,7 +452,7 @@ DEF_OP(AtomicXor) {
|
||||
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (SupportsAtomics) {
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
switch (IROp->Size) {
|
||||
case 1: steorlb(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
case 2: steorlh(GetReg<RA_32>(Op->Header.Args[1].ID()), MemOperand(MemSrc)); break;
|
||||
@@ -510,7 +510,7 @@ DEF_OP(AtomicSwap) {
|
||||
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (SupportsAtomics) {
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mov(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
switch (IROp->Size) {
|
||||
case 1: swplb(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
@@ -568,7 +568,7 @@ DEF_OP(AtomicFetchAdd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchAdd>();
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (SupportsAtomics) {
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
switch (IROp->Size) {
|
||||
case 1: ldaddalb(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: ldaddalh(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
@@ -629,7 +629,7 @@ DEF_OP(AtomicFetchSub) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchSub>();
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (SupportsAtomics) {
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
neg(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
switch (IROp->Size) {
|
||||
case 1: ldaddalb(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
@@ -691,7 +691,7 @@ DEF_OP(AtomicFetchAnd) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchAnd>();
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (SupportsAtomics) {
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
mvn(TMP2, GetReg<RA_64>(Op->Header.Args[1].ID()));
|
||||
switch (IROp->Size) {
|
||||
case 1: ldclralb(TMP2.W(), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
@@ -753,7 +753,7 @@ DEF_OP(AtomicFetchOr) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchOr>();
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (SupportsAtomics) {
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
switch (IROp->Size) {
|
||||
case 1: ldsetalb(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: ldsetalh(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
@@ -814,7 +814,7 @@ DEF_OP(AtomicFetchXor) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicFetchXor>();
|
||||
auto MemSrc = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (SupportsAtomics) {
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
switch (IROp->Size) {
|
||||
case 1: ldeoralb(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
case 2: ldeoralh(GetReg<RA_32>(Op->Header.Args[1].ID()), GetReg<RA_32>(Node), MemOperand(MemSrc)); break;
|
||||
|
||||
@@ -318,7 +318,7 @@ DEF_OP(InlineSyscall) {
|
||||
ldr(RegArgs[i].W(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RBX])));
|
||||
}
|
||||
else {
|
||||
uxtw(RegArgs[i], Reg);
|
||||
uxtw(RegArgs[i].W(), Reg);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -331,7 +331,7 @@ DEF_OP(InlineSyscall) {
|
||||
mov(RegArgs[i], GetReg<RA_64>(Op->Header.Args[i].ID()));
|
||||
}
|
||||
else {
|
||||
uxtw(RegArgs[i], GetReg<RA_32>(Op->Header.Args[i].ID()));
|
||||
uxtw(RegArgs[i], GetReg<RA_64>(Op->Header.Args[i].ID()));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -354,7 +354,7 @@ DEF_OP(InlineSyscall) {
|
||||
mov(GetReg<RA_64>(Node), x0);
|
||||
}
|
||||
else {
|
||||
uxtw(GetReg<RA_64>(Node), w0);
|
||||
uxtw(GetReg<RA_64>(Node), x0);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -39,11 +39,11 @@ DEF_OP(VCastFromGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VCastFromGPR>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1:
|
||||
uxtb(TMP1.W(), GetReg<RA_32>(Op->Header.Args[0].ID()).W());
|
||||
uxtb(TMP1.W(), GetReg<RA_32>(Op->Header.Args[0].ID()));
|
||||
fmov(GetDst(Node).S(), TMP1.W());
|
||||
break;
|
||||
case 2:
|
||||
uxth(TMP1.W(), GetReg<RA_32>(Op->Header.Args[0].ID()).W());
|
||||
uxth(TMP1.W(), GetReg<RA_32>(Op->Header.Args[0].ID()));
|
||||
fmov(GetDst(Node).S(), TMP1.W());
|
||||
break;
|
||||
case 4:
|
||||
|
||||
@@ -54,7 +54,7 @@ DEF_OP(AESDecLast) {
|
||||
DEF_OP(AESKeyGenAssist) {
|
||||
auto Op = IROp->C<IR::IROp_VAESKeyGenAssist>();
|
||||
|
||||
aarch64::Label Constant;
|
||||
aarch64::Literal ConstantLiteral (0x0C030609'0306090CULL, 0x040B0E01'0B0E0104ULL);
|
||||
aarch64::Label PastConstant;
|
||||
|
||||
// Do a "regular" AESE step
|
||||
@@ -63,8 +63,7 @@ DEF_OP(AESKeyGenAssist) {
|
||||
aese(VTMP1.V16B(), VTMP2.V16B());
|
||||
|
||||
// Do a table shuffle to undo ShiftRows
|
||||
adr(TMP1.X(), &Constant);
|
||||
ldr(VTMP3, MemOperand(TMP1.X()));
|
||||
ldr(VTMP3, &ConstantLiteral);
|
||||
|
||||
// Now EOR in the RCON
|
||||
if (Op->RCON) {
|
||||
@@ -80,14 +79,29 @@ DEF_OP(AESKeyGenAssist) {
|
||||
}
|
||||
|
||||
b(&PastConstant);
|
||||
bind(&Constant);
|
||||
dc32(0x0B0E0104);
|
||||
dc32(0x040B0E01);
|
||||
dc32(0x0306090C);
|
||||
dc32(0x0C030609);
|
||||
place(&ConstantLiteral);
|
||||
bind(&PastConstant);
|
||||
}
|
||||
|
||||
DEF_OP(CRC32) {
|
||||
auto Op = IROp->C<IR::IROp_CRC32>();
|
||||
switch (Op->SrcSize) {
|
||||
case 1:
|
||||
crc32cb(GetReg<RA_32>(Node), GetReg<RA_32>(Op->Src1.ID()), GetReg<RA_32>(Op->Src2.ID()));
|
||||
break;
|
||||
case 2:
|
||||
crc32ch(GetReg<RA_32>(Node), GetReg<RA_32>(Op->Src1.ID()), GetReg<RA_32>(Op->Src2.ID()));
|
||||
break;
|
||||
case 4:
|
||||
crc32cw(GetReg<RA_32>(Node), GetReg<RA_32>(Op->Src1.ID()), GetReg<RA_32>(Op->Src2.ID()));
|
||||
break;
|
||||
case 8:
|
||||
crc32cx(GetReg<RA_32>(Node), GetReg<RA_32>(Op->Src1.ID()), GetReg<RA_64>(Op->Src2.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown CRC32 size: {}", Op->SrcSize);
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterEncryptionHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
@@ -97,7 +111,7 @@ void Arm64JITCore::RegisterEncryptionHandlers() {
|
||||
REGISTER_OP(VAESDEC, AESDec);
|
||||
REGISTER_OP(VAESDECLAST, AESDecLast);
|
||||
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
|
||||
|
||||
REGISTER_OP(CRC32, CRC32);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
@@ -333,7 +333,7 @@ void Arm64JITCore::FreeCodeBuffer(CodeBuffer Buffer) {
|
||||
}
|
||||
|
||||
Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, bool CompileThread)
|
||||
: Arm64Emitter(0)
|
||||
: Arm64Emitter(ctx, 0)
|
||||
, CTX {ctx}
|
||||
, ThreadState {Thread} {
|
||||
{
|
||||
@@ -582,7 +582,12 @@ bool Arm64JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode,
|
||||
if (OpHeader->Op == IR::IROps::OP_INLINEENTRYPOINTOFFSET) {
|
||||
auto Op = OpHeader->C<IR::IROp_InlineEntrypointOffset>();
|
||||
if (Value) {
|
||||
*Value = Entry + Op->Offset;
|
||||
uint64_t Mask = ~0ULL;
|
||||
uint8_t OpSize = OpHeader->Size;
|
||||
if (OpSize == 4) {
|
||||
Mask = 0xFFFF'FFFFULL;
|
||||
}
|
||||
*Value = (Entry + Op->Offset) & Mask;
|
||||
}
|
||||
return true;
|
||||
} else {
|
||||
|
||||
@@ -176,6 +176,7 @@ private:
|
||||
struct CompilerSharedData {
|
||||
uint64_t SignalReturnInstruction{};
|
||||
uint64_t UnimplementedInstructionAddress{};
|
||||
uint64_t OverflowExceptionInstructionAddress{};
|
||||
|
||||
uint32_t *SignalHandlerRefCounterPtr{};
|
||||
FEXCore::CPU::Dispatcher *Dispatcher{};
|
||||
@@ -233,6 +234,8 @@ private:
|
||||
DEF_OP(Rol);
|
||||
DEF_OP(Ror);
|
||||
DEF_OP(Extr);
|
||||
DEF_OP(PDep);
|
||||
DEF_OP(PExt);
|
||||
DEF_OP(LDiv);
|
||||
DEF_OP(LUDiv);
|
||||
DEF_OP(LRem);
|
||||
@@ -302,7 +305,7 @@ private:
|
||||
DEF_OP(GetHostFlag);
|
||||
|
||||
///< Memory ops
|
||||
DEF_OP(LoadContext);
|
||||
DEF_OP(LoadContext);
|
||||
DEF_OP(StoreContext);
|
||||
DEF_OP(LoadRegister);
|
||||
DEF_OP(StoreRegister);
|
||||
@@ -321,6 +324,7 @@ private:
|
||||
DEF_OP(VLoadMemElement);
|
||||
DEF_OP(VStoreMemElement);
|
||||
DEF_OP(CacheLineClear);
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
///< Misc ops
|
||||
DEF_OP(EndBlock);
|
||||
@@ -331,6 +335,7 @@ private:
|
||||
DEF_OP(Print);
|
||||
DEF_OP(GetRoundingMode);
|
||||
DEF_OP(SetRoundingMode);
|
||||
DEF_OP(ProcessorID);
|
||||
|
||||
///< Move ops
|
||||
DEF_OP(ExtractElementPair);
|
||||
@@ -437,6 +442,7 @@ private:
|
||||
DEF_OP(AESDec);
|
||||
DEF_OP(AESDecLast);
|
||||
DEF_OP(AESKeyGenAssist);
|
||||
DEF_OP(CRC32);
|
||||
#undef DEF_OP
|
||||
};
|
||||
|
||||
|
||||
@@ -623,7 +623,7 @@ DEF_OP(LoadMemTSO) {
|
||||
LOGMAN_MSG_A_FMT("LoadMemTSO: No offset allowed");
|
||||
}
|
||||
|
||||
if (SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (IROp->Size == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
auto Dst = GetReg<RA_64>(Node);
|
||||
@@ -939,13 +939,35 @@ DEF_OP(CacheLineClear) {
|
||||
// Clear dcache only
|
||||
// icache doesn't matter here since the guest application shouldn't be calling clflush on JIT code.
|
||||
mov(TMP1, MemReg);
|
||||
for (size_t i = 0; i < std::max(1U, DCacheLineSize / 64U); ++i) {
|
||||
for (size_t i = 0; i < std::max(1U, CTX->HostFeatures.DCacheLineSize / 64U); ++i) {
|
||||
dc(DataCacheOp::CVAU, TMP1);
|
||||
add(TMP1, TMP1, DCacheLineSize);
|
||||
add(TMP1, TMP1, CTX->HostFeatures.DCacheLineSize);
|
||||
}
|
||||
dsb(InnerShareable, BarrierAll);
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineZero) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineZero>();
|
||||
|
||||
auto MemReg = GetReg<RA_64>(Op->Header.Args[0].ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsCLZERO) {
|
||||
// We can use this instruction directly
|
||||
dc(DataCacheOp::ZVA, MemReg);
|
||||
}
|
||||
else {
|
||||
// We must walk the cacheline ourselves
|
||||
// Force cacheline alignment
|
||||
and_(TMP1, MemReg, ~(CPUIDEmu::CACHELINE_SIZE - 1));
|
||||
// This will end up being four STPs
|
||||
// Depending on uarch it could be slightly more efficient in instructions emitted
|
||||
// and uops to use vector pair STP, but we want the non-temporal bit specifically here
|
||||
for (size_t i = 0; i < CPUIDEmu::CACHELINE_SIZE; i += 16) {
|
||||
stnp(xzr, xzr, MemOperand(TMP1, i, Offset));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterMemoryHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
@@ -972,6 +994,7 @@ void Arm64JITCore::RegisterMemoryHandlers() {
|
||||
REGISTER_OP(VLOADMEMELEMENT, VLoadMemElement);
|
||||
REGISTER_OP(VSTOREMEMELEMENT, VStoreMemElement);
|
||||
REGISTER_OP(CACHELINECLEAR, CacheLineClear);
|
||||
REGISTER_OP(CACHELINEZERO, CacheLineZero);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
@@ -40,9 +40,13 @@ DEF_OP(Break) {
|
||||
switch (Op->Reason) {
|
||||
case FEXCore::IR::Break_Unimplemented: // Hard fault
|
||||
case FEXCore::IR::Break_Interrupt: // Guest ud2
|
||||
case FEXCore::IR::Break_Overflow: // overflow
|
||||
hlt(4);
|
||||
break;
|
||||
case FEXCore::IR::Break_Overflow: // overflow
|
||||
ResetStack();
|
||||
LoadConstant(TMP1, ThreadSharedData.Dispatcher->OverflowExceptionInstructionAddress);
|
||||
br(TMP1);
|
||||
break;
|
||||
case FEXCore::IR::Break_Halt: { // HLT
|
||||
// Time to quit
|
||||
// Set our stack to the starting stack location
|
||||
@@ -154,6 +158,58 @@ DEF_OP(Print) {
|
||||
PopDynamicRegsAndLR();
|
||||
}
|
||||
|
||||
DEF_OP(ProcessorID) {
|
||||
// We always need to spill x8 since we can't know if it is live at this SSA location
|
||||
uint32_t SpillMask = 1U << 8;
|
||||
|
||||
// Ordering is incredibly important here
|
||||
// We must spill any overlapping registers first THEN claim we are in a syscall without invalidating state at all
|
||||
// Only spill the registers that intersect with our usage
|
||||
SpillStaticRegs(false, SpillMask);
|
||||
|
||||
// Now that we are spilled, store in the state that we are in a syscall
|
||||
// Still without overwriting registers that matter
|
||||
// 16bit LoadConstant to be a single instruction
|
||||
// We must always spill at least one register (x8) so this value always has a bit set
|
||||
// This gives the signal handler a value to check to see if we are in a syscall at all
|
||||
LoadConstant(x0, SpillMask & 0xFFFF);
|
||||
str(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo)));
|
||||
|
||||
// Allocate some temporary space for storing the uint32_t CPU and Node IDs
|
||||
sub(sp, sp, 16);
|
||||
|
||||
// Load the getcpu syscall number
|
||||
LoadConstant(x8, SYS_getcpu);
|
||||
|
||||
// CPU pointer in x0
|
||||
add(x0, sp, 0);
|
||||
// Node in x1
|
||||
add(x1, sp, 4);
|
||||
|
||||
svc(0);
|
||||
// On updated signal mask we can receive a signal RIGHT HERE
|
||||
|
||||
// Load the values returned by the kernel
|
||||
ldp(w0, w1, MemOperand(sp));
|
||||
// Deallocate stack space
|
||||
sub(sp, sp, 16);
|
||||
|
||||
// Now that we are done in the syscall we need to carefully peel back the state
|
||||
// First unspill the registers from before
|
||||
FillStaticRegs(false, SpillMask);
|
||||
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(xzr, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo)));
|
||||
|
||||
|
||||
// Now store the result in the destination in the expected format
|
||||
// uint32_t Res = (node << 12) | cpu;
|
||||
// CPU is in w0
|
||||
// Node is in w1
|
||||
orr(GetReg<RA_64>(Node), x0, Operand(x1, LSL, 12));
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterMiscHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
@@ -170,6 +226,7 @@ void Arm64JITCore::RegisterMiscHandlers() {
|
||||
REGISTER_OP(GETROUNDINGMODE, GetRoundingMode);
|
||||
REGISTER_OP(SETROUNDINGMODE, SetRoundingMode);
|
||||
REGISTER_OP(INVALIDATEFLAGS, NoOp);
|
||||
REGISTER_OP(PROCESSORID, ProcessorID);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
@@ -45,7 +45,13 @@ DEF_OP(EntrypointOffset) {
|
||||
auto Op = IROp->C<IR::IROp_EntrypointOffset>();
|
||||
|
||||
auto Constant = Entry + Op->Offset;
|
||||
mov(GetDst<RA_64>(Node), Constant);
|
||||
uint64_t Mask = ~0ULL;
|
||||
uint8_t OpSize = IROp->Size;
|
||||
if (OpSize == 4) {
|
||||
Mask = 0xFFFF'FFFFULL;
|
||||
}
|
||||
|
||||
mov(GetDst<RA_64>(Node), Constant & Mask);
|
||||
}
|
||||
|
||||
DEF_OP(InlineConstant) {
|
||||
@@ -671,6 +677,36 @@ DEF_OP(Extr) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(PDep) {
|
||||
const auto Op = IROp->C<IR::IROp_PExt>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Input = GRS(Op->Args(0).ID());
|
||||
const auto Mask = GRS(Op->Args(1).ID());
|
||||
const auto Dest = GRD(Node);
|
||||
|
||||
if (OpSize == 4) {
|
||||
pdep(Dest.cvt32(), Input.cvt32(), Mask.cvt32());
|
||||
} else {
|
||||
pdep(Dest.cvt64(), Input.cvt64(), Mask.cvt64());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(PExt) {
|
||||
const auto Op = IROp->C<IR::IROp_PExt>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Input = GRS(Op->Args(0).ID());
|
||||
const auto Mask = GRS(Op->Args(1).ID());
|
||||
const auto Dest = GRD(Node);
|
||||
|
||||
if (OpSize == 4) {
|
||||
pext(Dest.cvt32(), Input.cvt32(), Mask.cvt32());
|
||||
} else {
|
||||
pext(Dest.cvt64(), Input.cvt64(), Mask.cvt64());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LDiv) {
|
||||
auto Op = IROp->C<IR::IROp_LDiv>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
@@ -1248,6 +1284,8 @@ void X86JITCore::RegisterALUHandlers() {
|
||||
REGISTER_OP(ASHR, Ashr);
|
||||
REGISTER_OP(ROR, Ror);
|
||||
REGISTER_OP(EXTR, Extr);
|
||||
REGISTER_OP(PDEP, PDep);
|
||||
REGISTER_OP(PEXT, PExt);
|
||||
REGISTER_OP(LDIV, LDiv);
|
||||
REGISTER_OP(LUDIV, LUDiv);
|
||||
REGISTER_OP(LREM, LRem);
|
||||
|
||||
@@ -45,6 +45,36 @@ DEF_OP(AESKeyGenAssist) {
|
||||
vaeskeygenassist(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), Op->RCON);
|
||||
}
|
||||
|
||||
DEF_OP(CRC32) {
|
||||
auto Op = IROp->C<IR::IROp_CRC32>();
|
||||
switch (IROp->Size) {
|
||||
case 4:
|
||||
mov(TMP1, GetSrc<RA_32>(Op->Header.Args[1].ID()));
|
||||
mov(GetDst<RA_32>(Node), GetSrc<RA_32>(Op->Header.Args[0].ID()));
|
||||
break;
|
||||
case 8:
|
||||
mov(TMP1, GetSrc<RA_64>(Op->Header.Args[1].ID()));
|
||||
mov(GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Header.Args[0].ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown CRC32 size: {}", IROp->Size);
|
||||
}
|
||||
|
||||
switch (Op->SrcSize) {
|
||||
case 1:
|
||||
crc32(GetDst<RA_32>(Node).cvt32(), TMP1.cvt8());
|
||||
break;
|
||||
case 2:
|
||||
crc32(GetDst<RA_32>(Node).cvt32(), TMP1.cvt16());
|
||||
break;
|
||||
case 4:
|
||||
crc32(GetDst<RA_32>(Node).cvt32(), TMP1.cvt32());
|
||||
break;
|
||||
case 8:
|
||||
crc32(GetDst<RA_64>(Node).cvt64(), TMP1.cvt64());
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void X86JITCore::RegisterEncryptionHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
|
||||
@@ -54,7 +84,7 @@ void X86JITCore::RegisterEncryptionHandlers() {
|
||||
REGISTER_OP(VAESDEC, AESDec);
|
||||
REGISTER_OP(VAESDECLAST, AESDecLast);
|
||||
REGISTER_OP(VAESKEYGENASSIST, AESKeyGenAssist);
|
||||
|
||||
REGISTER_OP(CRC32, CRC32);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
@@ -358,6 +358,7 @@ X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalTh
|
||||
ThreadSharedData.SignalHandlerRefCounterPtr = &Dispatcher->SignalHandlerRefCounter;
|
||||
ThreadSharedData.SignalHandlerReturnAddress = Dispatcher->SignalHandlerReturnAddress;
|
||||
ThreadSharedData.UnimplementedInstructionAddress = Dispatcher->UnimplementedInstructionAddress;
|
||||
ThreadSharedData.OverflowExceptionInstructionAddress = Dispatcher->OverflowExceptionInstructionAddress;
|
||||
|
||||
ThreadSharedData.Dispatcher = Dispatcher.get();
|
||||
|
||||
@@ -552,7 +553,12 @@ bool X86JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, u
|
||||
if (OpHeader->Op == IR::IROps::OP_INLINEENTRYPOINTOFFSET) {
|
||||
auto Op = OpHeader->C<IR::IROp_InlineEntrypointOffset>();
|
||||
if (Value) {
|
||||
*Value = Entry + Op->Offset;
|
||||
uint64_t Mask = ~0ULL;
|
||||
uint8_t OpSize = OpHeader->Size;
|
||||
if (OpSize == 4) {
|
||||
Mask = 0xFFFF'FFFFULL;
|
||||
}
|
||||
*Value = (Entry + Op->Offset) & Mask;
|
||||
}
|
||||
return true;
|
||||
} else {
|
||||
|
||||
@@ -167,7 +167,7 @@ private:
|
||||
struct CompilerSharedData {
|
||||
uint64_t SignalHandlerReturnAddress{};
|
||||
uint64_t UnimplementedInstructionAddress{};
|
||||
|
||||
uint64_t OverflowExceptionInstructionAddress{};
|
||||
|
||||
uint32_t *SignalHandlerRefCounterPtr{};
|
||||
FEXCore::CPU::Dispatcher *Dispatcher{};
|
||||
@@ -233,6 +233,8 @@ private:
|
||||
DEF_OP(Rol);
|
||||
DEF_OP(Ror);
|
||||
DEF_OP(Extr);
|
||||
DEF_OP(PDep);
|
||||
DEF_OP(PExt);
|
||||
DEF_OP(LDiv);
|
||||
DEF_OP(LUDiv);
|
||||
DEF_OP(LRem);
|
||||
@@ -315,6 +317,7 @@ private:
|
||||
DEF_OP(VLoadMemElement);
|
||||
DEF_OP(VStoreMemElement);
|
||||
DEF_OP(CacheLineClear);
|
||||
DEF_OP(CacheLineZero);
|
||||
|
||||
///< Misc ops
|
||||
DEF_OP(EndBlock);
|
||||
@@ -325,6 +328,7 @@ private:
|
||||
DEF_OP(Print);
|
||||
DEF_OP(GetRoundingMode);
|
||||
DEF_OP(SetRoundingMode);
|
||||
DEF_OP(ProcessorID);
|
||||
|
||||
///< Move ops
|
||||
DEF_OP(ExtractElementPair);
|
||||
@@ -430,6 +434,7 @@ private:
|
||||
DEF_OP(AESDec);
|
||||
DEF_OP(AESDecLast);
|
||||
DEF_OP(AESKeyGenAssist);
|
||||
DEF_OP(CRC32);
|
||||
#undef DEF_OP
|
||||
};
|
||||
|
||||
|
||||
@@ -595,6 +595,23 @@ DEF_OP(CacheLineClear) {
|
||||
clflush(ptr [MemReg]);
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineZero) {
|
||||
auto Op = IROp->C<IR::IROp_CacheLineZero>();
|
||||
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
|
||||
// Align by cacheline
|
||||
mov (TMP1, CPUIDEmu::CACHELINE_SIZE - 1);
|
||||
andn(TMP1, TMP1, MemReg.cvt64());
|
||||
xor_(TMP2, TMP2);
|
||||
|
||||
using DataType = uint64_t;
|
||||
// 64-byte cache line zero
|
||||
for (size_t i = 0; i < CPUIDEmu::CACHELINE_SIZE; i += sizeof(DataType)) {
|
||||
mov (qword [TMP1 + i], TMP2);
|
||||
}
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void X86JITCore::RegisterMemoryHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
|
||||
@@ -615,6 +632,7 @@ void X86JITCore::RegisterMemoryHandlers() {
|
||||
REGISTER_OP(VLOADMEMELEMENT, VLoadMemElement);
|
||||
REGISTER_OP(VSTOREMEMELEMENT, VStoreMemElement);
|
||||
REGISTER_OP(CACHELINECLEAR, CacheLineClear);
|
||||
REGISTER_OP(CACHELINEZERO, CacheLineZero);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
@@ -49,9 +49,13 @@ DEF_OP(Break) {
|
||||
switch (Op->Reason) {
|
||||
case FEXCore::IR::Break_Unimplemented: // Hard fault
|
||||
case FEXCore::IR::Break_Interrupt: // Guest ud2
|
||||
case FEXCore::IR::Break_Overflow: // overflow
|
||||
ud2();
|
||||
break;
|
||||
case FEXCore::IR::Break_Overflow: // overflow
|
||||
// Need to be outside of JIT cache space to ensure cache clearing correctness
|
||||
mov(TMP1, ThreadSharedData.Dispatcher->OverflowExceptionInstructionAddress);
|
||||
jmp(TMP1);
|
||||
break;
|
||||
case FEXCore::IR::Break_Halt: { // HLT
|
||||
// Time to quit
|
||||
// Set our stack to the starting stack location
|
||||
@@ -155,6 +159,13 @@ DEF_OP(Print) {
|
||||
PopRegs();
|
||||
}
|
||||
|
||||
DEF_OP(ProcessorID) {
|
||||
// Cyclecounter in EDX:EAX
|
||||
// IA32_TSC_AUX in ECX
|
||||
rdtscp();
|
||||
mov (GetDst<RA_32>(Node), ecx);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void X86JITCore::RegisterMiscHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
|
||||
@@ -171,6 +182,7 @@ void X86JITCore::RegisterMiscHandlers() {
|
||||
REGISTER_OP(GETROUNDINGMODE, GetRoundingMode);
|
||||
REGISTER_OP(SETROUNDINGMODE, SetRoundingMode);
|
||||
REGISTER_OP(INVALIDATEFLAGS, NoOp);
|
||||
REGISTER_OP(PROCESSORID, ProcessorID);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
+307
-258
File diff suppressed because it is too large.
Load diff
+569
-20
@@ -14,6 +14,7 @@
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <fmt/format.h>
|
||||
#include <map>
|
||||
#include <stddef.h>
|
||||
#include <utility>
|
||||
@@ -35,6 +36,37 @@ enum class SelectionFlag {
|
||||
};
|
||||
|
||||
public:
|
||||
enum class FlagsGenerationType : uint8_t {
|
||||
TYPE_NONE,
|
||||
TYPE_ADC,
|
||||
TYPE_SBB,
|
||||
TYPE_SUB,
|
||||
TYPE_ADD,
|
||||
TYPE_MUL,
|
||||
TYPE_UMUL,
|
||||
TYPE_LOGICAL,
|
||||
TYPE_LSHL,
|
||||
TYPE_LSHLI,
|
||||
TYPE_LSHR,
|
||||
TYPE_LSHRI,
|
||||
TYPE_ASHR,
|
||||
TYPE_ASHRI,
|
||||
TYPE_ROR,
|
||||
TYPE_RORI,
|
||||
TYPE_ROL,
|
||||
TYPE_ROLI,
|
||||
TYPE_FCMP,
|
||||
TYPE_BEXTR,
|
||||
TYPE_BLSI,
|
||||
TYPE_BLSMSK,
|
||||
TYPE_BLSR,
|
||||
TYPE_POPCOUNT,
|
||||
TYPE_BZHI,
|
||||
TYPE_TZCNT,
|
||||
TYPE_LZCNT,
|
||||
TYPE_BITSELECT,
|
||||
};
|
||||
|
||||
SelectionFlag flagsOp{};
|
||||
uint8_t flagsOpSize{};
|
||||
OrderedNode* flagsOpDest{};
|
||||
@@ -87,6 +119,9 @@ public:
|
||||
// cmp qword [rdi-8], 0
|
||||
// jne .label
|
||||
if (LastOp && !BlockSetRIP) {
|
||||
// Calculate flags first
|
||||
CalculateDeferredFlags();
|
||||
|
||||
auto it = JumpTargets.find(NextRIP);
|
||||
if (it == JumpTargets.end()) {
|
||||
|
||||
@@ -101,6 +136,11 @@ public:
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
if (LastOp) {
|
||||
LOGMAN_THROW_A_FMT(IsDeferredFlagsStored(), "FinishOp: Deferred flags weren't generated at end of block");
|
||||
}
|
||||
|
||||
BlockSetRIP = false;
|
||||
|
||||
return false;
|
||||
@@ -338,6 +378,8 @@ public:
|
||||
void BMI2Shift(OpcodeArgs);
|
||||
void BZHI(OpcodeArgs);
|
||||
void MULX(OpcodeArgs);
|
||||
void PDEP(OpcodeArgs);
|
||||
void PEXT(OpcodeArgs);
|
||||
void RORX(OpcodeArgs);
|
||||
|
||||
// ADX Ops
|
||||
@@ -469,6 +511,8 @@ public:
|
||||
void FenceOp(OpcodeArgs);
|
||||
|
||||
void StoreFenceOrCLFlush(OpcodeArgs);
|
||||
void CLZeroOp(OpcodeArgs);
|
||||
void RDTSCPOp(OpcodeArgs);
|
||||
|
||||
void PSADBW(OpcodeArgs);
|
||||
|
||||
@@ -496,6 +540,8 @@ public:
|
||||
|
||||
void MPSADBWOp(OpcodeArgs);
|
||||
|
||||
void CRC32(OpcodeArgs);
|
||||
|
||||
void UnimplementedOp(OpcodeArgs);
|
||||
|
||||
void InvalidOp(OpcodeArgs);
|
||||
@@ -515,7 +561,7 @@ private:
|
||||
|
||||
OrderedNode *AppendSegmentOffset(OrderedNode *Value, uint32_t Flags, uint32_t DefaultPrefix = 0, bool Override = false);
|
||||
|
||||
OrderedNode *GetDynamicPC(FEXCore::X86Tables::DecodedOp const& Op, int64_t Offset = 0);
|
||||
OrderedNode *GetRelocatedPC(FEXCore::X86Tables::DecodedOp const& Op, int64_t Offset = 0);
|
||||
OrderedNode *LoadSource(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint32_t Flags, int8_t Align, bool LoadData = true, bool ForceLoad = false);
|
||||
OrderedNode *LoadSource_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint8_t OpSize, uint32_t Flags, int8_t Align, bool LoadData = true, bool ForceLoad = false);
|
||||
void StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, FEXCore::X86Tables::DecodedOperand const& Operand, OrderedNode *const Src, uint8_t OpSize, int8_t Align);
|
||||
@@ -553,23 +599,515 @@ private:
|
||||
|
||||
OrderedNode *SelectCC(uint8_t OP, OrderedNode *TrueValue, OrderedNode *FalseValue);
|
||||
|
||||
void GenerateFlags_ADC(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF);
|
||||
void GenerateFlags_SBB(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF);
|
||||
void GenerateFlags_SUB(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true);
|
||||
void GenerateFlags_ADD(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true);
|
||||
void GenerateFlags_MUL(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *High);
|
||||
void GenerateFlags_UMUL(FEXCore::X86Tables::DecodedOp Op, OrderedNode *High);
|
||||
void GenerateFlags_Logical(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void GenerateFlags_ShiftLeft(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void GenerateFlags_ShiftLeftImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void GenerateFlags_ShiftRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void GenerateFlags_ShiftRightImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void GenerateFlags_SignShiftRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void GenerateFlags_SignShiftRightImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void GenerateFlags_RotateRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void GenerateFlags_RotateLeft(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void GenerateFlags_RotateRightImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void GenerateFlags_RotateLeftImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
/**
|
||||
* @name Deferred RFLAG calculation and generation.
|
||||
*
|
||||
* Only handles the six flags that ALU ops typically generate.
|
||||
* Specifically: CF, PF, AF, ZF, SF, OF
|
||||
* These six flags are heavily generated through basic ALU ops and balloon the IR if not early eliminated.
|
||||
* This tracking structure only tracks single blocks and requires RFLAGS calculation at block-ending ops.
|
||||
* Some flags generating ALU ops only touch part of the registers, In these cases it will do calculation up front.
|
||||
* This means we still need our IR passes to eliminate all redundant flags accesses but this light OpcodeDispatcher optimization
|
||||
* doesn't take it to that level.
|
||||
* @{ */
|
||||
|
||||
// Deferred flag generation tracking structure.
|
||||
// This structure is used to track RFlags from ALU ops for invalidation.
|
||||
//
|
||||
// Future ideas: Use an invalidation mask to do partial generation of flags.
|
||||
// Particularly for the instructions that don't do the full set of flags calculations.
|
||||
// These instructions currently calculate the deferred RFLAGS immediately then overwrite rflags state.
|
||||
// RCLSE IR pass will catch and remove redundant rflags stores like this currently.
|
||||
struct DeferredFlagData {
|
||||
// What type of flags to generate
|
||||
FlagsGenerationType Type {FlagsGenerationType::TYPE_NONE};
|
||||
|
||||
// Source size of the op
|
||||
uint8_t SrcSize;
|
||||
|
||||
// Every flag generation type has a result
|
||||
OrderedNode *Res{};
|
||||
|
||||
union {
|
||||
// UMUL, BEXTR, BLSI, BLSMSK, POPCOUNT, TZCNT, LZCNT, BITSELECT
|
||||
struct {
|
||||
} NoSource;
|
||||
|
||||
// MUL, BLSR, BZHI
|
||||
struct {
|
||||
OrderedNode *Src1;
|
||||
} OneSource;
|
||||
|
||||
// Logical, LSHL, LSHR, ASHR, ROR, ROL
|
||||
struct {
|
||||
OrderedNode *Src1;
|
||||
OrderedNode *Src2;
|
||||
} TwoSource;
|
||||
|
||||
// ADC, SBB
|
||||
struct {
|
||||
OrderedNode *Src1;
|
||||
OrderedNode *Src2;
|
||||
OrderedNode *Src3;
|
||||
} ThreeSource;
|
||||
|
||||
// LSHLI, LSHRI, ASHRI, RORI, ROLI
|
||||
struct {
|
||||
OrderedNode *Src1;
|
||||
uint64_t Imm;
|
||||
} OneSrcImmediate;
|
||||
|
||||
// ADD, SUB
|
||||
struct {
|
||||
OrderedNode *Src1;
|
||||
OrderedNode *Src2;
|
||||
|
||||
bool UpdateCF;
|
||||
} TwoSrcImmediate;
|
||||
} Sources{};
|
||||
};
|
||||
|
||||
DeferredFlagData CurrentDeferredFlags{};
|
||||
|
||||
/**
|
||||
* @brief Takes the current deferred flag state and stores the result in to RFLAGS.
|
||||
*
|
||||
* Once executed there will no longer be any deferred flag state and RFLAGS will have the correct flags in it.
|
||||
* Necessary to do when leaving a IR block, or if an instruction is doing a partial overwrite of the flags.
|
||||
*/
|
||||
void CalculateDeferredFlags(uint32_t FlagsToCalculateMask = ~0U);
|
||||
|
||||
/**
|
||||
* @brief Invalidates the current deferred flags structure.
|
||||
*
|
||||
* If the emulated instruction is going to overwrite all of the flags but isn't tracked using the deferred flag system
|
||||
* then use this function to stop tracking the current active deferred flags.
|
||||
*/
|
||||
void InvalidateDeferredFlags() {
|
||||
CurrentDeferredFlags.Type = FlagsGenerationType::TYPE_NONE;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Checks if there is any deferred flag state active.
|
||||
*
|
||||
* @return True if RFLAGs contains the flags. False if deferred flags is tracking the data.
|
||||
*/
|
||||
bool IsDeferredFlagsStored() const {
|
||||
return CurrentDeferredFlags.Type == FlagsGenerationType::TYPE_NONE;
|
||||
}
|
||||
|
||||
/**
|
||||
* @name These functions are used by the deferred flag handling while it is calculating and storing flags in to RFLAGs.
|
||||
* @{ */
|
||||
void CalculcateFlags_ADC(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF);
|
||||
void CalculcateFlags_SBB(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF);
|
||||
void CalculcateFlags_SUB(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true);
|
||||
void CalculcateFlags_ADD(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true);
|
||||
void CalculcateFlags_MUL(uint8_t SrcSize, OrderedNode *Res, OrderedNode *High);
|
||||
void CalculcateFlags_UMUL(OrderedNode *High);
|
||||
void CalculcateFlags_Logical(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_ShiftLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_ShiftLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculcateFlags_ShiftRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_ShiftRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculcateFlags_SignShiftRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_SignShiftRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculcateFlags_RotateRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_RotateLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_RotateRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculcateFlags_RotateLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculcateFlags_FCMP(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_BEXTR(OrderedNode *Src);
|
||||
void CalculcateFlags_BLSI(uint8_t SrcSize, OrderedNode *Src);
|
||||
void CalculcateFlags_BLSMSK(OrderedNode *Src);
|
||||
void CalculcateFlags_BLSR(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src);
|
||||
void CalculcateFlags_POPCOUNT(OrderedNode *Src);
|
||||
void CalculcateFlags_BZHI(uint8_t SrcSize, OrderedNode *Result, OrderedNode *Src);
|
||||
void CalculcateFlags_TZCNT(OrderedNode *Src);
|
||||
void CalculcateFlags_LZCNT(uint8_t SrcSize, OrderedNode *Src);
|
||||
void CalculcateFlags_BITSELECT(OrderedNode *Src);
|
||||
/** @} */
|
||||
|
||||
/**
|
||||
* @name These functions generated deferred RFLAGs tracking.
|
||||
*
|
||||
* Depending on the operation it may force a RFLAGs calculation before storing the new deferred state.
|
||||
* @{ */
|
||||
void GenerateFlags_ADC(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_ADC,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.ThreeSource = {
|
||||
.Src1 = Src1,
|
||||
.Src2 = Src2,
|
||||
.Src3 = CF,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_SBB(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_SBB,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.ThreeSource = {
|
||||
.Src1 = Src1,
|
||||
.Src2 = Src2,
|
||||
.Src3 = CF,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_SUB(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true) {
|
||||
if (!UpdateCF) {
|
||||
// If we aren't updating CF then we need to calculate flags. Invalidation mask would make this not required.
|
||||
CalculateDeferredFlags();
|
||||
}
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_SUB,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.TwoSrcImmediate = {
|
||||
.Src1 = Src1,
|
||||
.Src2 = Src2,
|
||||
.UpdateCF = UpdateCF,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_ADD(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true) {
|
||||
if (!UpdateCF) {
|
||||
// If we aren't updating CF then we need to calculate flags. Invalidation mask would make this not required.
|
||||
CalculateDeferredFlags();
|
||||
}
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_ADD,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.TwoSrcImmediate = {
|
||||
.Src1 = Src1,
|
||||
.Src2 = Src2,
|
||||
.UpdateCF = UpdateCF,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_MUL(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *High) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_MUL,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.OneSource = {
|
||||
.Src1 = High,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_UMUL(FEXCore::X86Tables::DecodedOp Op, OrderedNode *High) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_UMUL,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = High,
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_Logical(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_LOGICAL,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.TwoSource = {
|
||||
.Src1 = Src1,
|
||||
.Src2 = Src2,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_ShiftLeft(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
// Flags need to be used, generate incoming flags first.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_LSHL,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.TwoSource = {
|
||||
.Src1 = Src1,
|
||||
.Src2 = Src2,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_ShiftRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
// Flags need to be used, generate incoming flags first.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_LSHR,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.TwoSource = {
|
||||
.Src1 = Src1,
|
||||
.Src2 = Src2,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_SignShiftRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
// Flags need to be used, generate incoming flags first.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_ASHR,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.TwoSource = {
|
||||
.Src1 = Src1,
|
||||
.Src2 = Src2,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_ShiftLeftImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero.
|
||||
if (Shift == 0) return;
|
||||
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_LSHLI,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.OneSrcImmediate = {
|
||||
.Src1 = Src1,
|
||||
.Imm = Shift,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_SignShiftRightImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero.
|
||||
if (Shift == 0) return;
|
||||
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_ASHRI,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.OneSrcImmediate = {
|
||||
.Src1 = Src1,
|
||||
.Imm = Shift,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_ShiftRightImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero.
|
||||
if (Shift == 0) return;
|
||||
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_LSHRI,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.OneSrcImmediate = {
|
||||
.Src1 = Src1,
|
||||
.Imm = Shift,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_RotateRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
// Doesn't set all the flags, needs to calculate.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_ROR,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.TwoSource = {
|
||||
.Src1 = Src1,
|
||||
.Src2 = Src2,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_RotateLeft(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
// Doesn't set all the flags, needs to calculate.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_ROL,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.TwoSource = {
|
||||
.Src1 = Src1,
|
||||
.Src2 = Src2,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_RotateRightImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
if (Shift == 0) return;
|
||||
|
||||
// Doesn't set all the flags, needs to calculate.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_RORI,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.OneSrcImmediate = {
|
||||
.Src1 = Src1,
|
||||
.Imm = Shift,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_RotateLeftImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
if (Shift == 0) return;
|
||||
|
||||
// Doesn't set all the flags, needs to calculate.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_ROLI,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.OneSrcImmediate = {
|
||||
.Src1 = Src1,
|
||||
.Imm = Shift,
|
||||
},
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_FCMP(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_FCMP,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.TwoSource = {
|
||||
.Src1 = Src1,
|
||||
.Src2 = Src2,
|
||||
},
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_BEXTR(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_BEXTR,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Src,
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_BLSI(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_BLSI,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Src,
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_BLSMSK(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_BLSMSK,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Src,
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_BLSR(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_BLSR,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.OneSource = {
|
||||
.Src1 = Src,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_POPCOUNT(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_POPCOUNT,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Src,
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_BZHI(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Result, OrderedNode *Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_BZHI,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Result,
|
||||
.Sources = {
|
||||
.OneSource = {
|
||||
.Src1 = Src,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_TZCNT(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_TZCNT,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Src,
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_LZCNT(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_LZCNT,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Src,
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_BITSELECT(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Src) {
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_BITSELECT,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Src,
|
||||
};
|
||||
}
|
||||
|
||||
/** @} */
|
||||
/** @} */
|
||||
|
||||
OrderedNode * GetX87Top();
|
||||
enum class X87Tag {
|
||||
@@ -609,11 +1147,22 @@ private:
|
||||
else
|
||||
return _LoadMem(ssa0, Invalid(), Align, Class, MEM_OFFSET_SXTX, 1, Size);
|
||||
}
|
||||
|
||||
|
||||
};
|
||||
|
||||
void InstallOpcodeHandlers(Context::OperatingMode Mode);
|
||||
|
||||
}
|
||||
template <>
|
||||
struct fmt::formatter<FEXCore::IR::OpDispatchBuilder::FlagsGenerationType> : fmt::formatter<int> {
|
||||
using Base = fmt::formatter<int>;
|
||||
|
||||
// Pass-through the underlying value, so IDs can
|
||||
// be formatted like any integral value.
|
||||
template <typename FormatContext>
|
||||
auto format(const FEXCore::IR::OpDispatchBuilder::FlagsGenerationType& ID, FormatContext& ctx) {
|
||||
return Base::format(static_cast<int>(ID), ctx);
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
|
||||
+475
-43
@@ -41,8 +41,17 @@ constexpr std::array<uint32_t, 17> FlagOffsets = {
|
||||
void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, OrderedNode *Src) {
|
||||
size_t NumFlags = FlagOffsets.size();
|
||||
if (Lower8) {
|
||||
// Calculate flags early.
|
||||
// Could use InvalidateDeferredFlags() if we had masked invalidation.
|
||||
// This is only a partial overwrite of flags since OF isn't stored here.
|
||||
CalculateDeferredFlags();
|
||||
NumFlags = 5;
|
||||
}
|
||||
else {
|
||||
// We are overwriting all RFLAGS. Invalidate the deferred flag state.
|
||||
InvalidateDeferredFlags();
|
||||
}
|
||||
|
||||
auto OneConst = _Constant(1);
|
||||
for (size_t i = 0; i < NumFlags; ++i) {
|
||||
const auto FlagOffset = FlagOffsets[i];
|
||||
@@ -52,6 +61,9 @@ void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, OrderedNode *Src) {
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::GetPackedRFLAG(bool Lower8) {
|
||||
// Calculate flags early.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
OrderedNode *Original = _Constant(2);
|
||||
size_t NumFlags = FlagOffsets.size();
|
||||
if (Lower8) {
|
||||
@@ -68,8 +80,185 @@ OrderedNode *OpDispatchBuilder::GetPackedRFLAG(bool Lower8) {
|
||||
return Original;
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_ADC(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF) {
|
||||
auto Size = GetSrcSize(Op) * 8;
|
||||
void OpDispatchBuilder::CalculateDeferredFlags(uint32_t FlagsToCalculateMask) {
|
||||
if (CurrentDeferredFlags.Type == FlagsGenerationType::TYPE_NONE) {
|
||||
// Nothing to do
|
||||
return;
|
||||
}
|
||||
|
||||
switch (CurrentDeferredFlags.Type) {
|
||||
case FlagsGenerationType::TYPE_ADC:
|
||||
CalculcateFlags_ADC(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.ThreeSource.Src1,
|
||||
CurrentDeferredFlags.Sources.ThreeSource.Src2,
|
||||
CurrentDeferredFlags.Sources.ThreeSource.Src3);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_SBB:
|
||||
CalculcateFlags_SBB(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.ThreeSource.Src1,
|
||||
CurrentDeferredFlags.Sources.ThreeSource.Src2,
|
||||
CurrentDeferredFlags.Sources.ThreeSource.Src3);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_SUB:
|
||||
CalculcateFlags_SUB(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.TwoSrcImmediate.Src1,
|
||||
CurrentDeferredFlags.Sources.TwoSrcImmediate.Src2,
|
||||
CurrentDeferredFlags.Sources.TwoSrcImmediate.UpdateCF);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_ADD:
|
||||
CalculcateFlags_ADD(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.TwoSrcImmediate.Src1,
|
||||
CurrentDeferredFlags.Sources.TwoSrcImmediate.Src2,
|
||||
CurrentDeferredFlags.Sources.TwoSrcImmediate.UpdateCF);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_MUL:
|
||||
CalculcateFlags_MUL(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSource.Src1);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_UMUL:
|
||||
CalculcateFlags_UMUL(CurrentDeferredFlags.Res);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_LOGICAL:
|
||||
CalculcateFlags_Logical(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src1,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src2);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_LSHL:
|
||||
CalculcateFlags_ShiftLeft(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src1,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src2);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_LSHLI:
|
||||
CalculcateFlags_ShiftLeftImmediate(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Src1,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Imm);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_LSHR:
|
||||
CalculcateFlags_ShiftRight(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src1,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src2);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_LSHRI:
|
||||
CalculcateFlags_ShiftRightImmediate(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Src1,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Imm);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_ASHR:
|
||||
CalculcateFlags_SignShiftRight(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src1,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src2);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_ASHRI:
|
||||
CalculcateFlags_SignShiftRightImmediate(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Src1,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Imm);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_ROR:
|
||||
CalculcateFlags_RotateRight(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src1,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src2);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_RORI:
|
||||
CalculcateFlags_RotateRightImmediate(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Src1,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Imm);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_ROL:
|
||||
CalculcateFlags_RotateLeft(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src1,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src2);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_ROLI:
|
||||
CalculcateFlags_RotateLeftImmediate(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Src1,
|
||||
CurrentDeferredFlags.Sources.OneSrcImmediate.Imm);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_FCMP:
|
||||
CalculcateFlags_FCMP(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src1,
|
||||
CurrentDeferredFlags.Sources.TwoSource.Src2);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_BEXTR:
|
||||
CalculcateFlags_BEXTR(CurrentDeferredFlags.Res);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_BLSI:
|
||||
CalculcateFlags_BLSI(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_BLSMSK:
|
||||
CalculcateFlags_BLSMSK(CurrentDeferredFlags.Res);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_BLSR:
|
||||
CalculcateFlags_BLSR(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSource.Src1);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_POPCOUNT:
|
||||
CalculcateFlags_POPCOUNT(CurrentDeferredFlags.Res);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_BZHI:
|
||||
CalculcateFlags_BZHI(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res,
|
||||
CurrentDeferredFlags.Sources.OneSource.Src1);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_TZCNT:
|
||||
CalculcateFlags_TZCNT(CurrentDeferredFlags.Res);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_LZCNT:
|
||||
CalculcateFlags_LZCNT(
|
||||
CurrentDeferredFlags.SrcSize,
|
||||
CurrentDeferredFlags.Res);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_BITSELECT:
|
||||
CalculcateFlags_BITSELECT(CurrentDeferredFlags.Res);
|
||||
break;
|
||||
case FlagsGenerationType::TYPE_NONE:
|
||||
default: ERROR_AND_DIE_FMT("Unhandled flags type {}", CurrentDeferredFlags.Type);
|
||||
}
|
||||
|
||||
// Done calculating
|
||||
CurrentDeferredFlags.Type = FlagsGenerationType::TYPE_NONE;
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculcateFlags_ADC(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF) {
|
||||
auto Size = SrcSize * 8;
|
||||
// AF
|
||||
{
|
||||
OrderedNode *AFRes = _Xor(_Xor(Src1, Src2), Res);
|
||||
@@ -79,7 +268,7 @@ void OpDispatchBuilder::GenerateFlags_ADC(FEXCore::X86Tables::DecodedOp Op, Orde
|
||||
|
||||
// SF
|
||||
{
|
||||
auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1);
|
||||
auto SignBitConst = _Constant(Size - 1);
|
||||
|
||||
auto LshrOp = _Lshr(Res, SignBitConst);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(LshrOp);
|
||||
@@ -140,9 +329,7 @@ void OpDispatchBuilder::GenerateFlags_ADC(FEXCore::X86Tables::DecodedOp Op, Orde
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_SBB(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF) {
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
|
||||
void OpDispatchBuilder::CalculcateFlags_SBB(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF) {
|
||||
// AF
|
||||
{
|
||||
OrderedNode *AFRes = _Xor(_Xor(Src1, Src2), Res);
|
||||
@@ -212,7 +399,7 @@ void OpDispatchBuilder::GenerateFlags_SBB(FEXCore::X86Tables::DecodedOp Op, Orde
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_SUB(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF) {
|
||||
void OpDispatchBuilder::CalculcateFlags_SUB(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF) {
|
||||
// AF
|
||||
{
|
||||
OrderedNode *AFRes = _Xor(_Xor(Src1, Src2), Res);
|
||||
@@ -222,7 +409,7 @@ void OpDispatchBuilder::GenerateFlags_SUB(FEXCore::X86Tables::DecodedOp Op, Orde
|
||||
|
||||
// SF
|
||||
{
|
||||
auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1);
|
||||
auto SignBitConst = _Constant(SrcSize * 8 - 1);
|
||||
|
||||
auto LshrOp = _Lshr(Res, SignBitConst);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(LshrOp);
|
||||
@@ -263,15 +450,13 @@ void OpDispatchBuilder::GenerateFlags_SUB(FEXCore::X86Tables::DecodedOp Op, Orde
|
||||
auto XorOp2 = _Xor(Res, Src1);
|
||||
OrderedNode *FinalAnd = _And(XorOp1, XorOp2);
|
||||
|
||||
FinalAnd = _Bfe(1, GetSrcSize(Op) * 8 - 1, FinalAnd);
|
||||
FinalAnd = _Bfe(1, SrcSize * 8 - 1, FinalAnd);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(FinalAnd);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_ADD(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF) {
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
|
||||
void OpDispatchBuilder::CalculcateFlags_ADD(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF) {
|
||||
// AF
|
||||
{
|
||||
OrderedNode *AFRes = _Xor(_Xor(Src1, Src2), Res);
|
||||
@@ -339,7 +524,7 @@ void OpDispatchBuilder::GenerateFlags_ADD(FEXCore::X86Tables::DecodedOp Op, Orde
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_MUL(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *High) {
|
||||
void OpDispatchBuilder::CalculcateFlags_MUL(uint8_t SrcSize, OrderedNode *Res, OrderedNode *High) {
|
||||
// PF/AF/ZF/SF
|
||||
// Undefined
|
||||
{
|
||||
@@ -354,7 +539,7 @@ void OpDispatchBuilder::GenerateFlags_MUL(FEXCore::X86Tables::DecodedOp Op, Orde
|
||||
// CF and OF are set if the result of the operation can't be fit in to the destination register
|
||||
// If the value can fit then the top bits will be zero
|
||||
|
||||
auto SignBit = _Sbfe(1, GetSrcSize(Op) * 8 - 1, Res);
|
||||
auto SignBit = _Sbfe(1, SrcSize * 8 - 1, Res);
|
||||
|
||||
auto SelectOp = _Select(FEXCore::IR::COND_EQ, High, SignBit, _Constant(0), _Constant(1));
|
||||
|
||||
@@ -363,7 +548,7 @@ void OpDispatchBuilder::GenerateFlags_MUL(FEXCore::X86Tables::DecodedOp Op, Orde
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_UMUL(FEXCore::X86Tables::DecodedOp Op, OrderedNode *High) {
|
||||
void OpDispatchBuilder::CalculcateFlags_UMUL(OrderedNode *High) {
|
||||
// AF/SF/PF/ZF
|
||||
// Undefined
|
||||
{
|
||||
@@ -385,7 +570,7 @@ void OpDispatchBuilder::GenerateFlags_UMUL(FEXCore::X86Tables::DecodedOp Op, Ord
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_Logical(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
void OpDispatchBuilder::CalculcateFlags_Logical(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
// AF
|
||||
{
|
||||
// Undefined
|
||||
@@ -395,7 +580,7 @@ void OpDispatchBuilder::GenerateFlags_Logical(FEXCore::X86Tables::DecodedOp Op,
|
||||
|
||||
// SF
|
||||
{
|
||||
auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1);
|
||||
auto SignBitConst = _Constant(SrcSize * 8 - 1);
|
||||
|
||||
auto LshrOp = _Lshr(Res, SignBitConst);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(LshrOp);
|
||||
@@ -430,11 +615,11 @@ auto oldflag = GetRFLAG(FEXCore::X86State::flag);\
|
||||
auto newval = _Select(FEXCore::IR::COND_EQ, cond, _Constant(0), oldflag, newflag);\
|
||||
SetRFLAG<FEXCore::X86State::flag>(newval);
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_ShiftLeft(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
void OpDispatchBuilder::CalculcateFlags_ShiftLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
auto Size = _Constant(GetSrcSize(Op) * 8);
|
||||
auto Size = _Constant(SrcSize * 8);
|
||||
auto ShiftAmt = _Sub(Size, Src2);
|
||||
auto LastBit = _And(_Lshr(Src1, ShiftAmt), _Constant(1));
|
||||
COND_FLAG_SET(Src2, RFLAG_CF_LOC, LastBit);
|
||||
@@ -466,7 +651,7 @@ void OpDispatchBuilder::GenerateFlags_ShiftLeft(FEXCore::X86Tables::DecodedOp Op
|
||||
|
||||
// SF
|
||||
{
|
||||
auto val = _Bfe(1, GetSrcSize(Op) * 8 - 1, Res);
|
||||
auto val = _Bfe(1, SrcSize * 8 - 1, Res);
|
||||
COND_FLAG_SET(Src2, RFLAG_SF_LOC, val);
|
||||
}
|
||||
|
||||
@@ -474,12 +659,12 @@ void OpDispatchBuilder::GenerateFlags_ShiftLeft(FEXCore::X86Tables::DecodedOp Op
|
||||
{
|
||||
// In the case of left shift. OF is only set from the result of <Top Source Bit> XOR <Top Result Bit>
|
||||
// When Shift > 1 then OF is undefined
|
||||
auto val = _Bfe(1, GetSrcSize(Op) * 8 - 1, _Xor(Src1, Res));
|
||||
auto val = _Bfe(1, SrcSize * 8 - 1, _Xor(Src1, Res));
|
||||
COND_FLAG_SET(Src2, RFLAG_OF_LOC, val);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_ShiftRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
void OpDispatchBuilder::CalculcateFlags_ShiftRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
@@ -514,7 +699,7 @@ void OpDispatchBuilder::GenerateFlags_ShiftRight(FEXCore::X86Tables::DecodedOp O
|
||||
|
||||
// SF
|
||||
{
|
||||
auto val =_Bfe(1, GetSrcSize(Op) * 8 - 1, Res);
|
||||
auto val =_Bfe(1, SrcSize * 8 - 1, Res);
|
||||
COND_FLAG_SET(Src2, RFLAG_SF_LOC, val);
|
||||
}
|
||||
|
||||
@@ -522,12 +707,12 @@ void OpDispatchBuilder::GenerateFlags_ShiftRight(FEXCore::X86Tables::DecodedOp O
|
||||
{
|
||||
// Only defined when Shift is 1 else undefined
|
||||
// OF flag is set if a sign change occurred
|
||||
auto val = _Bfe(1, GetSrcSize(Op) * 8 - 1, _Xor(Src1, Res));
|
||||
auto val = _Bfe(1, SrcSize * 8 - 1, _Xor(Src1, Res));
|
||||
COND_FLAG_SET(Src2, RFLAG_OF_LOC, val);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_SignShiftRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
void OpDispatchBuilder::CalculcateFlags_SignShiftRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
@@ -562,7 +747,7 @@ void OpDispatchBuilder::GenerateFlags_SignShiftRight(FEXCore::X86Tables::Decoded
|
||||
|
||||
// SF
|
||||
{
|
||||
auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1);
|
||||
auto SignBitConst = _Constant(SrcSize * 8 - 1);
|
||||
|
||||
auto LshrOp = _Lshr(Res, SignBitConst);
|
||||
COND_FLAG_SET(Src2, RFLAG_SF_LOC, LshrOp);
|
||||
@@ -574,14 +759,14 @@ void OpDispatchBuilder::GenerateFlags_SignShiftRight(FEXCore::X86Tables::Decoded
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_ShiftLeftImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculcateFlags_ShiftLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero
|
||||
if (Shift == 0) return;
|
||||
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(1, GetSrcSize(Op) * 8 - Shift, Src1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(1, SrcSize * 8 - Shift, Src1));
|
||||
}
|
||||
|
||||
// PF
|
||||
@@ -610,20 +795,20 @@ void OpDispatchBuilder::GenerateFlags_ShiftLeftImmediate(FEXCore::X86Tables::Dec
|
||||
|
||||
// SF
|
||||
{
|
||||
auto LshrOp = _Bfe(1, GetSrcSize(Op) * 8 - 1, Res);
|
||||
auto LshrOp = _Bfe(1, SrcSize * 8 - 1, Res);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(LshrOp);
|
||||
|
||||
// OF
|
||||
// In the case of left shift. OF is only set from the result of <Top Source Bit> XOR <Top Result Bit>
|
||||
if (Shift == 1) {
|
||||
auto SourceBit = _Bfe(1, GetSrcSize(Op) * 8 - 1, Src1);
|
||||
auto SourceBit = _Bfe(1, SrcSize * 8 - 1, Src1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(_Xor(SourceBit, LshrOp));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_SignShiftRightImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculcateFlags_SignShiftRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero
|
||||
if (Shift == 0) return;
|
||||
|
||||
@@ -659,7 +844,7 @@ void OpDispatchBuilder::GenerateFlags_SignShiftRightImmediate(FEXCore::X86Tables
|
||||
|
||||
// SF
|
||||
{
|
||||
auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1);
|
||||
auto SignBitConst = _Constant(SrcSize * 8 - 1);
|
||||
|
||||
auto LshrOp = _Lshr(Res, SignBitConst);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(LshrOp);
|
||||
@@ -674,7 +859,7 @@ void OpDispatchBuilder::GenerateFlags_SignShiftRightImmediate(FEXCore::X86Tables
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_ShiftRightImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculcateFlags_ShiftRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero
|
||||
if (Shift == 0) return;
|
||||
|
||||
@@ -710,7 +895,7 @@ void OpDispatchBuilder::GenerateFlags_ShiftRightImmediate(FEXCore::X86Tables::De
|
||||
|
||||
// SF
|
||||
{
|
||||
auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1);
|
||||
auto SignBitConst = _Constant(SrcSize * 8 - 1);
|
||||
|
||||
auto LshrOp = _Lshr(Res, SignBitConst);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(LshrOp);
|
||||
@@ -721,13 +906,13 @@ void OpDispatchBuilder::GenerateFlags_ShiftRightImmediate(FEXCore::X86Tables::De
|
||||
// Only defined when Shift is 1 else undefined
|
||||
// Is set to the MSB of the original value
|
||||
if (Shift == 1) {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(_Bfe(1, GetSrcSize(Op) * 8 - 1, Src1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(_Bfe(1, SrcSize * 8 - 1, Src1));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_RotateRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
auto OpSize = GetSrcSize(Op) * 8;
|
||||
void OpDispatchBuilder::CalculcateFlags_RotateRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
auto OpSize = SrcSize * 8;
|
||||
|
||||
// Extract the last bit shifted in to CF
|
||||
auto NewCF = _Bfe(1, OpSize - 1, Res);
|
||||
@@ -755,8 +940,8 @@ void OpDispatchBuilder::GenerateFlags_RotateRight(FEXCore::X86Tables::DecodedOp
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_RotateLeft(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
auto OpSize = GetSrcSize(Op) * 8;
|
||||
void OpDispatchBuilder::CalculcateFlags_RotateLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
auto OpSize = SrcSize * 8;
|
||||
|
||||
// Extract the last bit shifted in to CF
|
||||
//auto Size = _Constant(GetSrcSize(Res) * 8);
|
||||
@@ -785,10 +970,10 @@ void OpDispatchBuilder::GenerateFlags_RotateLeft(FEXCore::X86Tables::DecodedOp O
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_RotateRightImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculcateFlags_RotateRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
if (Shift == 0) return;
|
||||
|
||||
auto OpSize = GetSrcSize(Op) * 8;
|
||||
auto OpSize = SrcSize * 8;
|
||||
|
||||
auto NewCF = _Bfe(1, OpSize - Shift, Src1);
|
||||
|
||||
@@ -807,10 +992,10 @@ void OpDispatchBuilder::GenerateFlags_RotateRightImmediate(FEXCore::X86Tables::D
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::GenerateFlags_RotateLeftImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
void OpDispatchBuilder::CalculcateFlags_RotateLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
if (Shift == 0) return;
|
||||
|
||||
auto OpSize = GetSrcSize(Op) * 8;
|
||||
auto OpSize = SrcSize * 8;
|
||||
|
||||
// CF
|
||||
{
|
||||
@@ -827,4 +1012,251 @@ void OpDispatchBuilder::GenerateFlags_RotateLeftImmediate(FEXCore::X86Tables::De
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculcateFlags_FCMP(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
OrderedNode *HostFlag_CF = _GetHostFlag(Res, FCMP_FLAG_LT);
|
||||
OrderedNode *HostFlag_ZF = _GetHostFlag(Res, FCMP_FLAG_EQ);
|
||||
OrderedNode *HostFlag_Unordered = _GetHostFlag(Res, FCMP_FLAG_UNORDERED);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(HostFlag_ZF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(HostFlag_Unordered);
|
||||
|
||||
auto ZeroConst = _Constant(0);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(ZeroConst);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(ZeroConst);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(ZeroConst);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculcateFlags_BEXTR(OrderedNode *Src) {
|
||||
// Handle flag setting.
|
||||
//
|
||||
// All that matters primarily for this instruction is
|
||||
// that we only set the ZF flag properly.
|
||||
//
|
||||
// CF and OF are defined as being set to zero
|
||||
//
|
||||
SetRFLAG<X86State::RFLAG_CF_LOC>(_Constant(0));
|
||||
SetRFLAG<X86State::RFLAG_OF_LOC>(_Constant(0));
|
||||
|
||||
// Every other flag is considered undefined after a
|
||||
// BEXTR instruction, but we opt to reliably clear them.
|
||||
//
|
||||
SetRFLAG<X86State::RFLAG_AF_LOC>(_Constant(0));
|
||||
SetRFLAG<X86State::RFLAG_SF_LOC>(_Constant(0));
|
||||
|
||||
// PF
|
||||
if (CTX->Config.ABINoPF) {
|
||||
_InvalidateFlags(1UL << X86State::RFLAG_PF_LOC);
|
||||
} else {
|
||||
SetRFLAG<X86State::RFLAG_PF_LOC>(_Constant(0));
|
||||
}
|
||||
|
||||
// ZF
|
||||
auto ZeroOp = _Select(IR::COND_EQ,
|
||||
Src, _Constant(0),
|
||||
_Constant(1), _Constant(0));
|
||||
SetRFLAG<X86State::RFLAG_ZF_LOC>(ZeroOp);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculcateFlags_BLSI(uint8_t SrcSize, OrderedNode *Src) {
|
||||
// Now for the flags:
|
||||
//
|
||||
// Only CF, SF, ZF and OF are defined as being updated
|
||||
// CF is cleared if Src is zero, otherwise it's set.
|
||||
// SF is set to the value of the most significant operand bit of Result.
|
||||
// OF is always cleared
|
||||
// ZF is set, as usual, if Result is zero or not.
|
||||
//
|
||||
// AF and PF are documented as being in an undefined state after
|
||||
// a BLSI operation, however, we choose to reliably clear them.
|
||||
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
|
||||
SetRFLAG<X86State::RFLAG_OF_LOC>(Zero);
|
||||
SetRFLAG<X86State::RFLAG_AF_LOC>(Zero);
|
||||
if (CTX->Config.ABINoPF) {
|
||||
_InvalidateFlags(1UL << X86State::RFLAG_PF_LOC);
|
||||
} else {
|
||||
SetRFLAG<X86State::RFLAG_PF_LOC>(Zero);
|
||||
}
|
||||
|
||||
// ZF
|
||||
{
|
||||
auto ZFOp = _Select(IR::COND_EQ,
|
||||
Src, Zero,
|
||||
One, Zero);
|
||||
SetRFLAG<X86State::RFLAG_ZF_LOC>(ZFOp);
|
||||
}
|
||||
|
||||
// CF
|
||||
{
|
||||
auto CFOp = _Select(IR::COND_EQ,
|
||||
Src, Zero,
|
||||
Zero, One);
|
||||
SetRFLAG<X86State::RFLAG_CF_LOC>(CFOp);
|
||||
}
|
||||
|
||||
// SF
|
||||
{
|
||||
auto SignBit = _Constant(SrcSize * 8 - 1);
|
||||
auto SFOp = _Lshr(Src, SignBit);
|
||||
|
||||
SetRFLAG<X86State::RFLAG_SF_LOC>(SFOp);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculcateFlags_BLSMSK(OrderedNode *Src) {
|
||||
// Now for the flags.
|
||||
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
|
||||
SetRFLAG<X86State::RFLAG_ZF_LOC>(Zero);
|
||||
SetRFLAG<X86State::RFLAG_OF_LOC>(Zero);
|
||||
SetRFLAG<X86State::RFLAG_AF_LOC>(Zero);
|
||||
if (CTX->Config.ABINoPF) {
|
||||
_InvalidateFlags(1UL << X86State::RFLAG_PF_LOC);
|
||||
} else {
|
||||
SetRFLAG<X86State::RFLAG_PF_LOC>(Zero);
|
||||
}
|
||||
|
||||
auto CFOp = _Select(IR::COND_EQ,
|
||||
Src, Zero,
|
||||
Zero, One);
|
||||
SetRFLAG<X86State::RFLAG_CF_LOC>(CFOp);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculcateFlags_BLSR(uint8_t SrcSize, OrderedNode *Result, OrderedNode *Src) {
|
||||
// Now for flags.
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
|
||||
SetRFLAG<X86State::RFLAG_OF_LOC>(Zero);
|
||||
SetRFLAG<X86State::RFLAG_AF_LOC>(Zero);
|
||||
if (CTX->Config.ABINoPF) {
|
||||
_InvalidateFlags(1UL << X86State::RFLAG_PF_LOC);
|
||||
} else {
|
||||
SetRFLAG<X86State::RFLAG_PF_LOC>(Zero);
|
||||
}
|
||||
|
||||
// ZF
|
||||
{
|
||||
auto ZFOp = _Select(IR::COND_EQ,
|
||||
Result, Zero,
|
||||
One, Zero);
|
||||
SetRFLAG<X86State::RFLAG_ZF_LOC>(ZFOp);
|
||||
}
|
||||
|
||||
// CF
|
||||
{
|
||||
auto CFOp = _Select(IR::COND_EQ,
|
||||
Src, Zero,
|
||||
Zero, One);
|
||||
SetRFLAG<X86State::RFLAG_CF_LOC>(CFOp);
|
||||
}
|
||||
|
||||
// SF
|
||||
{
|
||||
auto SignBit = _Constant(SrcSize * 8 - 1);
|
||||
auto SFOp = _Lshr(Result, SignBit);
|
||||
|
||||
SetRFLAG<X86State::RFLAG_SF_LOC>(SFOp);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculcateFlags_POPCOUNT(OrderedNode *Src) {
|
||||
// Set ZF
|
||||
auto Zero = _Constant(0);
|
||||
auto ZFResult = _Select(FEXCore::IR::COND_EQ,
|
||||
Src, Zero,
|
||||
_Constant(1), Zero);
|
||||
|
||||
// Set flags
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(ZFResult);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(Zero);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculcateFlags_BZHI(uint8_t SrcSize, OrderedNode *Result, OrderedNode *Src) {
|
||||
// Now for the flags
|
||||
|
||||
auto Bounds = _Constant(SrcSize * 8- 1);
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
|
||||
SetRFLAG<X86State::RFLAG_OF_LOC>(Zero);
|
||||
SetRFLAG<X86State::RFLAG_AF_LOC>(Zero);
|
||||
if (CTX->Config.ABINoPF) {
|
||||
_InvalidateFlags(1UL << X86State::RFLAG_PF_LOC);
|
||||
} else {
|
||||
SetRFLAG<X86State::RFLAG_PF_LOC>(Zero);
|
||||
}
|
||||
|
||||
// ZF
|
||||
{
|
||||
auto ZFOp = _Select(IR::COND_EQ,
|
||||
Result, Zero,
|
||||
One, Zero);
|
||||
SetRFLAG<X86State::RFLAG_ZF_LOC>(ZFOp);
|
||||
}
|
||||
|
||||
// CF
|
||||
{
|
||||
auto CFOp = _Select(IR::COND_UGT,
|
||||
Src, Bounds,
|
||||
One, Zero);
|
||||
SetRFLAG<X86State::RFLAG_CF_LOC>(CFOp);
|
||||
}
|
||||
|
||||
// SF
|
||||
{
|
||||
auto SFOp = _Lshr(Result, Bounds);
|
||||
|
||||
SetRFLAG<X86State::RFLAG_SF_LOC>(SFOp);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculcateFlags_TZCNT(OrderedNode *Src) {
|
||||
// OF, SF, AF, PF all undefined
|
||||
auto Zero = _Constant(0);
|
||||
auto ZFResult = _Select(FEXCore::IR::COND_EQ,
|
||||
Src, Zero,
|
||||
_Constant(1), Zero);
|
||||
|
||||
// Set flags
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(ZFResult);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(_Bfe(1, 0, Src));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculcateFlags_LZCNT(uint8_t SrcSize, OrderedNode *Src) {
|
||||
// OF, SF, AF, PF all undefined
|
||||
|
||||
auto Zero = _Constant(0);
|
||||
auto ZFResult = _Select(FEXCore::IR::COND_EQ,
|
||||
Src, Zero,
|
||||
_Constant(1), Zero);
|
||||
|
||||
// Set flags
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(ZFResult);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(_Bfe(1, SrcSize * 8 - 1, Src));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculcateFlags_BITSELECT(OrderedNode *Src) {
|
||||
// OF, SF, AF, PF, CF all undefined
|
||||
|
||||
auto ZeroConst = _Constant(0);
|
||||
auto OneConst = _Constant(1);
|
||||
|
||||
// ZF is set to 1 if the source was zero
|
||||
auto ZFSelectOp = _Select(FEXCore::IR::COND_EQ,
|
||||
Src, ZeroConst,
|
||||
OneConst, ZeroConst);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(ZFSelectOp);
|
||||
}
|
||||
|
||||
}
|
||||
@@ -241,6 +241,8 @@ void OpDispatchBuilder::VectorALUOp<IR::OP_VCMPGT, 2>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorALUOp<IR::OP_VCMPGT, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorALUOp<IR::OP_VCMPGT, 8>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorALUOp<IR::OP_VCMPEQ, 1>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorALUOp<IR::OP_VCMPEQ, 2>(OpcodeArgs);
|
||||
@@ -1187,13 +1189,6 @@ void OpDispatchBuilder::VFCMPOp(OpcodeArgs) {
|
||||
auto Size = GetSrcSize(Op);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, GetDstSize(Op), Op->Flags, -1);
|
||||
OrderedNode *Src2{};
|
||||
if constexpr (Scalar) {
|
||||
Src2 = _VExtractElement(GetDstSize(Op), Size, Dest, 0);
|
||||
}
|
||||
else {
|
||||
Src2 = Dest;
|
||||
}
|
||||
uint8_t CompType = Op->Src[1].Data.Literal.Value;
|
||||
|
||||
OrderedNode *Result{};
|
||||
@@ -1201,30 +1196,30 @@ void OpDispatchBuilder::VFCMPOp(OpcodeArgs) {
|
||||
//auto ALUOp = _VCMPGT(Size, ElementSize, Dest, Src);
|
||||
switch (CompType) {
|
||||
case 0x00: case 0x08: case 0x10: case 0x18: // EQ
|
||||
Result = _VFCMPEQ(Size, ElementSize, Src2, Src);
|
||||
Result = _VFCMPEQ(Size, ElementSize, Dest, Src);
|
||||
break;
|
||||
case 0x01: case 0x09: case 0x11: case 0x19: // LT, GT(Swapped operand)
|
||||
Result = _VFCMPLT(Size, ElementSize, Src2, Src);
|
||||
Result = _VFCMPLT(Size, ElementSize, Dest, Src);
|
||||
break;
|
||||
case 0x02: case 0x0A: case 0x12: case 0x1A: // LE, GE(Swapped operand)
|
||||
Result = _VFCMPLE(Size, ElementSize, Src2, Src);
|
||||
Result = _VFCMPLE(Size, ElementSize, Dest, Src);
|
||||
break;
|
||||
case 0x03: case 0x0B: case 0x13: case 0x1B: // Unordered
|
||||
Result = _VFCMPUNO(Size, ElementSize, Src2, Src);
|
||||
Result = _VFCMPUNO(Size, ElementSize, Dest, Src);
|
||||
break;
|
||||
case 0x04: case 0x0C: case 0x14: case 0x1C: // NEQ
|
||||
Result = _VFCMPNEQ(Size, ElementSize, Src2, Src);
|
||||
Result = _VFCMPNEQ(Size, ElementSize, Dest, Src);
|
||||
break;
|
||||
case 0x05: case 0x0D: case 0x15: case 0x1D: // NLT, NGT(Swapped operand)
|
||||
Result = _VFCMPLT(Size, ElementSize, Src2, Src);
|
||||
Result = _VFCMPLT(Size, ElementSize, Dest, Src);
|
||||
Result = _VNot(Size, ElementSize, Result);
|
||||
break;
|
||||
case 0x06: case 0x0E: case 0x16: case 0x1E: // NLE, NGE(Swapped operand)
|
||||
Result = _VFCMPLE(Size, ElementSize, Src2, Src);
|
||||
Result = _VFCMPLE(Size, ElementSize, Dest, Src);
|
||||
Result = _VNot(Size, ElementSize, Result);
|
||||
break;
|
||||
case 0x07: case 0x0F: case 0x17: case 0x1F: // Ordered
|
||||
Result = _VFCMPORD(Size, ElementSize, Src2, Src);
|
||||
Result = _VFCMPORD(Size, ElementSize, Dest, Src);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Comparison type: {}", CompType);
|
||||
@@ -1432,18 +1427,7 @@ void OpDispatchBuilder::UCOMISxOp(OpcodeArgs) {
|
||||
(1 << FCMP_FLAG_LT) |
|
||||
(1 << FCMP_FLAG_UNORDERED));
|
||||
|
||||
OrderedNode *HostFlag_CF = _GetHostFlag(Res, FCMP_FLAG_LT);
|
||||
OrderedNode *HostFlag_ZF = _GetHostFlag(Res, FCMP_FLAG_EQ);
|
||||
OrderedNode *HostFlag_Unordered = _GetHostFlag(Res, FCMP_FLAG_UNORDERED);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(HostFlag_ZF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(HostFlag_Unordered);
|
||||
|
||||
auto ZeroConst = _Constant(0);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(ZeroConst);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(ZeroConst);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(ZeroConst);
|
||||
GenerateFlags_FCMP(Op, Res, Src1, Src2);
|
||||
|
||||
flagsOp = SelectionFlag::FCMP;
|
||||
flagsOpDest = Src1;
|
||||
@@ -2205,6 +2189,9 @@ template
|
||||
void OpDispatchBuilder::VectorVariableBlend<8>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::PTestOp(OpcodeArgs) {
|
||||
// Invalidate deferred flags early
|
||||
InvalidateDeferredFlags();
|
||||
|
||||
auto Size = GetSrcSize(Op);
|
||||
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
@@ -2233,8 +2220,14 @@ void OpDispatchBuilder::PTestOp(OpcodeArgs) {
|
||||
Test2 = _Select(FEXCore::IR::COND_EQ,
|
||||
Test2, ZeroConst, OneConst, ZeroConst);
|
||||
|
||||
// Careful, these flags are different between {V,}PTEST and VTESTP{S,D}
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(Test1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(Test2);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(ZeroConst);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(ZeroConst);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(ZeroConst);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(ZeroConst);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PHMINPOSUWOp(OpcodeArgs) {
|
||||
|
||||
@@ -685,12 +685,15 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs) {
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(HostFlag_ZF);
|
||||
}
|
||||
else {
|
||||
// Invalidate deferred flags early
|
||||
// OF, SF, AF, PF all undefined
|
||||
InvalidateDeferredFlags();
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(HostFlag_ZF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(HostFlag_Unordered);
|
||||
}
|
||||
|
||||
|
||||
if constexpr (poptwice) {
|
||||
// if we are popping then we must first mark this location as empty
|
||||
SetX87TopTag(top, X87Tag::Empty);
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include <unistd.h>
|
||||
#include <signal.h>
|
||||
@@ -91,7 +92,7 @@ namespace FEXCore {
|
||||
HostSignalHandler &Handler = HostHandlers[Signal];
|
||||
|
||||
if (!Thread) {
|
||||
LogMan::Msg::EFmt("[{}] Thread has received a signal and hasn't registered itself with the delegate! Programming error!", ::gettid());
|
||||
LogMan::Msg::EFmt("[{}] Thread has received a signal and hasn't registered itself with the delegate! Programming error!", FHU::Syscalls::gettid());
|
||||
}
|
||||
else {
|
||||
if (Handler.Handler &&
|
||||
|
||||
+4
-41
@@ -6,44 +6,7 @@
|
||||
|
||||
namespace FEXCore::X86Tables::X86InstDebugInfo {
|
||||
void InstallDebugInfo() {
|
||||
|
||||
using namespace FEXCore::X86Tables;
|
||||
auto NoFlags = Flags {0};
|
||||
|
||||
for (auto &BaseOp : BaseOps)
|
||||
BaseOp.DebugInfo = NoFlags;
|
||||
for (auto &BaseOp : SecondBaseOps)
|
||||
BaseOp.DebugInfo = NoFlags;
|
||||
for (auto &BaseOp : RepModOps)
|
||||
BaseOp.DebugInfo = NoFlags;
|
||||
for (auto &BaseOp : RepNEModOps)
|
||||
BaseOp.DebugInfo = NoFlags;
|
||||
for (auto &BaseOp : OpSizeModOps)
|
||||
BaseOp.DebugInfo = NoFlags;
|
||||
for (auto &BaseOp : PrimaryInstGroupOps)
|
||||
BaseOp.DebugInfo = NoFlags;
|
||||
for (auto &BaseOp : SecondInstGroupOps)
|
||||
BaseOp.DebugInfo = NoFlags;
|
||||
for (auto &BaseOp : SecondModRMTableOps)
|
||||
BaseOp.DebugInfo = NoFlags;
|
||||
for (auto &BaseOp : X87Ops)
|
||||
BaseOp.DebugInfo = NoFlags;
|
||||
for (auto &BaseOp : DDDNowOps)
|
||||
BaseOp.DebugInfo = NoFlags;
|
||||
for (auto &BaseOp : H0F38TableOps)
|
||||
BaseOp.DebugInfo = NoFlags;
|
||||
for (auto &BaseOp : H0F3ATableOps)
|
||||
BaseOp.DebugInfo = NoFlags;
|
||||
for (auto &BaseOp : VEXTableOps)
|
||||
BaseOp.DebugInfo = NoFlags;
|
||||
for (auto &BaseOp : VEXTableGroupOps)
|
||||
BaseOp.DebugInfo = NoFlags;
|
||||
for (auto &BaseOp : XOPTableOps)
|
||||
BaseOp.DebugInfo = NoFlags;
|
||||
for (auto &BaseOp : XOPTableGroupOps)
|
||||
BaseOp.DebugInfo = NoFlags;
|
||||
|
||||
const std::vector<std::tuple<uint8_t, uint8_t, Flags>> BaseOpTable = {
|
||||
const std::tuple<uint8_t, uint8_t, Flags> BaseOpTable[] = {
|
||||
{0x50, 8, {FLAGS_MEM_ACCESS}},
|
||||
{0x58, 8, {FLAGS_MEM_ACCESS}},
|
||||
|
||||
@@ -62,7 +25,7 @@ void InstallDebugInfo() {
|
||||
{0xF4, 1, {FLAGS_DEBUG}},
|
||||
};
|
||||
|
||||
const std::vector<std::tuple<uint8_t, uint8_t, Flags>> TwoByteOpTable = {
|
||||
const std::tuple<uint8_t, uint8_t, Flags> TwoByteOpTable[] = {
|
||||
{0x0B, 1, {FLAGS_DEBUG}},
|
||||
{0x19, 7, {FLAGS_DEBUG}},
|
||||
{0x28, 2, {FLAGS_MEM_ALIGN_16}},
|
||||
@@ -78,14 +41,14 @@ void InstallDebugInfo() {
|
||||
{0xFF, 1, {FLAGS_DEBUG}},
|
||||
};
|
||||
|
||||
const std::vector<std::tuple<uint8_t, uint8_t, Flags>> PrimaryGroupOpTable = {
|
||||
const std::tuple<uint8_t, uint8_t, Flags> PrimaryGroupOpTable[] = {
|
||||
#define OPD(group, prefix, Reg) (((group - FEXCore::X86Tables::TYPE_GROUP_1) << 6) | (prefix) << 3 | (Reg))
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 6), 2, {FLAGS_DIVIDE}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 6), 2, {FLAGS_DIVIDE}},
|
||||
#undef OPD
|
||||
};
|
||||
|
||||
const std::vector<std::tuple<uint16_t, uint8_t, Flags>> SecondaryExtensionOpTable = {
|
||||
const std::tuple<uint16_t, uint8_t, Flags> SecondaryExtensionOpTable[] = {
|
||||
#define PF_NONE 0
|
||||
#define PF_F3 1
|
||||
#define PF_66 2
|
||||
|
||||
+1
-1
@@ -9,12 +9,12 @@ $end_info$
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <cstring>
|
||||
#include <vector>
|
||||
#include <sys/mman.h>
|
||||
#include <bits/mman-map-flags-generic.h>
|
||||
|
||||
namespace FEXCore {
|
||||
constexpr size_t CODE_SIZE = 0x1000;
|
||||
|
||||
+17
-56
@@ -10,25 +10,24 @@ $end_info$
|
||||
|
||||
namespace FEXCore::X86Tables {
|
||||
|
||||
X86InstInfo BaseOps[MAX_PRIMARY_TABLE_SIZE];
|
||||
std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps{};
|
||||
std::array<X86InstInfo, MAX_SECOND_TABLE_SIZE> SecondBaseOps{};
|
||||
std::array<X86InstInfo, MAX_REP_MOD_TABLE_SIZE> RepModOps{};
|
||||
std::array<X86InstInfo, MAX_REPNE_MOD_TABLE_SIZE> RepNEModOps{};
|
||||
std::array<X86InstInfo, MAX_OPSIZE_MOD_TABLE_SIZE> OpSizeModOps{};
|
||||
|
||||
X86InstInfo SecondBaseOps[MAX_SECOND_TABLE_SIZE];
|
||||
X86InstInfo RepModOps[MAX_REP_MOD_TABLE_SIZE];
|
||||
X86InstInfo RepNEModOps[MAX_REPNE_MOD_TABLE_SIZE];
|
||||
X86InstInfo OpSizeModOps[MAX_OPSIZE_MOD_TABLE_SIZE];
|
||||
|
||||
X86InstInfo PrimaryInstGroupOps[MAX_INST_GROUP_TABLE_SIZE];
|
||||
X86InstInfo SecondInstGroupOps[MAX_INST_SECOND_GROUP_TABLE_SIZE];
|
||||
X86InstInfo SecondModRMTableOps[MAX_SECOND_MODRM_TABLE_SIZE];
|
||||
X86InstInfo X87Ops[MAX_X87_TABLE_SIZE];
|
||||
X86InstInfo DDDNowOps[MAX_3DNOW_TABLE_SIZE];
|
||||
X86InstInfo H0F38TableOps[MAX_0F_38_TABLE_SIZE];
|
||||
X86InstInfo H0F3ATableOps[MAX_0F_3A_TABLE_SIZE];
|
||||
X86InstInfo VEXTableOps[MAX_VEX_TABLE_SIZE];
|
||||
X86InstInfo VEXTableGroupOps[MAX_VEX_GROUP_TABLE_SIZE];
|
||||
X86InstInfo XOPTableOps[MAX_XOP_TABLE_SIZE];
|
||||
X86InstInfo XOPTableGroupOps[MAX_XOP_GROUP_TABLE_SIZE];
|
||||
X86InstInfo EVEXTableOps[MAX_EVEX_TABLE_SIZE];
|
||||
std::array<X86InstInfo, MAX_INST_GROUP_TABLE_SIZE> PrimaryInstGroupOps{};
|
||||
std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGroupOps{};
|
||||
std::array<X86InstInfo, MAX_SECOND_MODRM_TABLE_SIZE> SecondModRMTableOps{};
|
||||
std::array<X86InstInfo, MAX_X87_TABLE_SIZE> X87Ops{};
|
||||
std::array<X86InstInfo, MAX_3DNOW_TABLE_SIZE> DDDNowOps{};
|
||||
std::array<X86InstInfo, MAX_0F_38_TABLE_SIZE> H0F38TableOps{};
|
||||
std::array<X86InstInfo, MAX_0F_3A_TABLE_SIZE> H0F3ATableOps{};
|
||||
std::array<X86InstInfo, MAX_VEX_TABLE_SIZE> VEXTableOps{};
|
||||
std::array<X86InstInfo, MAX_VEX_GROUP_TABLE_SIZE> VEXTableGroupOps{};
|
||||
std::array<X86InstInfo, MAX_XOP_TABLE_SIZE> XOPTableOps{};
|
||||
std::array<X86InstInfo, MAX_XOP_GROUP_TABLE_SIZE> XOPTableGroupOps{};
|
||||
std::array<X86InstInfo, MAX_EVEX_TABLE_SIZE> EVEXTableOps{};
|
||||
|
||||
void InitializeBaseTables(Context::OperatingMode Mode);
|
||||
void InitializeSecondaryTables(Context::OperatingMode Mode);
|
||||
@@ -49,44 +48,6 @@ uint64_t NumInsts{};
|
||||
#endif
|
||||
|
||||
void InitializeInfoTables(Context::OperatingMode Mode) {
|
||||
using namespace FEXCore::X86Tables::InstFlags;
|
||||
auto UnknownOp = X86InstInfo{"UND", TYPE_UNKNOWN, FLAGS_NONE, 0, nullptr};
|
||||
|
||||
for (auto &BaseOp : BaseOps)
|
||||
BaseOp = UnknownOp;
|
||||
for (auto &BaseOp : SecondBaseOps)
|
||||
BaseOp = UnknownOp;
|
||||
for (auto &BaseOp : RepModOps)
|
||||
BaseOp = UnknownOp;
|
||||
for (auto &BaseOp : RepNEModOps)
|
||||
BaseOp = UnknownOp;
|
||||
for (auto &BaseOp : OpSizeModOps)
|
||||
BaseOp = UnknownOp;
|
||||
for (auto &BaseOp : PrimaryInstGroupOps)
|
||||
BaseOp = UnknownOp;
|
||||
for (auto &BaseOp : SecondInstGroupOps)
|
||||
BaseOp = UnknownOp;
|
||||
for (auto &BaseOp : SecondModRMTableOps)
|
||||
BaseOp = UnknownOp;
|
||||
for (auto &BaseOp : X87Ops)
|
||||
BaseOp = UnknownOp;
|
||||
for (auto &BaseOp : DDDNowOps)
|
||||
BaseOp = UnknownOp;
|
||||
for (auto &BaseOp : H0F38TableOps)
|
||||
BaseOp = UnknownOp;
|
||||
for (auto &BaseOp : H0F3ATableOps)
|
||||
BaseOp = UnknownOp;
|
||||
for (auto &BaseOp : VEXTableOps)
|
||||
BaseOp = UnknownOp;
|
||||
for (auto &BaseOp : VEXTableGroupOps)
|
||||
BaseOp = UnknownOp;
|
||||
for (auto &BaseOp : XOPTableOps)
|
||||
BaseOp = UnknownOp;
|
||||
for (auto &BaseOp : XOPTableGroupOps)
|
||||
BaseOp = UnknownOp;
|
||||
for (auto &BaseOp : EVEXTableOps)
|
||||
BaseOp = UnknownOp;
|
||||
|
||||
InitializeBaseTables(Mode);
|
||||
InitializeSecondaryTables(Mode);
|
||||
InitializePrimaryGroupTables(Mode);
|
||||
|
||||
@@ -288,13 +288,13 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0xEA, 1, X86InstInfo{"JMPF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
};
|
||||
|
||||
GenerateTable(BaseOps, BaseOpTable, std::size(BaseOpTable));
|
||||
GenerateTable(&BaseOps.at(0), BaseOpTable, std::size(BaseOpTable));
|
||||
|
||||
if (Mode == Context::MODE_64BIT) {
|
||||
GenerateTable(BaseOps, BaseOpTable_64, std::size(BaseOpTable_64));
|
||||
GenerateTable(&BaseOps.at(0), BaseOpTable_64, std::size(BaseOpTable_64));
|
||||
}
|
||||
else {
|
||||
GenerateTable(BaseOps, BaseOpTable_32, std::size(BaseOpTable_32));
|
||||
GenerateTable(&BaseOps.at(0), BaseOpTable_32, std::size(BaseOpTable_32));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -48,6 +48,6 @@ void InitializeDDDTables() {
|
||||
{0xB7, 1, X86InstInfo{"PMULHRW", TYPE_3DNOW_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
};
|
||||
|
||||
GenerateTable(DDDNowOps, DDDNowOpTable, std::size(DDDNowOpTable));
|
||||
GenerateTable(&DDDNowOps.at(0), DDDNowOpTable, std::size(DDDNowOpTable));
|
||||
}
|
||||
}
|
||||
@@ -30,6 +30,6 @@ void InitializeEVEXTables() {
|
||||
{0xE7, 1, X86InstInfo{"VMOVNTDQ", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
};
|
||||
|
||||
GenerateTable(EVEXTableOps, EVEXTable, std::size(EVEXTable));
|
||||
GenerateTable(&EVEXTableOps.at(0), EVEXTable, std::size(EVEXTable));
|
||||
}
|
||||
}
|
||||
@@ -14,11 +14,11 @@ namespace FEXCore::X86Tables {
|
||||
using namespace InstFlags;
|
||||
|
||||
void InitializeH0F38Tables() {
|
||||
#define OPD(prefix, opcode) ((prefix << 8) | opcode)
|
||||
#define OPD(prefix, opcode) (((prefix) << 8) | opcode)
|
||||
constexpr uint16_t PF_38_NONE = 0;
|
||||
constexpr uint16_t PF_38_66 = 1;
|
||||
constexpr uint16_t PF_38_F2 = 2;
|
||||
constexpr uint16_t PF_38_F3 = 3;
|
||||
constexpr uint16_t PF_38_66 = (1U << 0);
|
||||
constexpr uint16_t PF_38_F2 = (1U << 1);
|
||||
constexpr uint16_t PF_38_F3 = (1U << 2);
|
||||
|
||||
static constexpr U16U8InfoStruct H0F38Table[] = {
|
||||
{OPD(PF_38_NONE, 0x00), 1, X86InstInfo{"PSHUFB", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
@@ -74,6 +74,7 @@ void InitializeH0F38Tables() {
|
||||
{OPD(PF_38_66, 0x33), 1, X86InstInfo{"PMOVZXWD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x34), 1, X86InstInfo{"PMOVZXWQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x35), 1, X86InstInfo{"PMOVZXDQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x37), 1, X86InstInfo{"PCMPGTQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x38), 1, X86InstInfo{"PMINSB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x39), 1, X86InstInfo{"PMINSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x3A), 1, X86InstInfo{"PMINUW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
@@ -98,14 +99,16 @@ void InitializeH0F38Tables() {
|
||||
{OPD(PF_38_66, 0xF0), 1, X86InstInfo{"MOVBE", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0xF1), 1, X86InstInfo{"MOVBE", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0, nullptr}},
|
||||
|
||||
{OPD(PF_38_F2, 0xF0), 1, X86InstInfo{"CRC32", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_F2, 0xF1), 1, X86InstInfo{"CRC32", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(PF_38_F2, 0xF0), 1, X86InstInfo{"CRC32", TYPE_INST, GenFlagsSizes(SIZE_DEF, SIZE_8BIT) | FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(PF_38_F2, 0xF1), 1, X86InstInfo{"CRC32", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(PF_38_66 | PF_38_F2, 0xF0), 1, X86InstInfo{"CRC32", TYPE_INST, GenFlagsSizes(SIZE_DEF, SIZE_8BIT) | FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(PF_38_66 | PF_38_F2, 0xF1), 1, X86InstInfo{"CRC32", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
|
||||
{OPD(PF_38_66, 0xF6), 1, X86InstInfo{"ADCX", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(PF_38_F3, 0xF6), 1, X86InstInfo{"ADOX", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
GenerateTable(H0F38TableOps, H0F38Table, std::size(H0F38Table));
|
||||
GenerateTable(&H0F38TableOps.at(0), H0F38Table, std::size(H0F38Table));
|
||||
}
|
||||
}
|
||||
@@ -60,10 +60,10 @@ void InitializeH0F3ATables(Context::OperatingMode Mode) {
|
||||
|
||||
#undef OPD
|
||||
|
||||
GenerateTable(H0F3ATableOps, H0F3ATable, std::size(H0F3ATable));
|
||||
GenerateTable(&H0F3ATableOps.at(0), H0F3ATable, std::size(H0F3ATable));
|
||||
|
||||
if (Mode == Context::MODE_64BIT) {
|
||||
GenerateTable(H0F3ATableOps, H0F3ATable_64, std::size(H0F3ATable_64));
|
||||
GenerateTable(&H0F3ATableOps.at(0), H0F3ATable_64, std::size(H0F3ATable_64));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -163,14 +163,13 @@ void InitializePrimaryGroupTables(Context::OperatingMode Mode) {
|
||||
|
||||
#undef OPD
|
||||
|
||||
GenerateTable(PrimaryInstGroupOps, PrimaryGroupOpTable, std::size(PrimaryGroupOpTable));
|
||||
GenerateTable(&PrimaryInstGroupOps.at(0), PrimaryGroupOpTable, std::size(PrimaryGroupOpTable));
|
||||
if (Mode == Context::MODE_64BIT) {
|
||||
GenerateTable(PrimaryInstGroupOps, PrimaryGroupOpTable_64, std::size(PrimaryGroupOpTable_64));
|
||||
GenerateTable(&PrimaryInstGroupOps.at(0), PrimaryGroupOpTable_64, std::size(PrimaryGroupOpTable_64));
|
||||
}
|
||||
else {
|
||||
GenerateTable(PrimaryInstGroupOps, PrimaryGroupOpTable_32, std::size(PrimaryGroupOpTable_32));
|
||||
GenerateTable(&PrimaryInstGroupOps.at(0), PrimaryGroupOpTable_32, std::size(PrimaryGroupOpTable_32));
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
@@ -488,7 +488,7 @@ void InitializeSecondaryGroupTables() {
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
GenerateTable(SecondInstGroupOps, SecondaryExtensionOpTable, std::size(SecondaryExtensionOpTable));
|
||||
GenerateTable(&SecondInstGroupOps.at(0), SecondaryExtensionOpTable, std::size(SecondaryExtensionOpTable));
|
||||
}
|
||||
|
||||
}
|
||||
@@ -47,15 +47,15 @@ void InitializeSecondaryModRMTables() {
|
||||
|
||||
// REG /7
|
||||
{((3 << 3) | 0), 1, X86InstInfo{"SWAPGS", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{((3 << 3) | 1), 1, X86InstInfo{"RDTSCP", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{((3 << 3) | 1), 1, X86InstInfo{"RDTSCP", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{((3 << 3) | 2), 1, X86InstInfo{"MONITORX", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{((3 << 3) | 3), 1, X86InstInfo{"MWAITX", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{((3 << 3) | 4), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{((3 << 3) | 4), 1, X86InstInfo{"CLZERO", TYPE_INST, FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{((3 << 3) | 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{((3 << 3) | 6), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{((3 << 3) | 7), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
};
|
||||
|
||||
GenerateTable(SecondModRMTableOps, SecondaryModRMExtensionOpTable, std::size(SecondaryModRMExtensionOpTable));
|
||||
GenerateTable(&SecondModRMTableOps.at(0), SecondaryModRMExtensionOpTable, std::size(SecondaryModRMExtensionOpTable));
|
||||
}
|
||||
}
|
||||
@@ -582,18 +582,18 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0xFF, 1, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0, nullptr}},
|
||||
};
|
||||
|
||||
GenerateTable(SecondBaseOps, TwoByteOpTable, std::size(TwoByteOpTable));
|
||||
GenerateTable(&SecondBaseOps.at(0), TwoByteOpTable, std::size(TwoByteOpTable));
|
||||
|
||||
if (Mode == Context::MODE_64BIT) {
|
||||
GenerateTable(SecondBaseOps, TwoByteOpTable_64, std::size(TwoByteOpTable_64));
|
||||
if (Mode == Context::MODE_64BIT) {
|
||||
GenerateTable(&SecondBaseOps.at(0), TwoByteOpTable_64, std::size(TwoByteOpTable_64));
|
||||
}
|
||||
else {
|
||||
GenerateTable(SecondBaseOps, TwoByteOpTable_32, std::size(TwoByteOpTable_32));
|
||||
GenerateTable(&SecondBaseOps.at(0), TwoByteOpTable_32, std::size(TwoByteOpTable_32));
|
||||
}
|
||||
|
||||
GenerateTableWithCopy(RepModOps, RepModOpTable, std::size(RepModOpTable), SecondBaseOps);
|
||||
GenerateTableWithCopy(RepNEModOps, RepNEModOpTable, std::size(RepNEModOpTable), SecondBaseOps);
|
||||
GenerateTableWithCopy(OpSizeModOps, OpSizeModOpTable, std::size(OpSizeModOpTable), SecondBaseOps);
|
||||
GenerateTableWithCopy(&RepModOps.at(0), RepModOpTable, std::size(RepModOpTable), &SecondBaseOps.at(0));
|
||||
GenerateTableWithCopy(&RepNEModOps.at(0), RepNEModOpTable, std::size(RepNEModOpTable), &SecondBaseOps.at(0));
|
||||
GenerateTableWithCopy(&OpSizeModOps.at(0), OpSizeModOpTable, std::size(OpSizeModOpTable), &SecondBaseOps.at(0));
|
||||
|
||||
}
|
||||
|
||||
|
||||
@@ -394,8 +394,9 @@ void InitializeVEXTables() {
|
||||
{OPD(2, 0b11, 0xF3), 1, X86InstInfo{"", TYPE_VEX_GROUP_17, FLAGS_NONE, 0, nullptr}}, // VEX Group 17
|
||||
|
||||
{OPD(2, 0b00, 0xF5), 1, X86InstInfo{"BZHI", TYPE_INST, FLAGS_MODRM | FLAGS_VEX_2ND_SRC, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0xF5), 1, X86InstInfo{"PEXT", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b11, 0xF5), 1, X86InstInfo{"PDEP", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
// AMD reference manual is incorrect. PEXT actually maps to 0b10, not 0b01.
|
||||
{OPD(2, 0b10, 0xF5), 1, X86InstInfo{"PEXT", TYPE_INST, FLAGS_MODRM | FLAGS_VEX_1ST_SRC, 0, nullptr}},
|
||||
{OPD(2, 0b11, 0xF5), 1, X86InstInfo{"PDEP", TYPE_INST, FLAGS_MODRM | FLAGS_VEX_1ST_SRC, 0, nullptr}},
|
||||
|
||||
{OPD(2, 0b11, 0xF6), 1, X86InstInfo{"MULX", TYPE_INST, FLAGS_MODRM | FLAGS_VEX_1ST_SRC, 0, nullptr}},
|
||||
|
||||
@@ -509,7 +510,7 @@ void InitializeVEXTables() {
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
GenerateTable(VEXTableOps, VEXTable, std::size(VEXTable));
|
||||
GenerateTable(VEXTableGroupOps, VEXGroupTable, std::size(VEXGroupTable));
|
||||
GenerateTable(&VEXTableOps.at(0), VEXTable, std::size(VEXTable));
|
||||
GenerateTable(&VEXTableGroupOps.at(0), VEXGroupTable, std::size(VEXGroupTable));
|
||||
}
|
||||
}
|
||||
+14
-30
@@ -17,20 +17,19 @@ extern uint64_t Total;
|
||||
extern uint64_t NumInsts;
|
||||
#endif
|
||||
|
||||
struct U8U8InfoStruct {
|
||||
uint8_t first, second;
|
||||
X86InstInfo Info;
|
||||
};
|
||||
|
||||
struct U16U8InfoStruct {
|
||||
uint16_t first;
|
||||
template <typename OpcodeType>
|
||||
struct X86TablesInfoStruct {
|
||||
OpcodeType first;
|
||||
uint8_t second;
|
||||
X86InstInfo Info;
|
||||
};
|
||||
using U8U8InfoStruct = X86TablesInfoStruct<uint8_t>;
|
||||
using U16U8InfoStruct = X86TablesInfoStruct<uint16_t>;
|
||||
|
||||
static inline void GenerateTable(X86InstInfo *FinalTable, U8U8InfoStruct const *LocalTable, size_t TableSize) {
|
||||
template<typename OpcodeType>
|
||||
static inline void GenerateTable(X86InstInfo *FinalTable, X86TablesInfoStruct<OpcodeType> const *LocalTable, size_t TableSize) {
|
||||
for (size_t j = 0; j < TableSize; ++j) {
|
||||
U8U8InfoStruct const &Op = LocalTable[j];
|
||||
X86TablesInfoStruct<OpcodeType> const &Op = LocalTable[j];
|
||||
auto OpNum = Op.first;
|
||||
X86InstInfo const &Info = Op.Info;
|
||||
for (uint32_t i = 0; i < Op.second; ++i) {
|
||||
@@ -45,26 +44,10 @@ static inline void GenerateTable(X86InstInfo *FinalTable, U8U8InfoStruct const *
|
||||
}
|
||||
};
|
||||
|
||||
static inline void GenerateTable(X86InstInfo *FinalTable, U16U8InfoStruct const *LocalTable, size_t TableSize) {
|
||||
template<typename OpcodeType>
|
||||
static inline void GenerateTableWithCopy(X86InstInfo *FinalTable, X86TablesInfoStruct<OpcodeType> const *LocalTable, size_t TableSize, X86InstInfo *OtherLocal) {
|
||||
for (size_t j = 0; j < TableSize; ++j) {
|
||||
U16U8InfoStruct const &Op = LocalTable[j];
|
||||
auto OpNum = Op.first;
|
||||
X86InstInfo const &Info = Op.Info;
|
||||
for (uint32_t i = 0; i < Op.second; ++i) {
|
||||
LOGMAN_THROW_A_FMT(FinalTable[OpNum + i].Type == TYPE_UNKNOWN, "Duplicate Entry {}->{}", FinalTable[OpNum + i].Name, Info.Name);
|
||||
FinalTable[OpNum + i] = Info;
|
||||
#ifndef NDEBUG
|
||||
++Total;
|
||||
if (Info.Type == TYPE_INST)
|
||||
NumInsts++;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
static inline void GenerateTableWithCopy(X86InstInfo *FinalTable, U8U8InfoStruct const *LocalTable, size_t TableSize, X86InstInfo *OtherLocal) {
|
||||
for (size_t j = 0; j < TableSize; ++j) {
|
||||
U8U8InfoStruct const &Op = LocalTable[j];
|
||||
X86TablesInfoStruct<OpcodeType> const &Op = LocalTable[j];
|
||||
auto OpNum = Op.first;
|
||||
X86InstInfo const &Info = Op.Info;
|
||||
for (uint32_t i = 0; i < Op.second; ++i) {
|
||||
@@ -84,9 +67,10 @@ static inline void GenerateTableWithCopy(X86InstInfo *FinalTable, U8U8InfoStruct
|
||||
}
|
||||
};
|
||||
|
||||
static inline void GenerateX87Table(X86InstInfo *FinalTable, U16U8InfoStruct const *LocalTable, size_t TableSize) {
|
||||
template<typename OpcodeType>
|
||||
static inline void GenerateX87Table(X86InstInfo *FinalTable, X86TablesInfoStruct<OpcodeType> const *LocalTable, size_t TableSize) {
|
||||
for (size_t j = 0; j < TableSize; ++j) {
|
||||
U16U8InfoStruct const &Op = LocalTable[j];
|
||||
X86TablesInfoStruct<OpcodeType> const &Op = LocalTable[j];
|
||||
auto OpNum = Op.first;
|
||||
X86InstInfo const &Info = Op.Info;
|
||||
for (uint32_t i = 0; i < Op.second; ++i) {
|
||||
|
||||
@@ -263,6 +263,6 @@ void InitializeX87Tables() {
|
||||
#undef OPD
|
||||
#undef OPDReg
|
||||
|
||||
GenerateX87Table(X87Ops, X87OpTable, std::size(X87OpTable));
|
||||
GenerateX87Table(&X87Ops.at(0), X87OpTable, std::size(X87OpTable));
|
||||
}
|
||||
}
|
||||
@@ -130,7 +130,7 @@ void InitializeXOPTables() {
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
GenerateTable(XOPTableOps, XOPTable, std::size(XOPTable));
|
||||
GenerateTable(XOPTableGroupOps, XOPGroupTable, std::size(XOPGroupTable));
|
||||
GenerateTable(&XOPTableOps.at(0), XOPTable, std::size(XOPTable));
|
||||
GenerateTable(&XOPTableGroupOps.at(0), XOPGroupTable, std::size(XOPGroupTable));
|
||||
}
|
||||
}
|
||||
+408
@@ -0,0 +1,408 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/IR/AOTIR.h"
|
||||
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <filesystem>
|
||||
#include <fstream>
|
||||
#include <mutex>
|
||||
#include <sys/mman.h>
|
||||
#include <sys/stat.h>
|
||||
#include <unistd.h>
|
||||
#include <xxhash.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
AOTIRInlineEntry *AOTIRInlineIndex::GetInlineEntry(uint64_t DataOffset) {
|
||||
uintptr_t This = (uintptr_t)this;
|
||||
|
||||
return (AOTIRInlineEntry*)(This + DataBase + DataOffset);
|
||||
}
|
||||
|
||||
AOTIRInlineEntry *AOTIRInlineIndex::Find(uint64_t GuestStart) {
|
||||
ssize_t l = 0;
|
||||
ssize_t r = Count - 1;
|
||||
|
||||
while (l <= r) {
|
||||
size_t m = l + (r - l) / 2;
|
||||
|
||||
if (Entries[m].GuestStart == GuestStart)
|
||||
return GetInlineEntry(Entries[m].DataOffset);
|
||||
else if (Entries[m].GuestStart < GuestStart)
|
||||
l = m + 1;
|
||||
else
|
||||
r = m - 1;
|
||||
}
|
||||
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
IR::RegisterAllocationData *AOTIRInlineEntry::GetRAData() {
|
||||
return (IR::RegisterAllocationData *)InlineData;
|
||||
}
|
||||
|
||||
IR::IRListView *AOTIRInlineEntry::GetIRData() {
|
||||
auto RAData = GetRAData();
|
||||
auto Offset = RAData->Size(RAData->MapCount);
|
||||
|
||||
return (IR::IRListView *)&InlineData[Offset];
|
||||
}
|
||||
|
||||
void AOTIRCaptureCacheEntry::AppendAOTIRCaptureCache(uint64_t GuestRIP, uint64_t Start, uint64_t Length, uint64_t Hash, FEXCore::IR::IRListView *IRList, FEXCore::IR::RegisterAllocationData *RAData) {
|
||||
auto Inserted = Index.emplace(GuestRIP, Stream->tellp());
|
||||
|
||||
if (Inserted.second) {
|
||||
//GuestHash
|
||||
Stream->write((const char*)&Hash, sizeof(Hash));
|
||||
|
||||
//GuestLength
|
||||
Stream->write((const char*)&Length, sizeof(Length));
|
||||
|
||||
RAData->Serialize(*Stream);
|
||||
|
||||
// IRData (inline)
|
||||
IRList->Serialize(*Stream);
|
||||
}
|
||||
}
|
||||
|
||||
static bool readAll(int fd, void *data, size_t size) {
|
||||
int rv = read(fd, data, size);
|
||||
|
||||
if (rv != size)
|
||||
return false;
|
||||
else
|
||||
return true;
|
||||
}
|
||||
|
||||
bool LoadAOTIRCache(AOTCacheType *AOTIRCache, int streamfd) {
|
||||
uint64_t tag;
|
||||
|
||||
if (!readAll(streamfd, (char*)&tag, sizeof(tag)) || tag != FEXCore::IR::AOTIR_COOKIE)
|
||||
return false;
|
||||
|
||||
std::string Module;
|
||||
uint64_t ModSize;
|
||||
uint64_t IndexSize;
|
||||
|
||||
lseek(streamfd, -sizeof(ModSize), SEEK_END);
|
||||
|
||||
if (!readAll(streamfd, (char*)&ModSize, sizeof(ModSize)))
|
||||
return false;
|
||||
|
||||
Module.resize(ModSize);
|
||||
|
||||
lseek(streamfd, -sizeof(ModSize) - ModSize, SEEK_END);
|
||||
|
||||
if (!readAll(streamfd, (char*)&Module[0], Module.size()))
|
||||
return false;
|
||||
|
||||
lseek(streamfd, -sizeof(ModSize) - ModSize - sizeof(IndexSize), SEEK_END);
|
||||
|
||||
if (!readAll(streamfd, (char*)&IndexSize, sizeof(IndexSize)))
|
||||
return false;
|
||||
|
||||
struct stat fileinfo;
|
||||
if (fstat(streamfd, &fileinfo) < 0)
|
||||
return false;
|
||||
size_t Size = (fileinfo.st_size + 4095) & ~4095;
|
||||
|
||||
size_t IndexOffset = fileinfo.st_size - IndexSize -sizeof(ModSize) - ModSize - sizeof(IndexSize);
|
||||
|
||||
void *FilePtr = FEXCore::Allocator::mmap(nullptr, Size, PROT_READ, MAP_SHARED, streamfd, 0);
|
||||
|
||||
if (FilePtr == MAP_FAILED) {
|
||||
return false;
|
||||
}
|
||||
|
||||
auto Array = (AOTIRInlineIndex *)((char*)FilePtr + IndexOffset);
|
||||
|
||||
AOTIRCache->insert({Module, {Array, FilePtr, Size}});
|
||||
|
||||
LogMan::Msg::DFmt("AOTIR: Module {} has {} functions", Module, Array->Count);
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
AOTIRCaptureCache::~AOTIRCaptureCache() {
|
||||
for (auto &Mod: AOTIRCache) {
|
||||
FEXCore::Allocator::munmap(Mod.second.mapping, Mod.second.size);
|
||||
}
|
||||
}
|
||||
|
||||
void AOTIRCaptureCache::FinalizeAOTIRCache() {
|
||||
AOTIRCaptureCacheWriteoutQueue_Flush();
|
||||
|
||||
std::unique_lock lk(AOTIRCacheLock);
|
||||
|
||||
for (auto& [String, Entry] : AOTIRCaptureCacheMap) {
|
||||
if (!Entry.Stream) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const auto ModSize = String.size();
|
||||
auto &stream = Entry.Stream;
|
||||
|
||||
// pad to 32 bytes
|
||||
constexpr char Zero = 0;
|
||||
while(stream->tellp() & 31)
|
||||
stream->write(&Zero, 1);
|
||||
|
||||
// AOTIRInlineIndex
|
||||
const auto FnCount = Entry.Index.size();
|
||||
const size_t DataBase = -stream->tellp();
|
||||
|
||||
stream->write((const char*)&FnCount, sizeof(FnCount));
|
||||
stream->write((const char*)&DataBase, sizeof(DataBase));
|
||||
|
||||
for (const auto& [GuestStart, DataOffset] : Entry.Index) {
|
||||
//AOTIRInlineIndexEntry
|
||||
|
||||
// GuestStart
|
||||
stream->write((const char*)&GuestStart, sizeof(GuestStart));
|
||||
|
||||
// DataOffset
|
||||
stream->write((const char*)&DataOffset, sizeof(DataOffset));
|
||||
}
|
||||
|
||||
// End of file header
|
||||
const auto IndexSize = FnCount * sizeof(FEXCore::IR::AOTIRInlineIndexEntry) + sizeof(DataBase) + sizeof(FnCount);
|
||||
stream->write((const char*)&IndexSize, sizeof(IndexSize));
|
||||
stream->write(String.c_str(), ModSize);
|
||||
stream->write((const char*)&ModSize, sizeof(ModSize));
|
||||
|
||||
// Close the stream
|
||||
stream->close();
|
||||
|
||||
// Rename the file to atomically update the cache with the temporary file
|
||||
AOTIRRenamer(String);
|
||||
}
|
||||
}
|
||||
|
||||
void AOTIRCaptureCache::AOTIRCaptureCacheWriteoutQueue_Flush() {
|
||||
{
|
||||
std::shared_lock lk{AOTIRCaptureCacheWriteoutLock};
|
||||
if (AOTIRCaptureCacheWriteoutQueue.size() == 0) {
|
||||
AOTIRCaptureCacheWriteoutFlusing.store(false);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
for (;;) {
|
||||
AOTIRCaptureCacheWriteoutLock.lock();
|
||||
std::function<void()> fn = std::move(AOTIRCaptureCacheWriteoutQueue.front());
|
||||
bool MaybeEmpty = false;
|
||||
AOTIRCaptureCacheWriteoutQueue.pop();
|
||||
MaybeEmpty = AOTIRCaptureCacheWriteoutQueue.size() == 0;
|
||||
AOTIRCaptureCacheWriteoutLock.unlock();
|
||||
|
||||
fn();
|
||||
if (MaybeEmpty) {
|
||||
std::shared_lock lk{AOTIRCaptureCacheWriteoutLock};
|
||||
if (AOTIRCaptureCacheWriteoutQueue.size() == 0) {
|
||||
AOTIRCaptureCacheWriteoutFlusing.store(false);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
LOGMAN_MSG_A_FMT("Must never get here");
|
||||
}
|
||||
|
||||
void AOTIRCaptureCache::AOTIRCaptureCacheWriteoutQueue_Append(const std::function<void()> &fn) {
|
||||
bool Flush = false;
|
||||
|
||||
{
|
||||
std::unique_lock lk{AOTIRCaptureCacheWriteoutLock};
|
||||
AOTIRCaptureCacheWriteoutQueue.push(fn);
|
||||
if (AOTIRCaptureCacheWriteoutQueue.size() > 10000) {
|
||||
Flush = true;
|
||||
}
|
||||
}
|
||||
|
||||
bool test_val = false;
|
||||
if (Flush && AOTIRCaptureCacheWriteoutFlusing.compare_exchange_strong(test_val, true)) {
|
||||
AOTIRCaptureCacheWriteoutQueue_Flush();
|
||||
}
|
||||
}
|
||||
|
||||
void AOTIRCaptureCache::WriteFilesWithCode(std::function<void(const std::string& fileid, const std::string& filename)> Writer) {
|
||||
std::shared_lock lk(AOTIRCacheLock);
|
||||
for( const auto &File: FilesWithCode) {
|
||||
Writer(File.first, File.second);
|
||||
}
|
||||
}
|
||||
|
||||
AOTIRCaptureCache::PreGenerateIRFetchResult AOTIRCaptureCache::PreGenerateIRFetch(uint64_t GuestRIP, FEXCore::IR::IRListView *IRList) {
|
||||
{
|
||||
std::shared_lock lk(AOTIRCacheLock);
|
||||
auto file = AddrToFile.lower_bound(GuestRIP);
|
||||
if (file != AddrToFile.begin()) {
|
||||
--file;
|
||||
if (!file->second.ContainsCode) {
|
||||
file->second.ContainsCode = true;
|
||||
FilesWithCode[file->second.fileid] = file->second.filename;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
PreGenerateIRFetchResult Result{};
|
||||
if (IRList == nullptr && CTX->Config.AOTIRLoad()) {
|
||||
std::shared_lock lk(AOTIRCacheLock);
|
||||
auto file = AddrToFile.lower_bound(GuestRIP);
|
||||
if (file != AddrToFile.begin()) {
|
||||
--file;
|
||||
auto Mod = (FEXCore::IR::AOTIRInlineIndex*)file->second.CachedFileEntry;
|
||||
|
||||
if (Mod == nullptr) {
|
||||
file->second.CachedFileEntry = Mod = AOTIRCache[file->second.fileid].Array;
|
||||
}
|
||||
|
||||
if (Mod != nullptr)
|
||||
{
|
||||
auto AOTEntry = Mod->Find(GuestRIP - file->second.Start + file->second.Offset);
|
||||
|
||||
if (AOTEntry) {
|
||||
// verify hash
|
||||
auto MappedStart = GuestRIP;
|
||||
auto hash = XXH3_64bits((void*)MappedStart, AOTEntry->GuestLength);
|
||||
if (hash == AOTEntry->GuestHash) {
|
||||
Result.IRList = AOTEntry->GetIRData();
|
||||
//LogMan::Msg::DFmt("using {} + {:x} -> {:x}\n", file->second.fileid, AOTEntry->first, GuestRIP);
|
||||
|
||||
Result.RAData = AOTEntry->GetRAData();;
|
||||
Result.DebugData = new FEXCore::Core::DebugData();
|
||||
Result.StartAddr = MappedStart;
|
||||
Result.Length = AOTEntry->GuestLength;
|
||||
Result.GeneratedIR = true;
|
||||
} else {
|
||||
LogMan::Msg::IFmt("AOTIR: hash check failed {:x}\n", MappedStart);
|
||||
}
|
||||
} else {
|
||||
//LogMan::Msg::IFmt("AOTIR: Failed to find {:x}, {:x}, {}\n", GuestRIP, GuestRIP - file->second.Start + file->second.Offset, file->second.fileid);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return Result;
|
||||
}
|
||||
|
||||
bool AOTIRCaptureCache::PostCompileCode(
|
||||
FEXCore::Core::InternalThreadState *Thread,
|
||||
void* CodePtr,
|
||||
uint64_t GuestRIP,
|
||||
uint64_t StartAddr,
|
||||
uint64_t Length,
|
||||
FEXCore::IR::RegisterAllocationData *RAData,
|
||||
FEXCore::IR::IRListView *IRList,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
bool GeneratedIR,
|
||||
bool DecrementRefCount) {
|
||||
// Both generated ir and LibraryJITName need a named region lookup
|
||||
if (GeneratedIR || CTX->Config.LibraryJITNaming()) {
|
||||
std::shared_lock lk(AOTIRCacheLock);
|
||||
|
||||
auto file = FindAddrForFile(StartAddr, Length);
|
||||
|
||||
// Only go down this path if we actually found a library region
|
||||
if (file != AddrToFile.end()) {
|
||||
if (DebugData && CTX->Config.LibraryJITNaming()) {
|
||||
CTX->Symbols.RegisterNamedRegion(CodePtr, DebugData->HostCodeSize, file->second.filename);
|
||||
}
|
||||
|
||||
// Add to AOT cache if aot generation is enabled
|
||||
if (GeneratedIR && RAData &&
|
||||
(CTX->Config.AOTIRCapture() || CTX->Config.AOTIRGenerate())) {
|
||||
|
||||
auto hash = XXH3_64bits((void*)StartAddr, Length);
|
||||
|
||||
auto LocalRIP = GuestRIP - file->second.Start + file->second.Offset;
|
||||
auto LocalStartAddr = StartAddr - file->second.Start + file->second.Offset;
|
||||
auto fileid = file->second.fileid;
|
||||
AOTIRCaptureCacheWriteoutQueue_Append([this, LocalRIP, LocalStartAddr, Length, hash, IRList, RAData, fileid]() {
|
||||
auto *AotFile = &AOTIRCaptureCacheMap[fileid];
|
||||
|
||||
if (!AotFile->Stream) {
|
||||
AotFile->Stream = AOTIRWriter(fileid);
|
||||
uint64_t tag = FEXCore::IR::AOTIR_COOKIE;
|
||||
AotFile->Stream->write((char*)&tag, sizeof(tag));
|
||||
}
|
||||
AotFile->AppendAOTIRCaptureCache(LocalRIP, LocalStartAddr, Length, hash, IRList, RAData);
|
||||
});
|
||||
|
||||
if (CTX->Config.AOTIRGenerate()) {
|
||||
// cleanup memory and early exit here -- we're not running the application
|
||||
|
||||
if (DecrementRefCount) {
|
||||
--Thread->CompileBlockReentrantRefCount;
|
||||
}
|
||||
|
||||
Thread->CPUBackend->ClearCache();
|
||||
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Insert to caches if we generated IR
|
||||
if (GeneratedIR) {
|
||||
// Add to thread local ir cache
|
||||
Core::LocalIREntry Entry = {StartAddr, Length, decltype(Entry.IR)(IRList), decltype(Entry.RAData)(RAData), decltype(Entry.DebugData)(DebugData)};
|
||||
Thread->LocalIRCache.insert({GuestRIP, std::move(Entry)});
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
AOTIRCaptureCache::AddrToFileMapType::iterator AOTIRCaptureCache::FindAddrForFile(uint64_t Entry, uint64_t Length) {
|
||||
// Thread safety here! We are returning an iterator to the map object
|
||||
// This needs the AOTIRCacheLock locked prior to coming in to the function
|
||||
auto file = AddrToFile.lower_bound(Entry);
|
||||
if (file != AddrToFile.begin()) {
|
||||
--file;
|
||||
if (file->second.Start <= Entry && (file->second.Start + file->second.Len) >= (Entry + Length)) {
|
||||
return file;
|
||||
}
|
||||
}
|
||||
return AddrToFile.end();
|
||||
}
|
||||
|
||||
void AOTIRCaptureCache::AddNamedRegion(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename) {
|
||||
// TODO: Support overlapping maps and region splitting
|
||||
auto base_filename = std::filesystem::path(filename).filename().string();
|
||||
|
||||
if (!base_filename.empty()) {
|
||||
auto filename_hash = XXH3_64bits(filename.c_str(), filename.size());
|
||||
|
||||
auto fileid = base_filename + "-" + std::to_string(filename_hash) + "-";
|
||||
|
||||
// append optimization flags to the fileid
|
||||
fileid += (CTX->Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL) ? "S" : "s";
|
||||
fileid += CTX->Config.TSOEnabled ? "T" : "t";
|
||||
fileid += CTX->Config.ABILocalFlags ? "L" : "l";
|
||||
fileid += CTX->Config.ABINoPF ? "p" : "P";
|
||||
|
||||
std::unique_lock lk(AOTIRCacheLock);
|
||||
|
||||
AddrToFile.insert({ Base, { Base, Size, Offset, fileid, filename, nullptr, false} });
|
||||
|
||||
if (CTX->Config.AOTIRLoad && !AOTIRCache.contains(fileid) && AOTIRLoader) {
|
||||
auto streamfd = AOTIRLoader(fileid);
|
||||
if (streamfd != -1) {
|
||||
FEXCore::IR::LoadAOTIRCache(&AOTIRCache, streamfd);
|
||||
close(streamfd);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void AOTIRCaptureCache::RemoveNamedRegion(uintptr_t Base, uintptr_t Size) {
|
||||
std::unique_lock lk(AOTIRCacheLock);
|
||||
// TODO: Support partial removing
|
||||
AddrToFile.erase(Base);
|
||||
}
|
||||
}
|
||||
+162
@@ -0,0 +1,162 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cstdint>
|
||||
#include <functional>
|
||||
#include <fstream>
|
||||
#include <memory>
|
||||
#include <map>
|
||||
#include <unordered_map>
|
||||
#include <shared_mutex>
|
||||
#include <queue>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct DebugData;
|
||||
}
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class RegisterAllocationData;
|
||||
class IRListView;
|
||||
|
||||
constexpr auto COOKIE_VERSION = [](const char CookieText[4], uint32_t Version) {
|
||||
uint64_t Cookie = Version;
|
||||
Cookie <<= 32;
|
||||
|
||||
// Make the cookie text be the lower bits
|
||||
Cookie |= CookieText[3];
|
||||
Cookie <<= 8;
|
||||
Cookie |= CookieText[2];
|
||||
Cookie <<= 8;
|
||||
Cookie |= CookieText[1];
|
||||
Cookie <<= 8;
|
||||
Cookie |= CookieText[0];
|
||||
|
||||
return Cookie;
|
||||
};
|
||||
constexpr static uint32_t AOTIR_VERSION = 0x0000'00004;
|
||||
constexpr static uint64_t AOTIR_COOKIE = COOKIE_VERSION("FEXI", AOTIR_VERSION);
|
||||
|
||||
struct AOTIRInlineEntry {
|
||||
uint64_t GuestHash;
|
||||
uint64_t GuestLength;
|
||||
|
||||
/* RAData followed by IRData */
|
||||
uint8_t InlineData[0];
|
||||
|
||||
IR::RegisterAllocationData *GetRAData();
|
||||
IR::IRListView *GetIRData();
|
||||
};
|
||||
|
||||
struct AOTIRInlineIndexEntry {
|
||||
uint64_t GuestStart;
|
||||
uint64_t DataOffset;
|
||||
};
|
||||
|
||||
struct AOTIRInlineIndex {
|
||||
uint64_t Count;
|
||||
uint64_t DataBase;
|
||||
AOTIRInlineIndexEntry Entries[0];
|
||||
|
||||
AOTIRInlineEntry *Find(uint64_t GuestStart);
|
||||
AOTIRInlineEntry *GetInlineEntry(uint64_t DataOffset);
|
||||
};
|
||||
|
||||
struct AOTIRCaptureCacheEntry {
|
||||
std::unique_ptr<std::ofstream> Stream;
|
||||
std::map<uint64_t, uint64_t> Index;
|
||||
|
||||
void AppendAOTIRCaptureCache(uint64_t GuestRIP, uint64_t Start, uint64_t Length, uint64_t Hash, FEXCore::IR::IRListView *IRList, FEXCore::IR::RegisterAllocationData *RAData);
|
||||
};
|
||||
|
||||
struct AOTIRCacheEntry {
|
||||
AOTIRInlineIndex *Array;
|
||||
void *mapping;
|
||||
size_t size;
|
||||
};
|
||||
|
||||
using AOTCacheType = std::unordered_map<std::string, FEXCore::IR::AOTIRCacheEntry>;
|
||||
bool LoadAOTIRCache(AOTCacheType *AOTIRCache, int streamfd);
|
||||
|
||||
class AOTIRCaptureCache final {
|
||||
public:
|
||||
|
||||
AOTIRCaptureCache(FEXCore::Context::Context *ctx) : CTX {ctx} {}
|
||||
~AOTIRCaptureCache();
|
||||
|
||||
void FinalizeAOTIRCache();
|
||||
void AOTIRCaptureCacheWriteoutQueue_Flush();
|
||||
void AOTIRCaptureCacheWriteoutQueue_Append(const std::function<void()> &fn);
|
||||
void WriteFilesWithCode(std::function<void(const std::string& fileid, const std::string& filename)> Writer);
|
||||
|
||||
struct PreGenerateIRFetchResult {
|
||||
FEXCore::IR::IRListView *IRList {};
|
||||
FEXCore::IR::RegisterAllocationData *RAData {};
|
||||
FEXCore::Core::DebugData *DebugData {};
|
||||
uint64_t StartAddr {};
|
||||
uint64_t Length {};
|
||||
bool GeneratedIR {};
|
||||
};
|
||||
[[nodiscard]] PreGenerateIRFetchResult PreGenerateIRFetch(uint64_t GuestRIP, FEXCore::IR::IRListView *IRList);
|
||||
|
||||
bool PostCompileCode(FEXCore::Core::InternalThreadState *Thread,
|
||||
void* CodePtr,
|
||||
uint64_t GuestRIP,
|
||||
uint64_t StartAddr,
|
||||
uint64_t Length,
|
||||
FEXCore::IR::RegisterAllocationData *RAData,
|
||||
FEXCore::IR::IRListView *IRList,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
bool GeneratedIR,
|
||||
bool DecrementRefCount);
|
||||
|
||||
void AddNamedRegion(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const std::string &filename);
|
||||
void RemoveNamedRegion(uintptr_t Base, uintptr_t Size);
|
||||
|
||||
// Callbacks
|
||||
void SetAOTIRLoader(std::function<int(const std::string&)> CacheReader) {
|
||||
AOTIRLoader = CacheReader;
|
||||
}
|
||||
|
||||
void SetAOTIRWriter(std::function<std::unique_ptr<std::ofstream>(const std::string&)> CacheWriter) {
|
||||
AOTIRWriter = CacheWriter;
|
||||
}
|
||||
|
||||
void SetAOTIRRenamer(std::function<void(const std::string&)> CacheRenamer) {
|
||||
AOTIRRenamer = CacheRenamer;
|
||||
}
|
||||
|
||||
private:
|
||||
FEXCore::Context::Context *CTX;
|
||||
|
||||
std::shared_mutex AOTIRCacheLock;
|
||||
std::shared_mutex AOTIRCaptureCacheWriteoutLock;
|
||||
std::atomic<bool> AOTIRCaptureCacheWriteoutFlusing;
|
||||
|
||||
std::queue<std::function<void()>> AOTIRCaptureCacheWriteoutQueue;
|
||||
|
||||
std::map<std::string, std::string> FilesWithCode;
|
||||
|
||||
struct AddrToFileEntry {
|
||||
uint64_t Start;
|
||||
uint64_t Len;
|
||||
uint64_t Offset;
|
||||
std::string fileid;
|
||||
std::string filename;
|
||||
void *CachedFileEntry;
|
||||
bool ContainsCode;
|
||||
};
|
||||
|
||||
using AddrToFileMapType = std::map<uint64_t, AddrToFileEntry>;
|
||||
AddrToFileMapType AddrToFile;
|
||||
FEXCore::IR::AOTCacheType AOTIRCache;
|
||||
|
||||
std::function<int(const std::string&)> AOTIRLoader;
|
||||
std::function<std::unique_ptr<std::ofstream>(const std::string&)> AOTIRWriter;
|
||||
std::function<void(const std::string&)> AOTIRRenamer;
|
||||
std::unordered_map<std::string, FEXCore::IR::AOTIRCaptureCacheEntry> AOTIRCaptureCacheMap;
|
||||
|
||||
AddrToFileMapType::iterator FindAddrForFile(uint64_t Entry, uint64_t Length);
|
||||
};
|
||||
}
|
||||
+84
-3
@@ -191,6 +191,19 @@
|
||||
]
|
||||
},
|
||||
|
||||
"ProcessorID": {
|
||||
"Desc": ["Returns the processor ID correlating to the current running CPU",
|
||||
"This may be out of date by time this instruction is executed so care must be taken",
|
||||
"This same information can be gotten from syscall getcpu(&cpu, &node)",
|
||||
"uint32_t Res = (node << 12) | cpu;",
|
||||
"This means it has a limitation of 4096 CPU cores. Which is fine and matches x86 behaviour"
|
||||
],
|
||||
"OpClass": "Misc",
|
||||
"HasDest": true,
|
||||
"DestClass": "GPR",
|
||||
"DestSize": 8
|
||||
},
|
||||
|
||||
"SignalReturn": {
|
||||
"HasSideEffects": true,
|
||||
"OpClass": "Branch"
|
||||
@@ -293,7 +306,9 @@
|
||||
},
|
||||
|
||||
"EntrypointOffset": {
|
||||
"Desc": ["Returns the <entrypoint> + Offset address"],
|
||||
"Desc": ["Returns the <entrypoint> + Offset address",
|
||||
"When the size is 4 bytes then 32-bit overflow and underflow needs to work"
|
||||
],
|
||||
"OpClass": "ALU",
|
||||
"HasDest": true,
|
||||
"DestClass": "GPR",
|
||||
@@ -307,7 +322,9 @@
|
||||
},
|
||||
|
||||
"InlineEntrypointOffset": {
|
||||
"Desc": ["Returns the <entrypoint> + Offset address"],
|
||||
"Desc": ["Returns the <entrypoint> + Offset address",
|
||||
"When the size is 4 bytes then 32-bit overflow and underflow needs to work"
|
||||
],
|
||||
"OpClass": "ALU",
|
||||
"DestSize": "RegisterSize",
|
||||
"Args": [
|
||||
@@ -858,7 +875,22 @@
|
||||
},
|
||||
|
||||
"CacheLineClear": {
|
||||
"Desc": ["Does a 64 byte cacheline clear at the address specified"
|
||||
"Desc": ["Does a 64 byte cacheline clear at the address specified",
|
||||
"Only clears the data cachelines. Doesn't do any zeroing"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"OpClass": "Memory",
|
||||
"SSAArgs": "1",
|
||||
"SSANames": [
|
||||
"Addr"
|
||||
]
|
||||
},
|
||||
|
||||
"CacheLineZero": {
|
||||
"Desc": ["Does a 64 byte zero at the address specified",
|
||||
"Writing zeroes to memory",
|
||||
"It is specifically non-temporal and weakly ordered",
|
||||
"This matches CLZero behaviour"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"OpClass": "Memory",
|
||||
@@ -1074,6 +1106,38 @@
|
||||
]
|
||||
},
|
||||
|
||||
"PDep": {
|
||||
"Desc": [
|
||||
"Performs a parallel bit deposit.",
|
||||
"Takes the contiguous low-order bits and deposits them into",
|
||||
"the destination at the locations specified by the Mask."
|
||||
],
|
||||
"OpClass": "ALU",
|
||||
"HasDest": true,
|
||||
"DestClass": "GPR",
|
||||
"SSAArgs": "2",
|
||||
"SSANames": [
|
||||
"Input",
|
||||
"Mask"
|
||||
]
|
||||
},
|
||||
|
||||
"PExt": {
|
||||
"Desc": [
|
||||
"Performs a parallel bit extract.",
|
||||
"Each bit set in the mask will select the corresponding bit in the Input",
|
||||
"and transfers them to the lower contiguous bits in the destination."
|
||||
],
|
||||
"OpClass": "ALU",
|
||||
"HasDest": true,
|
||||
"DestClass": "GPR",
|
||||
"SSAArgs": "2",
|
||||
"SSANames": [
|
||||
"Input",
|
||||
"Mask"
|
||||
]
|
||||
},
|
||||
|
||||
"LDiv": {
|
||||
"Desc": ["Integer long signed division returning lower bits",
|
||||
"The Lower and Upper registers will be concated together to generate a dividend twice the size",
|
||||
@@ -3491,6 +3555,23 @@
|
||||
]
|
||||
},
|
||||
|
||||
"CRC32": {
|
||||
"Desc": ["CRC32 using polynomial 0x1EDC6F41"
|
||||
],
|
||||
"OpClass": "Crypto",
|
||||
"HasDest": true,
|
||||
"DestSize": "std::max<uint8_t>(4, GetOpSize(ssa0))",
|
||||
"DestClass": "GPR",
|
||||
"SSAArgs": "2",
|
||||
"SSANames": [
|
||||
"Src1",
|
||||
"Src2"
|
||||
],
|
||||
"Args": [
|
||||
"uint8_t", "SrcSize"
|
||||
]
|
||||
},
|
||||
|
||||
"GetHostFlag": {
|
||||
"OpClass": "Flags",
|
||||
"HasDest": true,
|
||||
|
||||
+2
-2
@@ -225,7 +225,7 @@ void Dump(std::stringstream *out, IRListView const* IR, IR::RegisterAllocationDa
|
||||
NumElements /= ElementSize;
|
||||
}
|
||||
|
||||
*out << "%ssa" << ID;
|
||||
*out << "%ssa" << std::dec << ID;
|
||||
|
||||
if (RAData) {
|
||||
auto PhyReg = RAData->GetNodeRegister(ID);
|
||||
@@ -265,7 +265,7 @@ void Dump(std::stringstream *out, IRListView const* IR, IR::RegisterAllocationDa
|
||||
NumElements = IROp->Size / ElementSize;
|
||||
}
|
||||
|
||||
*out << "(%ssa" << ID << ' ';
|
||||
*out << "(%ssa" << std::dec << ID << ' ';
|
||||
*out << 'i' << std::dec << (ElementSize * 8);
|
||||
if (NumElements > 1) {
|
||||
*out << 'v' << std::dec << NumElements;
|
||||
|
||||
+3
-1
@@ -47,10 +47,12 @@ namespace {
|
||||
return (Type & ACCESS_TYPE_MASK) == ACCESS_READ;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool IsInvalidAccess(LastAccessType Type) {
|
||||
return (Type & ACCESS_TYPE_MASK) == ACCESS_INVALID;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool IsPartialAccess(LastAccessType Type) {
|
||||
return (Type & ACCESS_PARTIAL) == ACCESS_PARTIAL;
|
||||
}
|
||||
@@ -254,7 +256,7 @@ namespace {
|
||||
});
|
||||
|
||||
|
||||
size_t ClassifiedStructSize{};
|
||||
[[maybe_unused]] size_t ClassifiedStructSize{};
|
||||
ContextClassificationInfo->Lookup.reserve(sizeof(FEXCore::Core::CPUState));
|
||||
for (auto &it : *ContextClassification) {
|
||||
LOGMAN_THROW_A_FMT(it.Class.Offset == ContextClassificationInfo->Lookup.size(), "Offset mismatch (offset={})", it.Class.Offset);
|
||||
|
||||
@@ -77,7 +77,7 @@ struct FPRInfo {
|
||||
bool IsFPR(uint32_t Offset) {
|
||||
|
||||
auto begin = offsetof(FEXCore::Core::CpuStateFrame, State.xmm[0][0]);
|
||||
auto end = offsetof(FEXCore::Core::CpuStateFrame, State.xmm[17][0]);
|
||||
auto end = offsetof(FEXCore::Core::CpuStateFrame, State.xmm[16][0]);
|
||||
|
||||
if (Offset < begin || Offset >= end)
|
||||
return false;
|
||||
|
||||
@@ -570,10 +570,10 @@ namespace {
|
||||
// Get SRA Reg and Class from a Context offset
|
||||
auto GetRegAndClassFromOffset = [](uint32_t Offset) {
|
||||
auto beginGpr = offsetof(FEXCore::Core::CpuStateFrame, State.gregs[0]);
|
||||
auto endGpr = offsetof(FEXCore::Core::CpuStateFrame, State.gregs[17]);
|
||||
auto endGpr = offsetof(FEXCore::Core::CpuStateFrame, State.gregs[16]);
|
||||
|
||||
auto beginFpr = offsetof(FEXCore::Core::CpuStateFrame, State.xmm[0][0]);
|
||||
auto endFpr = offsetof(FEXCore::Core::CpuStateFrame, State.xmm[17][0]);
|
||||
auto endFpr = offsetof(FEXCore::Core::CpuStateFrame, State.xmm[16][0]);
|
||||
|
||||
if (Offset >= beginGpr && Offset < endGpr) {
|
||||
auto reg = (Offset - beginGpr) / 8;
|
||||
@@ -594,10 +594,10 @@ namespace {
|
||||
// Get a StaticMap entry from context offset
|
||||
const auto GetStaticMapFromOffset = [&](uint32_t Offset) -> LiveRange** {
|
||||
auto beginGpr = offsetof(FEXCore::Core::CpuStateFrame, State.gregs[0]);
|
||||
auto endGpr = offsetof(FEXCore::Core::CpuStateFrame, State.gregs[17]);
|
||||
auto endGpr = offsetof(FEXCore::Core::CpuStateFrame, State.gregs[16]);
|
||||
|
||||
auto beginFpr = offsetof(FEXCore::Core::CpuStateFrame, State.xmm[0][0]);
|
||||
auto endFpr = offsetof(FEXCore::Core::CpuStateFrame, State.xmm[17][0]);
|
||||
auto endFpr = offsetof(FEXCore::Core::CpuStateFrame, State.xmm[16][0]);
|
||||
|
||||
if (Offset >= beginGpr && Offset < endGpr) {
|
||||
auto reg = (Offset - beginGpr) / 8;
|
||||
@@ -1115,8 +1115,6 @@ namespace {
|
||||
auto InterferenceNodeOpBeginIter = IR.at(InterferenceLiveRange->Begin);
|
||||
auto InterferenceNodeOpEndIter = IR.at(InterferenceLiveRange->End);
|
||||
|
||||
bool Found{};
|
||||
|
||||
// If the nodes live range is entirely encompassed by the interference node's range
|
||||
// then spilling that range will /potentially/ lower RA
|
||||
// Will only lower register pressure if the interference node does NOT have a use inside of
|
||||
@@ -1153,7 +1151,6 @@ namespace {
|
||||
|
||||
const auto NextUseDistance = InterferenceNodeNextUse.ID().Value - CurrentLocation.Value;
|
||||
if (NextUseDistance >= InterferenceFarthestNextUse) {
|
||||
Found = true;
|
||||
InterferenceIdToSpill = InterferenceNode;
|
||||
InterferenceFarthestNextUse = NextUseDistance;
|
||||
}
|
||||
|
||||
+2
-2
@@ -25,7 +25,7 @@ public:
|
||||
|
||||
bool IsStaticAllocGpr(uint32_t Offset, RegisterClassType Class) {
|
||||
const auto begin = offsetof(FEXCore::Core::CPUState, gregs[0]);
|
||||
const auto end = offsetof(FEXCore::Core::CPUState, gregs[17]);
|
||||
const auto end = offsetof(FEXCore::Core::CPUState, gregs[16]);
|
||||
|
||||
if (Offset >= begin && Offset < end) {
|
||||
const auto reg = (Offset - begin) / 8;
|
||||
@@ -40,7 +40,7 @@ bool IsStaticAllocGpr(uint32_t Offset, RegisterClassType Class) {
|
||||
|
||||
bool IsStaticAllocFpr(uint32_t Offset, RegisterClassType Class, bool AllowGpr) {
|
||||
const auto begin = offsetof(FEXCore::Core::CPUState, xmm[0][0]);
|
||||
const auto end = offsetof(FEXCore::Core::CPUState, xmm[17][0]);
|
||||
const auto end = offsetof(FEXCore::Core::CPUState, xmm[16][0]);
|
||||
|
||||
if (Offset >= begin && Offset < end) {
|
||||
const auto reg = (Offset - begin) / 16;
|
||||
|
||||
+3
@@ -2,6 +2,9 @@
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include <array>
|
||||
#include <sys/mman.h>
|
||||
#ifdef ENABLE_JEMALLOC
|
||||
#include <jemalloc/jemalloc.h>
|
||||
|
||||
@@ -4,6 +4,8 @@
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXHeaderUtils/ScopedSignalMask.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
@@ -17,7 +19,6 @@
|
||||
#include <new>
|
||||
#include <sstream>
|
||||
#include <sys/mman.h>
|
||||
#include <bits/mman-map-flags-generic.h>
|
||||
#include <sys/utsname.h>
|
||||
#include <type_traits>
|
||||
#include <utility>
|
||||
@@ -183,7 +184,7 @@ void *OSAllocator_64Bit::Mmap(void *addr, size_t length, int prot, int flags, in
|
||||
size_t NumberOfPages = length / PAGE_SIZE;
|
||||
|
||||
// This needs a mutex to be thread safe
|
||||
std::scoped_lock<std::mutex> lk{AllocationMutex};
|
||||
FHU::ScopedSignalMaskWithMutex lk(AllocationMutex);
|
||||
|
||||
uint64_t AllocatedOffset{};
|
||||
LiveVMARegion *LiveRegion{};
|
||||
@@ -442,7 +443,7 @@ int OSAllocator_64Bit::Munmap(void *addr, size_t length) {
|
||||
}
|
||||
|
||||
// This needs a mutex to be thread safe
|
||||
std::scoped_lock<std::mutex> lk{AllocationMutex};
|
||||
FHU::ScopedSignalMaskWithMutex lk(AllocationMutex);
|
||||
|
||||
length = FEXCore::AlignUp(length, PAGE_SIZE);
|
||||
|
||||
@@ -621,8 +622,8 @@ OSAllocator_64Bit::OSAllocator_64Bit() {
|
||||
}
|
||||
|
||||
OSAllocator_64Bit::~OSAllocator_64Bit() {
|
||||
// For consistency, pull the mutex
|
||||
std::scoped_lock<std::mutex> lk{AllocationMutex};
|
||||
// This needs a mutex to be thread safe
|
||||
FHU::ScopedSignalMaskWithMutex lk(AllocationMutex);
|
||||
|
||||
// Walk the pages and deallocate
|
||||
// First walk the live regions
|
||||
|
||||
+42
-11
@@ -1,11 +1,42 @@
|
||||
#include <FEXCore/Utils/NetStream.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstring>
|
||||
#include <iterator>
|
||||
#include <sys/socket.h>
|
||||
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEXCore::Utils {
|
||||
int NetStream::NetBuf::flushBuffer(const char *buffer, size_t size) {
|
||||
namespace {
|
||||
class NetBuf final : public std::streambuf {
|
||||
public:
|
||||
explicit NetBuf(int socketfd) : socket{socketfd} {
|
||||
reset_output_buffer();
|
||||
}
|
||||
~NetBuf() override {
|
||||
close(socket);
|
||||
}
|
||||
|
||||
private:
|
||||
std::streamsize xsputn(const char* buffer, std::streamsize size) override;
|
||||
|
||||
std::streambuf::int_type underflow() override;
|
||||
std::streambuf::int_type overflow(std::streambuf::int_type ch) override;
|
||||
int sync() override;
|
||||
|
||||
void reset_output_buffer() {
|
||||
// we always leave room for one extra char
|
||||
setp(std::begin(output_buffer), std::end(output_buffer) -1);
|
||||
}
|
||||
|
||||
int flushBuffer(const char *buffer, size_t size);
|
||||
|
||||
int socket;
|
||||
std::array<char, 1400> output_buffer;
|
||||
std::array<char, 1500> input_buffer; // enough for a typical packet
|
||||
};
|
||||
|
||||
int NetBuf::flushBuffer(const char *buffer, size_t size) {
|
||||
size_t total = 0;
|
||||
|
||||
// Send data
|
||||
@@ -21,7 +52,7 @@ int NetStream::NetBuf::flushBuffer(const char *buffer, size_t size) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
std::streamsize NetStream::NetBuf::xsputn(const char* buffer, std::streamsize size) {
|
||||
std::streamsize NetBuf::xsputn(const char* buffer, std::streamsize size) {
|
||||
size_t buf_remaining = epptr() - pptr();
|
||||
|
||||
// Check if the string fits neatly in our buffer
|
||||
@@ -45,23 +76,23 @@ std::streamsize NetStream::NetBuf::xsputn(const char* buffer, std::streamsize si
|
||||
}
|
||||
}
|
||||
|
||||
std::streambuf::int_type NetStream::NetBuf::overflow(std::streambuf::int_type ch) {
|
||||
std::streambuf::int_type NetBuf::overflow(std::streambuf::int_type ch) {
|
||||
// we always leave room for one extra char
|
||||
*pptr() = (char) ch;
|
||||
pbump(1);
|
||||
return sync();
|
||||
}
|
||||
|
||||
int NetStream::NetBuf::sync() {
|
||||
int NetBuf::sync() {
|
||||
// Flush and reset output buffer to zero
|
||||
if(flushBuffer(pbase(), pptr() - pbase()) < 0) {
|
||||
if (flushBuffer(pbase(), pptr() - pbase()) < 0) {
|
||||
return -1;
|
||||
}
|
||||
reset_output_buffer();
|
||||
return 0;
|
||||
}
|
||||
|
||||
std::streambuf::int_type NetStream::NetBuf::underflow() {
|
||||
std::streambuf::int_type NetBuf::underflow() {
|
||||
ssize_t size = recv(socket, (void *)std::begin(input_buffer), sizeof(input_buffer), 0);
|
||||
|
||||
if (size <= 0) {
|
||||
@@ -73,12 +104,12 @@ std::streambuf::int_type NetStream::NetBuf::underflow() {
|
||||
|
||||
return traits_type::to_int_type(*gptr());
|
||||
}
|
||||
} // Anonymous namespace
|
||||
|
||||
NetStream::NetStream(int socketfd) : std::iostream(new NetBuf(socketfd)) {}
|
||||
|
||||
NetStream::~NetStream() {
|
||||
delete rdbuf();
|
||||
}
|
||||
|
||||
NetStream::NetBuf::~NetBuf() {
|
||||
close(socket);
|
||||
}
|
||||
}
|
||||
} // namespace FEXCore::Utils
|
||||
-1
@@ -12,7 +12,6 @@
|
||||
#include <sys/mman.h>
|
||||
#include <sys/signal.h>
|
||||
#include <sys/syscall.h>
|
||||
#include <bits/mman-map-flags-generic.h>
|
||||
#include <deque>
|
||||
#include <unistd.h>
|
||||
|
||||
|
||||
+12
-7
@@ -94,7 +94,7 @@ namespace FEXCore::Context {
|
||||
FEX_DEFAULT_VISIBILITY FEXCore::Core::InternalThreadState* InitCore(FEXCore::Context::Context *CTX, FEXCore::CodeLoader *Loader);
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void SetExitHandler(FEXCore::Context::Context *CTX, ExitHandler handler);
|
||||
FEX_DEFAULT_VISIBILITY ExitHandler GetExitHandler(FEXCore::Context::Context *CTX);
|
||||
FEX_DEFAULT_VISIBILITY ExitHandler GetExitHandler(const FEXCore::Context::Context *CTX);
|
||||
|
||||
/**
|
||||
* @brief Pauses execution on the CPU core
|
||||
@@ -134,7 +134,7 @@ namespace FEXCore::Context {
|
||||
*
|
||||
* @return The program exit status
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY int GetProgramStatus(FEXCore::Context::Context *CTX);
|
||||
FEX_DEFAULT_VISIBILITY int GetProgramStatus(const FEXCore::Context::Context *CTX);
|
||||
|
||||
/**
|
||||
* @brief Tells the core to shutdown
|
||||
@@ -157,7 +157,7 @@ namespace FEXCore::Context {
|
||||
*
|
||||
* @return The ExitReason for the parentthread
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY ExitReason GetExitReason(FEXCore::Context::Context *CTX);
|
||||
FEX_DEFAULT_VISIBILITY ExitReason GetExitReason(const FEXCore::Context::Context *CTX);
|
||||
|
||||
/**
|
||||
* @brief [[theadsafe]] Checks if the Context is either done working or paused(in the case of single stepping)
|
||||
@@ -168,7 +168,7 @@ namespace FEXCore::Context {
|
||||
*
|
||||
* @return true if the core is done or paused
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY bool IsDone(FEXCore::Context::Context *CTX);
|
||||
FEX_DEFAULT_VISIBILITY bool IsDone(const FEXCore::Context::Context *CTX);
|
||||
|
||||
/**
|
||||
* @brief Gets a copy the CPUState of the parent thread
|
||||
@@ -176,7 +176,8 @@ namespace FEXCore::Context {
|
||||
* @param CTX The context that we created
|
||||
* @param State The state object to populate
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY void GetCPUState(FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *State);
|
||||
FEX_DEFAULT_VISIBILITY void GetCPUState(const FEXCore::Context::Context *CTX,
|
||||
FEXCore::Core::CPUState *State);
|
||||
|
||||
/**
|
||||
* @brief Copies the CPUState provided to the parent thread
|
||||
@@ -184,7 +185,8 @@ namespace FEXCore::Context {
|
||||
* @param CTX The context that we created
|
||||
* @param State The satate object to copy from
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY void SetCPUState(FEXCore::Context::Context *CTX, FEXCore::Core::CPUState *State);
|
||||
FEX_DEFAULT_VISIBILITY void SetCPUState(FEXCore::Context::Context *CTX,
|
||||
const FEXCore::Core::CPUState *State);
|
||||
|
||||
/**
|
||||
* @brief Allows the frontend to pass in a custom CPUBackend creation factory
|
||||
@@ -232,11 +234,14 @@ namespace FEXCore::Context {
|
||||
FEX_DEFAULT_VISIBILITY void SetSignalDelegator(FEXCore::Context::Context *CTX, FEXCore::SignalDelegator *SignalDelegation);
|
||||
FEX_DEFAULT_VISIBILITY void SetSyscallHandler(FEXCore::Context::Context *CTX, FEXCore::HLE::SyscallHandler *Handler);
|
||||
FEX_DEFAULT_VISIBILITY FEXCore::CPUID::FunctionResults RunCPUIDFunction(FEXCore::Context::Context *CTX, uint32_t Function, uint32_t Leaf);
|
||||
FEX_DEFAULT_VISIBILITY FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(FEXCore::Context::Context *CTX, uint32_t Function, uint32_t Leaf, uint32_t CPU);
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void AddNamedRegion(FEXCore::Context::Context *CTX, uintptr_t Base, uintptr_t Length, uintptr_t Offset, const std::string& Name);
|
||||
FEX_DEFAULT_VISIBILITY void RemoveNamedRegion(FEXCore::Context::Context *CTX, uintptr_t Base, uintptr_t Length);
|
||||
FEX_DEFAULT_VISIBILITY void SetAOTIRLoader(FEXCore::Context::Context *CTX, std::function<int(const std::string&)> CacheReader);
|
||||
FEX_DEFAULT_VISIBILITY void SetAOTIRWriter(FEXCore::Context::Context *CTX, std::function<std::unique_ptr<std::ostream>(const std::string&)> CacheWriter);
|
||||
FEX_DEFAULT_VISIBILITY void SetAOTIRWriter(FEXCore::Context::Context *CTX, std::function<std::unique_ptr<std::ofstream>(const std::string&)> CacheWriter);
|
||||
FEX_DEFAULT_VISIBILITY void SetAOTIRRenamer(FEXCore::Context::Context *CTX, std::function<void(const std::string&)> CacheRenamer);
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void FinalizeAOTIRCache(FEXCore::Context::Context *CTX);
|
||||
FEX_DEFAULT_VISIBILITY void WriteFilesWithCode(FEXCore::Context::Context *CTX, std::function<void(const std::string& fileid, const std::string& filename)> Writer);
|
||||
FEX_DEFAULT_VISIBILITY void FlushCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length);
|
||||
|
||||
@@ -2,8 +2,11 @@
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <cstring>
|
||||
#include <signal.h>
|
||||
|
||||
namespace FEXCore {
|
||||
namespace x86_64 {
|
||||
@@ -151,6 +154,34 @@ namespace FEXCore {
|
||||
FEXCore::x86::sigval_t sigval;
|
||||
} _timer;
|
||||
} _sifields;
|
||||
|
||||
siginfo_t() = delete;
|
||||
|
||||
operator ::siginfo_t() const {
|
||||
::siginfo_t val{};
|
||||
val.si_signo = si_signo;
|
||||
val.si_errno = si_errno;
|
||||
val.si_code = si_code;
|
||||
|
||||
// Host siginfo has a pad member that is set to zeros
|
||||
val.__pad0 = 0;
|
||||
|
||||
// Copy over the union
|
||||
// The union is different sizes on 64-bit versus 32-bit
|
||||
memcpy(val._sifields._pad, _sifields.pad, std::min(sizeof(val._sifields._pad), sizeof(_sifields.pad)));
|
||||
|
||||
return val;
|
||||
}
|
||||
|
||||
siginfo_t(::siginfo_t val) {
|
||||
si_signo = val.si_signo;
|
||||
si_errno = val.si_errno;
|
||||
si_code = val.si_code;
|
||||
|
||||
// Copy over the union
|
||||
// The union is different sizes on 64-bit versus 32-bit
|
||||
memcpy(val._sifields._pad, _sifields.pad, std::min(sizeof(val._sifields._pad), sizeof(_sifields.pad)));
|
||||
}
|
||||
};
|
||||
static_assert(sizeof(FEXCore::x86::siginfo_t) == 128, "This needs to be the right size");
|
||||
|
||||
|
||||
@@ -5,6 +5,7 @@
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/InterruptableConditionVariable.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
|
||||
#include <unordered_map>
|
||||
@@ -83,7 +84,7 @@ namespace FEXCore::Core {
|
||||
std::atomic<SignalEvent> SignalReason{SignalEvent::Nothing};
|
||||
|
||||
std::unique_ptr<FEXCore::Threads::Thread> ExecutionThread;
|
||||
Event StartRunning;
|
||||
InterruptableConditionVariable StartRunning;
|
||||
Event ThreadWaiting;
|
||||
|
||||
std::unique_ptr<FEXCore::IR::OpDispatchBuilder> OpDispatcher;
|
||||
|
||||
+30
-18
@@ -2,10 +2,13 @@
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <cstring>
|
||||
#include <type_traits>
|
||||
|
||||
#include <fmt/format.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
///< Forward declaration of OpDispatchBuilder
|
||||
class OpDispatchBuilder;
|
||||
@@ -465,7 +468,7 @@ constexpr size_t MAX_INST_GROUP_TABLE_SIZE = 512;
|
||||
constexpr size_t MAX_INST_SECOND_GROUP_TABLE_SIZE = 512;
|
||||
constexpr size_t MAX_X87_TABLE_SIZE = 1 << 11;
|
||||
constexpr size_t MAX_SECOND_MODRM_TABLE_SIZE = 32;
|
||||
// 3 prefixes | 8 bit opcode
|
||||
// (3 bit prefixes) | 8 bit opcode
|
||||
constexpr size_t MAX_0F_38_TABLE_SIZE = (1 << 11);
|
||||
// 1 REX | 1 prefixes | 8 bit opcode
|
||||
constexpr size_t MAX_0F_3A_TABLE_SIZE = (1 << 11);
|
||||
@@ -487,27 +490,36 @@ constexpr size_t MAX_XOP_GROUP_TABLE_SIZE = (1 << 6);
|
||||
|
||||
constexpr size_t MAX_EVEX_TABLE_SIZE = 256;
|
||||
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo BaseOps[MAX_PRIMARY_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo SecondBaseOps[MAX_SECOND_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo RepModOps[MAX_REP_MOD_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo RepNEModOps[MAX_REPNE_MOD_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo OpSizeModOps[MAX_OPSIZE_MOD_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo PrimaryInstGroupOps[MAX_INST_GROUP_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo SecondInstGroupOps[MAX_INST_SECOND_GROUP_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo SecondModRMTableOps[MAX_SECOND_MODRM_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo X87Ops[MAX_X87_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo DDDNowOps[MAX_3DNOW_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo H0F38TableOps[MAX_0F_38_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo H0F3ATableOps[MAX_0F_3A_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps;
|
||||
extern FEX_DEFAULT_VISIBILITY std::array<X86InstInfo, MAX_SECOND_TABLE_SIZE> SecondBaseOps;
|
||||
extern FEX_DEFAULT_VISIBILITY std::array<X86InstInfo, MAX_REP_MOD_TABLE_SIZE> RepModOps;
|
||||
extern FEX_DEFAULT_VISIBILITY std::array<X86InstInfo, MAX_REPNE_MOD_TABLE_SIZE> RepNEModOps;
|
||||
extern FEX_DEFAULT_VISIBILITY std::array<X86InstInfo, MAX_OPSIZE_MOD_TABLE_SIZE> OpSizeModOps;
|
||||
|
||||
extern FEX_DEFAULT_VISIBILITY std::array<X86InstInfo, MAX_INST_GROUP_TABLE_SIZE> PrimaryInstGroupOps;
|
||||
extern FEX_DEFAULT_VISIBILITY std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGroupOps;
|
||||
extern FEX_DEFAULT_VISIBILITY std::array<X86InstInfo, MAX_SECOND_MODRM_TABLE_SIZE> SecondModRMTableOps;
|
||||
extern FEX_DEFAULT_VISIBILITY std::array<X86InstInfo, MAX_X87_TABLE_SIZE> X87Ops;
|
||||
extern FEX_DEFAULT_VISIBILITY std::array<X86InstInfo, MAX_3DNOW_TABLE_SIZE> DDDNowOps;
|
||||
extern FEX_DEFAULT_VISIBILITY std::array<X86InstInfo, MAX_0F_38_TABLE_SIZE> H0F38TableOps;
|
||||
extern FEX_DEFAULT_VISIBILITY std::array<X86InstInfo, MAX_0F_3A_TABLE_SIZE> H0F3ATableOps;
|
||||
|
||||
// VEX
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo VEXTableOps[MAX_VEX_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo VEXTableGroupOps[MAX_VEX_GROUP_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY std::array<X86InstInfo, MAX_VEX_TABLE_SIZE> VEXTableOps;
|
||||
extern FEX_DEFAULT_VISIBILITY std::array<X86InstInfo, MAX_VEX_GROUP_TABLE_SIZE> VEXTableGroupOps;
|
||||
|
||||
// XOP
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo XOPTableOps[MAX_XOP_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo XOPTableGroupOps[MAX_XOP_GROUP_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY std::array<X86InstInfo, MAX_XOP_TABLE_SIZE> XOPTableOps;
|
||||
extern FEX_DEFAULT_VISIBILITY std::array<X86InstInfo, MAX_XOP_GROUP_TABLE_SIZE> XOPTableGroupOps;
|
||||
|
||||
// EVEX
|
||||
extern FEX_DEFAULT_VISIBILITY X86InstInfo EVEXTableOps[MAX_EVEX_TABLE_SIZE];
|
||||
extern FEX_DEFAULT_VISIBILITY std::array<X86InstInfo, MAX_EVEX_TABLE_SIZE> EVEXTableOps;
|
||||
}
|
||||
|
||||
template <>
|
||||
struct fmt::formatter<FEXCore::X86Tables::DecodedOperand::OpType> : formatter<uint32_t> {
|
||||
template <typename FormatContext>
|
||||
auto format(FEXCore::X86Tables::DecodedOperand::OpType type, FormatContext& ctx) const {
|
||||
return fmt::formatter<uint32_t>::format(static_cast<uint32_t>(type), ctx);
|
||||
}
|
||||
};
|
||||
+1
-1
@@ -43,7 +43,7 @@ friend class FEXCore::IR::PassManager;
|
||||
|
||||
IRPair<IROp_Constant> _Constant(uint8_t Size, uint64_t Constant) {
|
||||
auto Op = AllocateOp<IROp_Constant, IROps::OP_CONSTANT>();
|
||||
uint64_t Mask = ~0ULL >> (Size - 64);
|
||||
uint64_t Mask = ~0ULL >> (64 - Size);
|
||||
Op.first->Constant = (Constant & Mask);
|
||||
Op.first->Header.Size = Size / 8;
|
||||
Op.first->Header.ElementSize = Size / 8;
|
||||
|
||||
@@ -28,7 +28,9 @@ union PhysicalRegister {
|
||||
|
||||
static_assert(sizeof(PhysicalRegister) == 1);
|
||||
|
||||
class RegisterAllocationData {
|
||||
// This class is serialized, can't have any holes in the structure
|
||||
// otherwise ASAN complains about reading uninitialized memory
|
||||
class FEX_PACKED RegisterAllocationData {
|
||||
public:
|
||||
uint32_t SpillSlotCount {};
|
||||
uint32_t MapCount {};
|
||||
@@ -43,6 +45,16 @@ class RegisterAllocationData {
|
||||
static size_t Size(uint32_t NodeCount) {
|
||||
return sizeof(RegisterAllocationData) + NodeCount * sizeof(Map[0]);
|
||||
}
|
||||
|
||||
void Serialize(std::ostream& stream) const {
|
||||
stream.write((const char*)&SpillSlotCount, sizeof(SpillSlotCount));
|
||||
stream.write((const char*)&MapCount, sizeof(MapCount));
|
||||
// RAData (inline)
|
||||
// In file, IsShared is always set
|
||||
bool _IsShared = true;
|
||||
stream.write((const char*)&_IsShared, sizeof(IsShared));
|
||||
stream.write((const char*)&Map[0], sizeof(Map[0]) * MapCount);
|
||||
}
|
||||
};
|
||||
|
||||
struct RegisterAllocationDataDeleter {
|
||||
|
||||
@@ -0,0 +1,89 @@
|
||||
#pragma once
|
||||
|
||||
#include <atomic>
|
||||
#include <chrono>
|
||||
#include <climits>
|
||||
#include <cstdint>
|
||||
#include <linux/futex.h>
|
||||
#include <sys/syscall.h>
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEXCore {
|
||||
/**
|
||||
* @brief A condition variable that is robust against use of longjmp in signal handlers.
|
||||
*
|
||||
* This is opposed to common `std::condition_variable` implementations:
|
||||
* Longjmp'ing in a signal handler while interrupting a pending `wait_for()`
|
||||
* call can leave the condition variable in an invalid state that breaks later
|
||||
* uses of that object and may cause hangs as a consequence.
|
||||
*/
|
||||
class InterruptableConditionVariable final {
|
||||
public:
|
||||
bool Wait(struct timespec *Timeout = nullptr) {
|
||||
while (true) {
|
||||
uint32_t Expected = SIGNALED;
|
||||
uint32_t Desired = UNSIGNALED;
|
||||
|
||||
// If the mutex was already signaled then we can early exit
|
||||
if (Mutex.compare_exchange_strong(Expected, Desired)) {
|
||||
return true;
|
||||
}
|
||||
|
||||
constexpr int Op = FUTEX_WAIT | FUTEX_PRIVATE_FLAG;
|
||||
// WAIT will keep sleeping on the futex word while it is `val`
|
||||
int Result = ::syscall(SYS_futex,
|
||||
&Mutex,
|
||||
Op,
|
||||
Desired, // val
|
||||
Timeout, // Timeout/val2
|
||||
nullptr, // Addr2
|
||||
0); // val3
|
||||
|
||||
if (Timeout && Result == -1 && errno == ETIMEDOUT) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template<class Rep, class Period>
|
||||
bool WaitFor(std::chrono::duration<Rep, Period> const& time) {
|
||||
struct timespec Timeout{};
|
||||
auto SecondsDuration = std::chrono::duration_cast<std::chrono::seconds>(time);
|
||||
Timeout.tv_sec = SecondsDuration.count();
|
||||
Timeout.tv_nsec = std::chrono::duration_cast<std::chrono::nanoseconds>(time - SecondsDuration).count();
|
||||
return Wait(&Timeout);
|
||||
}
|
||||
|
||||
void NotifyOne() {
|
||||
DoNotify(1);
|
||||
}
|
||||
|
||||
void NotifyAll() {
|
||||
// Maximum number of waiters
|
||||
DoNotify(INT_MAX);
|
||||
}
|
||||
|
||||
private:
|
||||
std::atomic<uint32_t> Mutex{};
|
||||
constexpr static uint32_t SIGNALED = 1;
|
||||
constexpr static uint32_t UNSIGNALED = 0;
|
||||
|
||||
void DoNotify(int Waiters) {
|
||||
uint32_t Expected = UNSIGNALED;
|
||||
uint32_t Desired = SIGNALED;
|
||||
|
||||
// If the mutex was in an unsignaled state then signal
|
||||
if (Mutex.compare_exchange_strong(Expected, Desired)) {
|
||||
constexpr int Op = FUTEX_WAKE | FUTEX_PRIVATE_FLAG;
|
||||
|
||||
::syscall(SYS_futex,
|
||||
&Mutex,
|
||||
Op,
|
||||
Waiters, // val - Number of waiters to wake
|
||||
0, // val2
|
||||
&Mutex, // Addr2 - Mutex to do the operation on
|
||||
0); // val3
|
||||
}
|
||||
}
|
||||
};
|
||||
}
|
||||
+3
-36
@@ -2,45 +2,12 @@
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
#include <array>
|
||||
#include <iostream>
|
||||
#include <iterator>
|
||||
#include <string.h>
|
||||
|
||||
namespace FEXCore::Utils {
|
||||
class FEX_DEFAULT_VISIBILITY NetStream : public std::iostream {
|
||||
public:
|
||||
NetStream(int socketfd) : std::iostream(new NetBuf(socketfd)) {}
|
||||
virtual ~NetStream();
|
||||
|
||||
private:
|
||||
class NetBuf : public std::streambuf {
|
||||
|
||||
public:
|
||||
NetBuf(int socketfd) {
|
||||
socket = socketfd;
|
||||
reset_output_buffer();
|
||||
}
|
||||
virtual ~NetBuf();
|
||||
|
||||
protected:
|
||||
virtual std::streamsize xsputn(const char* buffer, std::streamsize size);
|
||||
|
||||
virtual std::streambuf::int_type underflow();
|
||||
virtual std::streambuf::int_type overflow(std::streambuf::int_type ch);
|
||||
virtual int sync();
|
||||
|
||||
private:
|
||||
void reset_output_buffer() {
|
||||
// we always leave room for one extra char
|
||||
setp(std::begin(output_buffer), std::end(output_buffer) -1);
|
||||
}
|
||||
|
||||
int flushBuffer(const char *buffer, size_t size);
|
||||
|
||||
int socket;
|
||||
std::array<char, 1400> output_buffer;
|
||||
std::array<char, 1500> input_buffer; // enough for a typical packet
|
||||
};
|
||||
explicit NetStream(int socketfd);
|
||||
~NetStream() override;
|
||||
};
|
||||
}
|
||||
} // namespace FEXCore::Utils
|
||||
Vendored
-1
Submodule External/Vulkan-Docs deleted from a0960966d5.
Vendored
+1
-1
Submodule External/fmt updated: 7bdf0628b1...b6f4ceaed0.
Vendored
+1
-1
Submodule External/jemalloc updated: dea850b30a...b700fc4030.
Vendored
+1
-1
Submodule External/vixl updated: 4d6c1d44a5...00ada80548.
Loaded 100 of 298 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user