mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-08 20:00:17 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e9e5b6fb0b | ||
|
|
68cb6e61d1 | ||
|
|
5d0b2060e2 | ||
|
|
b7d05a65c7 | ||
|
|
93fe2fe06c | ||
|
|
21eb6e03c7 | ||
|
|
1b53337925 | ||
|
|
52e5b8ccd9 | ||
|
|
8c8a8c84df | ||
|
|
0d6837f1a1 | ||
|
|
7ef3cb88f9 | ||
|
|
617977357a | ||
|
|
5a53c9231b | ||
|
|
a996e5300e | ||
|
|
25b2af14fd | ||
|
|
76059949ea | ||
|
|
6e15c9c213 | ||
|
|
01fcca884b | ||
|
|
91bd3aa62a | ||
|
|
7a0119b092 | ||
|
|
fa42c1616e | ||
|
|
18783948f7 | ||
|
|
969d2e4b6a | ||
|
|
7ceaf56407 | ||
|
|
7c52375267 | ||
|
|
a4ea792d03 | ||
|
|
cec98637c9 | ||
|
|
d5026f5815 | ||
|
|
1e4456ec40 | ||
|
|
e285c7c9a0 | ||
|
|
8244c7f2a6 | ||
|
|
747c5e17f8 | ||
|
|
52ce027e9b | ||
|
|
0fb3903889 | ||
|
|
2832afd0f7 | ||
|
|
4af42477a7 | ||
|
|
4786aa479c | ||
|
|
8a049aa0c3 | ||
|
|
64bac687b5 | ||
|
|
6e05494ad0 | ||
|
|
4d13f5d97d | ||
|
|
1ad928dc2c | ||
|
|
c235ab883d | ||
|
|
4bf3a0888b | ||
|
|
6834fe32e4 | ||
|
|
d196709162 | ||
|
|
475a6a38fa | ||
|
|
0f748bd724 | ||
|
|
47bd331239 | ||
|
|
173b70d191 | ||
|
|
96aa0a844e | ||
|
|
48121cd585 | ||
|
|
52d7efda10 | ||
|
|
30ab4d3f58 | ||
|
|
f9fa0f25e2 | ||
|
|
eebcbfda96 | ||
|
|
1483ddb538 | ||
|
|
d2bca9b997 | ||
|
|
8f48021a63 | ||
|
|
1da5d7f2e6 | ||
|
|
0a5db6c404 | ||
|
|
4681061011 | ||
|
|
4e9eeb1a16 | ||
|
|
95d728b634 | ||
|
|
187e551a0c | ||
|
|
7c4e4c4409 | ||
|
|
bdd8df2100 | ||
|
|
1d1bdfb96d | ||
|
|
2fc6542d15 | ||
|
|
5e88be3c99 | ||
|
|
be52844ce0 | ||
|
|
d629884147 | ||
|
|
64c72430bf | ||
|
|
1b13e8db7d | ||
|
|
1ce0ea8f3b | ||
|
|
8557656259 | ||
|
|
c87c04db59 | ||
|
|
f5262446a0 | ||
|
|
0a1820d444 | ||
|
|
a255813e99 | ||
|
|
4c6336b560 | ||
|
|
145c7799a5 | ||
|
|
91df503e55 | ||
|
|
3c417cee84 | ||
|
|
857780b46e | ||
|
|
cb34aa3dee | ||
|
|
0d74777954 | ||
|
|
092a023900 | ||
|
|
b57dd5c3aa | ||
|
|
0bea508935 | ||
|
|
9c175da0e2 | ||
|
|
cc359f09ad | ||
|
|
dedeba0575 | ||
|
|
dc3ccbcc43 | ||
|
|
ebd9b49dbe | ||
|
|
43e21ad749 | ||
|
|
1b3beec085 | ||
|
|
c88f2e0730 | ||
|
|
1645f6e582 | ||
|
|
af35e18979 | ||
|
|
012c0ae062 | ||
|
|
dc8f063a4b | ||
|
|
759747b3c5 | ||
|
|
4c6d26f167 | ||
|
|
7562ff2308 | ||
|
|
e2baa1106b | ||
|
|
68880fd66b | ||
|
|
62eef6c548 | ||
|
|
cfdc1bda47 | ||
|
|
f9a9645ef3 | ||
|
|
8fc903cc51 | ||
|
|
c81b1c3432 | ||
|
|
db9ba16a53 | ||
|
|
0be68a54d0 | ||
|
|
6e140e9ff2 | ||
|
|
d793f7295a | ||
|
|
8f6018a1b4 | ||
|
|
0a97d4456d | ||
|
|
c2dbe8309b | ||
|
|
06d66a0188 | ||
|
|
ef254da7ca | ||
|
|
162dcf3cd0 | ||
|
|
dcc47074d1 | ||
|
|
4dfee31012 | ||
|
|
3749fa6569 | ||
|
|
94273fbf4f | ||
|
|
162bbf2937 | ||
|
|
296adf1830 | ||
|
|
a86109280a | ||
|
|
d83960d4dd | ||
|
|
bba97823a8 | ||
|
|
abf5e8c6a5 | ||
|
|
8d288300c4 | ||
|
|
0caf263c77 | ||
|
|
29dc77c44b | ||
|
|
4c1f53c1ff | ||
|
|
5821175ddb | ||
|
|
003c88e537 | ||
|
|
27a1ebc2f5 | ||
|
|
e689c6fcfa | ||
|
|
61e905a339 | ||
|
|
10fdcaa109 | ||
|
|
ba672f868c | ||
|
|
d6697fce32 | ||
|
|
9c3a843df7 | ||
|
|
8defa2b55f | ||
|
|
77c88ffe53 | ||
|
|
5b261b0d2e | ||
|
|
2ce15ddc89 | ||
|
|
3d1b55383e | ||
|
|
69deaa0976 | ||
|
|
fc72fa9e5f | ||
|
|
6baee3b7c1 | ||
|
|
362a5a019a | ||
|
|
d529a58893 | ||
|
|
c77ea3b392 | ||
|
|
20aaad15e4 | ||
|
|
6f2452e2ea | ||
|
|
b34401bb33 | ||
|
|
dff9868e9a | ||
|
|
036a196984 | ||
|
|
a7ac4fa6e4 | ||
|
|
82295b2943 | ||
|
|
eca80b6046 | ||
|
|
02fc02964b | ||
|
|
e8fcb070b3 | ||
|
|
597da88035 | ||
|
|
1e829a47fa | ||
|
|
e633ef7cf5 | ||
|
|
ff3b40400c | ||
|
|
801106faf4 | ||
|
|
536b2ed495 | ||
|
|
072f027885 | ||
|
|
754bc18813 | ||
|
|
ef257418d3 | ||
|
|
842b71cf83 | ||
|
|
8d8b64d2b7 | ||
|
|
85c6ef8097 | ||
|
|
2be16d9054 | ||
|
|
41b3c52663 | ||
|
|
f1f50b7a98 | ||
|
|
d7200e2a1e | ||
|
|
3f884fe2d0 | ||
|
|
1d453f10a0 | ||
|
|
5ebd21ca6a | ||
|
|
fb34b507a1 | ||
|
|
2c91b5cff6 | ||
|
|
79f7dcbaa5 | ||
|
|
f7b7997c77 | ||
|
|
0674dfab0a | ||
|
|
6979dc9c4e | ||
|
|
fe5f17d92e | ||
|
|
be71886990 | ||
|
|
5043e5fbc0 | ||
|
|
2acfde3cad | ||
|
|
c1eeeaf688 | ||
|
|
f2b3229a87 | ||
|
|
e20bfc0701 | ||
|
|
4caee5c9be | ||
|
|
46d3f283b8 | ||
|
|
cee5512a56 | ||
|
|
8c53a373bf | ||
|
|
f73176d5dc | ||
|
|
80ae3e632d | ||
|
|
ed75c19324 | ||
|
|
4a7fa7f2bc | ||
|
|
98f51c47fa | ||
|
|
724a8e13bf | ||
|
|
d9b52fd67d | ||
|
|
d3a2795106 | ||
|
|
3cd6c2d91a | ||
|
|
b86abfbccf | ||
|
|
ee66985ae0 | ||
|
|
daeba0625f | ||
|
|
5e6af25194 | ||
|
|
7f4528a6b0 | ||
|
|
776b7674e4 | ||
|
|
24cb2610a2 | ||
|
|
54b7a43b95 | ||
|
|
1b1e9e0fb5 | ||
|
|
ead43c6a51 | ||
|
|
e49de77225 | ||
|
|
8f4fe39b7d | ||
|
|
f250509718 | ||
|
|
6179c5a13e | ||
|
|
9e14a83442 | ||
|
|
95bfd003d2 | ||
|
|
08ca43c3c4 | ||
|
|
96b428dcaa | ||
|
|
421214e723 | ||
|
|
1a18bbb966 | ||
|
|
699c3f5762 | ||
|
|
da0a1710c0 | ||
|
|
c1205eb809 | ||
|
|
31b7cd77e9 | ||
|
|
9def04c705 | ||
|
|
68a2441e65 | ||
|
|
4ac0dec568 | ||
|
|
1d7b4bb522 | ||
|
|
22f95e627d | ||
|
|
599b64e975 | ||
|
|
5c27febbc2 | ||
|
|
58c93568f6 | ||
|
|
491e4e2c23 | ||
|
|
1ed2e24fba | ||
|
|
1fbb8bd78f | ||
|
|
b953433404 | ||
|
|
eae950be16 | ||
|
|
7e6bb04db1 | ||
|
|
68555546bc | ||
|
|
716cac35a8 | ||
|
|
9722c4c5a4 | ||
|
|
e8c0e19afc | ||
|
|
c559fec959 | ||
|
|
8d2fabe705 | ||
|
|
6455c4817a | ||
|
|
5dbd1b8dc2 | ||
|
|
7765bbc7b8 | ||
|
|
2283c73fae | ||
|
|
5fef0c29aa | ||
|
|
d387c46aab | ||
|
|
3bb7f9d6b5 | ||
|
|
70d54122b2 | ||
|
|
4e266cf9fd | ||
|
|
ddd6dbfdcc | ||
|
|
7f2557e322 | ||
|
|
810c7d926c | ||
|
|
04c325661c | ||
|
|
0121e858f2 | ||
|
|
2c3361be9e | ||
|
|
2d800b2627 | ||
|
|
55ed3e0549 | ||
|
|
55d084ebb0 | ||
|
|
457dc5dd90 | ||
|
|
98eda5e163 | ||
|
|
592935790e | ||
|
|
92d0344d6a | ||
|
|
9327435f97 | ||
|
|
573f339647 | ||
|
|
c1f18951ab | ||
|
|
8a4bfba47c | ||
|
|
69ea03f0eb | ||
|
|
462feec2a6 | ||
|
|
15f5fe658b | ||
|
|
052aa4317b | ||
|
|
debcb0e047 | ||
|
|
baf04b6a41 | ||
|
|
cc85a6a722 | ||
|
|
e72fa02897 | ||
|
|
8a4c5bcc65 | ||
|
|
7a13a24c05 | ||
|
|
5a53931b92 | ||
|
|
f9b352a093 | ||
|
|
f444b03317 | ||
|
|
ed05846dd0 | ||
|
|
f609990f90 | ||
|
|
d2032da452 | ||
|
|
8047007a7a | ||
|
|
1f7e82ea09 | ||
|
|
35c52f20f9 | ||
|
|
17c82c22a6 | ||
|
|
df03a7b101 | ||
|
|
c859540d7e | ||
|
|
20794593e7 | ||
|
|
a80a2bf569 | ||
|
|
e86a792189 | ||
|
|
d6c9b549df | ||
|
|
1a4d5a1abb | ||
|
|
c3e123df25 | ||
|
|
cac798574a | ||
|
|
1506a19229 | ||
|
|
51861234bc | ||
|
|
677b72c9a5 | ||
|
|
71a8c66c95 | ||
|
|
3372e9bdbb | ||
|
|
df0723e14b | ||
|
|
7ee6fc0d7f | ||
|
|
bf773452ac | ||
|
|
95dbccc0ab | ||
|
|
e5189d63a2 | ||
|
|
7d5442357a | ||
|
|
9dcc1deec0 | ||
|
|
66d4206cd7 | ||
|
|
f39163b1e1 | ||
|
|
01837b3ad6 | ||
|
|
bdb68840e3 | ||
|
|
4e2dcf3298 | ||
|
|
628f825416 | ||
|
|
a80327f6df | ||
|
|
16f7002222 | ||
|
|
c9712e45cb | ||
|
|
a082161d72 | ||
|
|
9017325c95 | ||
|
|
ae536e44d7 | ||
|
|
7679485cc3 | ||
|
|
cb8bf1add6 | ||
|
|
a69c457715 | ||
|
|
7c4729678b | ||
|
|
537562fab7 | ||
|
|
f8721992c2 | ||
|
|
e652399fd3 | ||
|
|
e7c92c43a0 | ||
|
|
9b5e1c44c8 | ||
|
|
755600c371 | ||
|
|
bec8b70e5d | ||
|
|
bef8ddde48 | ||
|
|
fe06f1b151 | ||
|
|
92a15e00c7 | ||
|
|
2997257d6d | ||
|
|
7ceadc6b5b | ||
|
|
8c41e8f7d8 | ||
|
|
784b3064fc |
No files matched your search
@@ -172,6 +172,17 @@ jobs:
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_APITests.log || true
|
||||
|
||||
- name: FEXCore APITest tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target fexcore_apitests
|
||||
|
||||
- name: FEXCore APITest Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_FEXCoreAPITests.log || true
|
||||
|
||||
- name: ARMEmitter tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
@@ -254,6 +265,7 @@ jobs:
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
|
||||
@@ -157,6 +157,17 @@ jobs:
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_APITests.log || true
|
||||
|
||||
- name: FEXCore APITest tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target fexcore_apitests
|
||||
|
||||
- name: FEXCore APITest Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_FEXCoreAPITests.log || true
|
||||
|
||||
- name: FEXLinuxTests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
@@ -189,6 +200,7 @@ jobs:
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
|
||||
@@ -0,0 +1,96 @@
|
||||
name: Mingw build
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
pull_request:
|
||||
branches:
|
||||
- main
|
||||
|
||||
env:
|
||||
BUILD_TYPE: Debug
|
||||
FEX_ENABLEAVX: 1
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ${{ matrix.arch }}
|
||||
strategy:
|
||||
matrix:
|
||||
arch: [[self-hosted, x64, mingw], [self-hosted, ARM64, mingw]]
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
|
||||
- name: Set CC x86
|
||||
if: matrix.arch[1] == 'x64'
|
||||
run: |
|
||||
echo "CC=$HOME/llvm-mingw/build/bin/x86_64-w64-mingw32-clang" >> $GITHUB_ENV
|
||||
echo "CXX=$HOME/llvm-mingw/build/bin/x86_64-w64-mingw32-clang++" >> $GITHUB_ENV
|
||||
|
||||
- name: Set CC Arm64
|
||||
if: matrix.arch[1] == 'ARM64'
|
||||
run: |
|
||||
echo "CC=$HOME/llvm-mingw/build/bin/aarch64-w64-mingw32-clang" >> $GITHUB_ENV
|
||||
echo "CXX=$HOME/llvm-mingw/build/bin/aarch64-w64-mingw32-clang++" >> $GITHUB_ENV
|
||||
|
||||
- name: Set rootfs paths
|
||||
run: |
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
|
||||
|
||||
- name : submodule checkout
|
||||
# Need to update submodules
|
||||
run: |
|
||||
git submodule sync --recursive
|
||||
git submodule update --init --depth 1
|
||||
|
||||
- name: Clean Build Environment
|
||||
run: rm -Rf ${{runner.workspace}}/build
|
||||
|
||||
- name: Create Build Environment
|
||||
# Some projects don't allow in-source building, so create a separate build directory
|
||||
# We'll use this as our working directory for all subsequent commands
|
||||
run: cmake -E make_directory ${{runner.workspace}}/build
|
||||
|
||||
- name: Configure CMake
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DENABLE_INTERPRETER=False -DBUILD_TESTS=False -DENABLE_JEMALLOC=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the build. You can specify a specific target with "--target <NAME>"
|
||||
run: cmake --build . --config $BUILD_TYPE
|
||||
|
||||
- name: Set runner name
|
||||
if: ${{ always() }}
|
||||
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
retention-days: 3
|
||||
|
||||
@@ -112,6 +112,7 @@ jobs:
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
|
||||
@@ -0,0 +1,30 @@
|
||||
# Extracts a version from the passed in version string in the form of "<Major>.<Minor>.<Patch>".
|
||||
# If a part of the version is missing then it gets set as zero.
|
||||
# Version variables returned in:
|
||||
# ${Package}_VERSION_MAJOR
|
||||
# ${Package}_VERSION_MINOR
|
||||
# ${Package}_VERSION_PATCH
|
||||
function(version_to_variables VERSION _Package)
|
||||
string(REPLACE "." ";" VERSION_LIST "${VERSION}")
|
||||
list (LENGTH VERSION_LIST VERSION_LEN)
|
||||
if (${VERSION_LEN} GREATER 0)
|
||||
list(GET VERSION_LIST 0 VERSION_MAJOR)
|
||||
set(${_Package}_VERSION_MAJOR ${VERSION_MAJOR} PARENT_SCOPE)
|
||||
else()
|
||||
set(${_Package}_VERSION_MAJOR 0 PARENT_SCOPE)
|
||||
endif()
|
||||
|
||||
if (${VERSION_LEN} GREATER 1)
|
||||
list(GET VERSION_LIST 1 VERSION_MINOR)
|
||||
set(${_Package}_VERSION_MINOR ${VERSION_MINOR} PARENT_SCOPE)
|
||||
else()
|
||||
set(${_Package}_VERSION_MINOR 0 PARENT_SCOPE)
|
||||
endif()
|
||||
|
||||
if (${VERSION_LEN} GREATER 2)
|
||||
list(GET VERSION_LIST 2 VERSION_PATCH)
|
||||
set(${_Package}_VERSION_PATCH ${VERSION_PATCH} PARENT_SCOPE)
|
||||
else()
|
||||
set(${_Package}_VERSION_PATCH 0 PARENT_SCOPE)
|
||||
endif()
|
||||
endfunction()
|
||||
+12
-9
@@ -7,13 +7,13 @@ CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
|
||||
option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
|
||||
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
|
||||
option(BUILD_THUNKS "Build thunks" FALSE)
|
||||
option(BUILD_FEXCONFIG "Build FEXConfig, requires SDL2 and X11" TRUE)
|
||||
option(ENABLE_CLANG_THUNKS "Build thunks with clang" FALSE)
|
||||
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
|
||||
option(ENABLE_IWYU "Enables include what you use program" FALSE)
|
||||
option(ENABLE_LTO "Enable LTO with compilation" TRUE)
|
||||
option(ENABLE_XRAY "Enable building with LLVM X-Ray" FALSE)
|
||||
option(ENABLE_LLD "Enable linking with lld" FALSE)
|
||||
option(ENABLE_MOLD "Enable linking with mold" FALSE)
|
||||
set(USE_LINKER "" CACHE STRING "Allow overriding the linker path directly")
|
||||
option(ENABLE_ASAN "Enables Clang ASAN" FALSE)
|
||||
option(ENABLE_TSAN "Enables Clang TSAN" FALSE)
|
||||
option(ENABLE_ASSERTIONS "Enables assertions in build" FALSE)
|
||||
@@ -156,13 +156,9 @@ endif()
|
||||
|
||||
set (PTHREAD_LIB pthread)
|
||||
|
||||
if (ENABLE_LLD AND ENABLE_MOLD)
|
||||
message (FATAL_ERROR "Cannot enable both lld and mold")
|
||||
elseif (ENABLE_LLD)
|
||||
set (LD_OVERRIDE "-fuse-ld=lld")
|
||||
add_link_options(${LD_OVERRIDE})
|
||||
elseif (ENABLE_MOLD)
|
||||
add_link_options("-fuse-ld=mold")
|
||||
if (USE_LINKER)
|
||||
message(STATUS "Overriding linker to: ${USE_LINKER}")
|
||||
add_link_options("-fuse-ld=${USE_LINKER}")
|
||||
endif()
|
||||
|
||||
if (ENABLE_LIBCXX)
|
||||
@@ -179,8 +175,13 @@ endif()
|
||||
if(DEFINED ENV{TERMUX_VERSION} OR ENABLE_TERMUX_BUILD)
|
||||
add_definitions(-DTERMUX_BUILD=1)
|
||||
set(TERMUX_BUILD 1)
|
||||
|
||||
# Termux doesn't support Jemalloc due to bad interactions between emutls, jemalloc, and scudo
|
||||
set(ENABLE_JEMALLOC FALSE)
|
||||
|
||||
# Termux builds can't rely on X11 packages
|
||||
# SDL2 isn't even compiled with GL support so our GUIs wouldn't even work
|
||||
set(BUILD_FEXCONFIG FALSE)
|
||||
endif()
|
||||
|
||||
if (ENABLE_ASAN)
|
||||
@@ -261,6 +262,8 @@ if (BUILD_TESTS)
|
||||
include(Catch)
|
||||
endif()
|
||||
|
||||
# Disable fmt install
|
||||
set(FMT_INSTALL OFF)
|
||||
add_subdirectory(External/fmt/)
|
||||
|
||||
add_subdirectory(External/imgui/)
|
||||
|
||||
+66
-63
@@ -6,10 +6,10 @@
|
||||
"X11"
|
||||
],
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGL.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGL.so.1",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGL.so.1.2.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGL.so.1.7.0"
|
||||
"@PREFIX_LIB@/libGL.so",
|
||||
"@PREFIX_LIB@/libGL.so.1",
|
||||
"@PREFIX_LIB@/libGL.so.1.2.0",
|
||||
"@PREFIX_LIB@/libGL.so.1.7.0"
|
||||
]
|
||||
},
|
||||
"GLESv2": {
|
||||
@@ -18,17 +18,17 @@
|
||||
"X11"
|
||||
],
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGLESv2.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGLESv2.so.2",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGLESv2.so.2.0.0"
|
||||
"@PREFIX_LIB@/libGLESv2.so",
|
||||
"@PREFIX_LIB@/libGLESv2.so.2",
|
||||
"@PREFIX_LIB@/libGLESv2.so.2.0.0"
|
||||
]
|
||||
},
|
||||
"X11": {
|
||||
"Library": "libX11-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libX11.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libX11.so.6",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libX11.so.6.4.0"
|
||||
"@PREFIX_LIB@/libX11.so",
|
||||
"@PREFIX_LIB@/libX11.so.6",
|
||||
"@PREFIX_LIB@/libX11.so.6.4.0"
|
||||
]
|
||||
},
|
||||
"Vulkan": {
|
||||
@@ -37,8 +37,8 @@
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libvulkan.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libvulkan.so.1",
|
||||
"@PREFIX_LIB@/libvulkan.so",
|
||||
"@PREFIX_LIB@/libvulkan.so.1",
|
||||
"@HOME@/.local/share/Steam/ubuntu12_32/steam-runtime/pinned_libs_64/libvulkan.so.1"
|
||||
],
|
||||
"Comment": [
|
||||
@@ -46,139 +46,142 @@
|
||||
]
|
||||
},
|
||||
"xcb": {
|
||||
"Depends": [
|
||||
"X11"
|
||||
],
|
||||
"Library": "libxcb-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb.so.1",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb.so.1.1.0"
|
||||
"@PREFIX_LIB@/libxcb.so",
|
||||
"@PREFIX_LIB@/libxcb.so.1",
|
||||
"@PREFIX_LIB@/libxcb.so.1.1.0"
|
||||
]
|
||||
},
|
||||
"xcb-dri2": {
|
||||
"Library": "libxcb-dri2-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri2.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri2.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri2.so.0.0.0"
|
||||
"@PREFIX_LIB@/libxcb-dri2.so",
|
||||
"@PREFIX_LIB@/libxcb-dri2.so.0",
|
||||
"@PREFIX_LIB@/libxcb-dri2.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-dri3": {
|
||||
"Library": "libxcb-dri3-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri3.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri3.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri3.so.0.0.0"
|
||||
"@PREFIX_LIB@/libxcb-dri3.so",
|
||||
"@PREFIX_LIB@/libxcb-dri3.so.0",
|
||||
"@PREFIX_LIB@/libxcb-dri3.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-xfixes": {
|
||||
"Library": "libxcb-xfixes-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-xfixes.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-xfixes.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-xfixes.so.0.0.0"
|
||||
"@PREFIX_LIB@/libxcb-xfixes.so",
|
||||
"@PREFIX_LIB@/libxcb-xfixes.so.0",
|
||||
"@PREFIX_LIB@/libxcb-xfixes.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-shm": {
|
||||
"Library": "libxcb-shm-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-shm.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-shm.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-shm.so.0.0.0"
|
||||
"@PREFIX_LIB@/libxcb-shm.so",
|
||||
"@PREFIX_LIB@/libxcb-shm.so.0",
|
||||
"@PREFIX_LIB@/libxcb-shm.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-sync": {
|
||||
"Library": "libxcb-sync-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-sync.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-sync.so.1",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-sync.so.1.0.0"
|
||||
"@PREFIX_LIB@/libxcb-sync.so",
|
||||
"@PREFIX_LIB@/libxcb-sync.so.1",
|
||||
"@PREFIX_LIB@/libxcb-sync.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-randr": {
|
||||
"Library": "libxcb-randr-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-randr.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-randr.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-randr.so.0.1.0"
|
||||
"@PREFIX_LIB@/libxcb-randr.so",
|
||||
"@PREFIX_LIB@/libxcb-randr.so.0",
|
||||
"@PREFIX_LIB@/libxcb-randr.so.0.1.0"
|
||||
]
|
||||
},
|
||||
"xcb-present": {
|
||||
"Library": "libxcb-present-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-present.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-present.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-present.so.0.0.0"
|
||||
"@PREFIX_LIB@/libxcb-present.so",
|
||||
"@PREFIX_LIB@/libxcb-present.so.0",
|
||||
"@PREFIX_LIB@/libxcb-present.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-glx": {
|
||||
"Library": "libxcb-glx-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-glx.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-glx.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-glx.so.0.0.0"
|
||||
"@PREFIX_LIB@/libxcb-glx.so",
|
||||
"@PREFIX_LIB@/libxcb-glx.so.0",
|
||||
"@PREFIX_LIB@/libxcb-glx.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xshmfence": {
|
||||
"Library": "libxshmfence-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxshmfence.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxshmfence.so.1",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxshmfence.so.1.0.0"
|
||||
"@PREFIX_LIB@/libxshmfence.so",
|
||||
"@PREFIX_LIB@/libxshmfence.so.1",
|
||||
"@PREFIX_LIB@/libxshmfence.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"drm": {
|
||||
"Library": "libdrm-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libdrm.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libdrm.so.2",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libdrm.so.2.4.0"
|
||||
"@PREFIX_LIB@/libdrm.so",
|
||||
"@PREFIX_LIB@/libdrm.so.2",
|
||||
"@PREFIX_LIB@/libdrm.so.2.4.0"
|
||||
]
|
||||
},
|
||||
"asound": {
|
||||
"Library": "libasound-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libasound.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libasound.so.2",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libasound.so.2.0.0"
|
||||
"@PREFIX_LIB@/libasound.so",
|
||||
"@PREFIX_LIB@/libasound.so.2",
|
||||
"@PREFIX_LIB@/libasound.so.2.0.0"
|
||||
]
|
||||
},
|
||||
"Xrender": {
|
||||
"Library": "libXrender-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXrender.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXrender.so.1",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXrender.so.1.3.0"
|
||||
"@PREFIX_LIB@/libXrender.so",
|
||||
"@PREFIX_LIB@/libXrender.so.1",
|
||||
"@PREFIX_LIB@/libXrender.so.1.3.0"
|
||||
]
|
||||
},
|
||||
"Xext": {
|
||||
"Library": "libXext-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXext.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXext.so.6",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXext.so.6.4.0"
|
||||
"@PREFIX_LIB@/libXext.so",
|
||||
"@PREFIX_LIB@/libXext.so.6",
|
||||
"@PREFIX_LIB@/libXext.so.6.4.0"
|
||||
]
|
||||
},
|
||||
"Xfixes": {
|
||||
"Library": "libXfixes-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXfixes.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXfixes.so.3",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXfixes.so.3.1.0"
|
||||
"@PREFIX_LIB@/libXfixes.so",
|
||||
"@PREFIX_LIB@/libXfixes.so.3",
|
||||
"@PREFIX_LIB@/libXfixes.so.3.1.0"
|
||||
]
|
||||
},
|
||||
"OpenCL": {
|
||||
"Library" : "libOpenCL-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libOpenCL.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libOpenCL.so.1",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libOpenCL.so.1.0.0"
|
||||
"@PREFIX_LIB@/libOpenCL.so",
|
||||
"@PREFIX_LIB@/libOpenCL.so.1",
|
||||
"@PREFIX_LIB@/libOpenCL.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"WaylandClient": {
|
||||
"Library" : "libwayland-client-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libwayland-client.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libwayland-client.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libwayland-client.so.0.20.0"
|
||||
"@PREFIX_LIB@/libwayland-client.so",
|
||||
"@PREFIX_LIB@/libwayland-client.so.0",
|
||||
"@PREFIX_LIB@/libwayland-client.so.0.20.0"
|
||||
]
|
||||
},
|
||||
"":{}
|
||||
|
||||
+70
-5
@@ -27,7 +27,9 @@ def print_header():
|
||||
#ifndef OPT_STRARRAY
|
||||
#define OPT_STRARRAY(group, enum, json, default) OPT_BASE(fextl::string, group, enum, json, default)
|
||||
#endif
|
||||
|
||||
#ifndef OPT_STRENUM
|
||||
#define OPT_STRENUM(group, enum, json, default) OPT_BASE(uint64_t, group, enum, json, default)
|
||||
#endif
|
||||
'''
|
||||
output_file.write(header)
|
||||
|
||||
@@ -40,6 +42,7 @@ def print_tail():
|
||||
#undef OPT_UINT64
|
||||
#undef OPT_STR
|
||||
#undef OPT_STRARRAY
|
||||
#undef OPT_STRENUM
|
||||
'''
|
||||
output_file.write(tail)
|
||||
|
||||
@@ -127,12 +130,13 @@ def print_man_options(options):
|
||||
short = op_vals["ShortArg"]
|
||||
|
||||
default = op_vals["Default"]
|
||||
value_type = op_vals["Type"]
|
||||
|
||||
# Textual default rather than enum based
|
||||
if ("TextDefault" in op_vals):
|
||||
default = op_vals["TextDefault"]
|
||||
|
||||
if (op_vals["Type"] == "str" or op_vals["Type"] == "strarray"):
|
||||
if (value_type == "str" or value_type == "strarray" or value_type == "strenum"):
|
||||
# Wrap the string argument in quotes
|
||||
default = "'" + default + "'"
|
||||
print_man_option(
|
||||
@@ -141,6 +145,12 @@ def print_man_options(options):
|
||||
op_vals["Desc"],
|
||||
default
|
||||
)
|
||||
if (value_type == "strenum"):
|
||||
Enums = op_vals["Enums"]
|
||||
output_man.write("\\fBAvailable Options:\\fR\n")
|
||||
for enum_op_key, enum_op_vals in Enums.items():
|
||||
output_man.write("{}, ".format(enum_op_vals))
|
||||
output_man.write("\n")
|
||||
|
||||
output_man.write(".El\n")
|
||||
|
||||
@@ -150,12 +160,13 @@ def print_man_environment(options):
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
default = op_vals["Default"]
|
||||
value_type = op_vals["Type"]
|
||||
|
||||
# Textual default rather than enum based
|
||||
if ("TextDefault" in op_vals):
|
||||
default = op_vals["TextDefault"]
|
||||
|
||||
if (op_vals["Type"] == "str" or op_vals["Type"] == "strarray"):
|
||||
if (value_type == "str" or value_type == "strarray" or value_type == "strenum"):
|
||||
# Wrap the string argument in quotes
|
||||
default = "'" + default + "'"
|
||||
print_man_env_option(
|
||||
@@ -165,6 +176,13 @@ def print_man_environment(options):
|
||||
False
|
||||
)
|
||||
|
||||
if (value_type == "strenum"):
|
||||
Enums = op_vals["Enums"]
|
||||
output_man.write("\\fBAvailable Options:\\fR\n")
|
||||
for enum_op_key, enum_op_vals in Enums.items():
|
||||
output_man.write("{}, ".format(enum_op_vals))
|
||||
output_man.write("\n")
|
||||
|
||||
print_man_environment_tail()
|
||||
output_man.write(".El\n")
|
||||
|
||||
@@ -334,7 +352,7 @@ def print_argloader_options(options):
|
||||
for op_key, op_vals in group_vals.items():
|
||||
default = op_vals["Default"]
|
||||
|
||||
if (op_vals["Type"] == "str" or op_vals["Type"] == "strarray"):
|
||||
if (op_vals["Type"] == "str" or op_vals["Type"] == "strarray" or op_vals["Type"] == "strenum"):
|
||||
# Wrap the string argument in quotes
|
||||
default = "\"" + default + "\""
|
||||
|
||||
@@ -382,7 +400,10 @@ def print_parse_argloader_options(options):
|
||||
# boolean values need a decimal specifier. Otherwise fmt prints strings.
|
||||
conversion_func = "fextl::fmt::format(\"{:d}\", "
|
||||
|
||||
if (value_type == "strarray"):
|
||||
if (value_type == "strenum"):
|
||||
output_argloader.write("\tfextl::string UserValue = Options[\"{0}\"];\n".format(op_key))
|
||||
output_argloader.write("\tSet(FEXCore::Config::ConfigOption::CONFIG_{}, FEXCore::Config::EnumParser(FEXCore::Config::{}_EnumPairs, UserValue));\n".format(op_key.upper(), op_key, op_key))
|
||||
elif (value_type == "strarray"):
|
||||
# these need a bit more help
|
||||
output_argloader.write("\tauto Array = Options.all(\"{0}\");\n".format(op_key))
|
||||
output_argloader.write("\tfor (auto iter = Array.begin(); iter != Array.end(); ++iter) {\n")
|
||||
@@ -407,6 +428,12 @@ def print_parse_envloader_options(options):
|
||||
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
value_type = op_vals["Type"]
|
||||
if (value_type == "strenum"):
|
||||
output_argloader.write("else if (Key == \"FEX_{0}\") {{\n".format(op_key.upper()))
|
||||
output_argloader.write("Value = FEXCore::Config::EnumParser(FEXCore::Config::{}_EnumPairs, Value);\n".format(op_key, op_key))
|
||||
output_argloader.write("}\n")
|
||||
|
||||
if ("ArgumentHandler" in op_vals):
|
||||
conversion_func = "FEXCore::Config::Handler::{0}".format(op_vals["ArgumentHandler"])
|
||||
output_argloader.write("else if (Key == \"FEX_{0}\") {{\n".format(op_key.upper()))
|
||||
@@ -414,6 +441,41 @@ def print_parse_envloader_options(options):
|
||||
output_argloader.write("}\n")
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
def print_parse_enum_options(options):
|
||||
output_argloader.write("#ifdef ENUMDEFINES\n")
|
||||
output_argloader.write("#undef ENUMDEFINES\n")
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
if (op_vals["Type"] == "strenum"):
|
||||
output_argloader.write("enum {} : uint64_t {{\n".format(op_key))
|
||||
Enums = op_vals["Enums"]
|
||||
i = 0
|
||||
# Always have an OFF.
|
||||
output_argloader.write("\tOFF = 0,\n")
|
||||
for enum_op_key, enum_op_vals in Enums.items():
|
||||
output_argloader.write("\t{} = 1ULL << {},\n".format(enum_op_key.upper(), i))
|
||||
i += 1
|
||||
|
||||
output_argloader.write("};\n")
|
||||
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
if (op_vals["Type"] == "strenum"):
|
||||
Enums = op_vals["Enums"]
|
||||
|
||||
output_argloader.write("using {}ConfigPair = std::pair<std::string_view, FEXCore::Config::{}>;\n".format(op_key, op_key))
|
||||
output_argloader.write("constexpr static std::array<{}ConfigPair, {}> {}_EnumPairs = {{{{\n".format(op_key, len(Enums) + 1, op_key))
|
||||
i = 0
|
||||
# Always have an OFF.
|
||||
output_argloader.write("\t{{ \"off\", FEXCore::Config::{}::OFF }},\n".format(op_key))
|
||||
for enum_op_key, enum_op_vals in Enums.items():
|
||||
output_argloader.write("\t{{ \"{}\", FEXCore::Config::{}::{} }},\n".format(enum_op_vals, op_key, enum_op_key.upper()))
|
||||
i += 1
|
||||
|
||||
output_argloader.write("}};\n")
|
||||
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
def check_for_duplicate_options(options):
|
||||
short_map = []
|
||||
long_map = []
|
||||
@@ -492,4 +554,7 @@ print_parse_argloader_options(options);
|
||||
# Generate environment loader code
|
||||
print_parse_envloader_options(options);
|
||||
|
||||
# Generate enum variable options
|
||||
print_parse_enum_options(options);
|
||||
|
||||
output_argloader.close()
|
||||
+3
-5
@@ -251,11 +251,8 @@ def parse_ops(ops):
|
||||
|
||||
# Print out enum values
|
||||
def print_enums():
|
||||
if len(IROps) > 255:
|
||||
ExitError("We have more than uint8_t ops. We have {}. Time to upgrade to uint16_t".format(len(IROps)))
|
||||
|
||||
output_file.write("#ifdef IROP_ENUM\n")
|
||||
output_file.write("enum IROps : uint8_t {\n")
|
||||
output_file.write("enum IROps : uint16_t {\n")
|
||||
|
||||
for op in IROps:
|
||||
output_file.write("\tOP_{},\n" .format(op.Name.upper()))
|
||||
@@ -291,6 +288,7 @@ def print_ir_structs(defines):
|
||||
output_file.write("\tOrderedNodeWrapper Args[0];\n")
|
||||
|
||||
output_file.write("};\n\n");
|
||||
output_file.write("static_assert(sizeof(IROp_Header) == sizeof(uint32_t), \"IROp_Header should be 32-bits in size\");\n\n");
|
||||
|
||||
# Now the user defined types
|
||||
output_file.write("// User defined IR Op structs\n")
|
||||
@@ -677,7 +675,7 @@ def print_ir_allocator_helpers():
|
||||
output_file.write("\t\t#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED\n")
|
||||
|
||||
for Validation in op.EmitValidation:
|
||||
output_file.write("\t\tassert({});\n".format(Validation))
|
||||
output_file.write("\tLOGMAN_THROW_A_FMT({}, \"\");\n".format(Validation))
|
||||
output_file.write("\t\t#endif\n")
|
||||
|
||||
output_file.write("\t\treturn Op;\n")
|
||||
|
||||
+2
-2
@@ -1,4 +1,4 @@
|
||||
set (MAN_DIR ${CMAKE_INSTALL_PREFIX}/share/man CACHE PATH "MAN_DIR")
|
||||
set (MAN_DIR share/man CACHE PATH "MAN_DIR")
|
||||
|
||||
set (FEXCORE_BASE_SRCS
|
||||
Interface/Config/Config.cpp
|
||||
@@ -231,7 +231,7 @@ if (ENABLE_JIT_ARM64)
|
||||
)
|
||||
endif()
|
||||
|
||||
set (LIBS fmt::fmt vixl xxhash tiny-json FEXHeaderUtils)
|
||||
set (LIBS fmt::fmt vixl xxhash FEXHeaderUtils)
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
list (APPEND LIBS dl)
|
||||
|
||||
+5
-1
@@ -17,8 +17,12 @@ namespace FEXCore {
|
||||
|
||||
void JITSymbols::InitFile() {
|
||||
// We can't use FILE here since we must be robust against forking processes closing our FD from under us.
|
||||
#ifdef __ANDROID__
|
||||
// Android simpleperf looks in /data/local/tmp instead of /tmp
|
||||
const auto PerfMap = fextl::fmt::format("/data/local/tmp/perf-{}.map", getpid());
|
||||
#else
|
||||
const auto PerfMap = fextl::fmt::format("/tmp/perf-{}.map", getpid());
|
||||
|
||||
#endif
|
||||
fd = open(PerfMap.c_str(), O_CREAT | O_TRUNC | O_WRONLY | O_APPEND, 0644);
|
||||
}
|
||||
|
||||
|
||||
+3
-218
@@ -1,5 +1,6 @@
|
||||
#include "Common/StringConv.h"
|
||||
#include "Common/StringUtils.h"
|
||||
#include "FEXCore/Utils/EnumUtils.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
@@ -16,7 +17,6 @@
|
||||
#include <FEXHeaderUtils/Filesystem.h>
|
||||
|
||||
#include <array>
|
||||
#include <assert.h>
|
||||
#include <cstdlib>
|
||||
#include <functional>
|
||||
#include <optional>
|
||||
@@ -27,8 +27,6 @@
|
||||
#include <type_traits>
|
||||
#include <utility>
|
||||
|
||||
#include <tiny-json.h>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class Context;
|
||||
}
|
||||
@@ -39,74 +37,10 @@ namespace DefaultValues {
|
||||
#define OPT_BASE(type, group, enum, json, default) const P(type) P(enum) = P(default);
|
||||
#define OPT_STR(group, enum, json, default) const std::string_view P(enum) = P(default);
|
||||
#define OPT_STRARRAY(group, enum, json, default) OPT_STR(group, enum, json, default)
|
||||
#define OPT_STRENUM(group, enum, json, default) const uint64_t P(enum) = FEXCore::ToUnderlying(P(default));
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}
|
||||
|
||||
namespace JSON {
|
||||
struct JsonAllocator {
|
||||
jsonPool_t PoolObject;
|
||||
fextl::unique_ptr<fextl::list<json_t>> json_objects;
|
||||
};
|
||||
static_assert(offsetof(JsonAllocator, PoolObject) == 0, "This needs to be at offset zero");
|
||||
|
||||
json_t* PoolInit(jsonPool_t* Pool) {
|
||||
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
|
||||
alloc->json_objects = fextl::make_unique<fextl::list<json_t>>();
|
||||
return &*alloc->json_objects->emplace(alloc->json_objects->end());
|
||||
}
|
||||
|
||||
json_t* PoolAlloc(jsonPool_t* Pool) {
|
||||
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
|
||||
return &*alloc->json_objects->emplace(alloc->json_objects->end());
|
||||
}
|
||||
|
||||
static void LoadJSonConfig(const fextl::string &Config, std::function<void(const char *Name, const char *ConfigSring)> Func) {
|
||||
fextl::vector<char> Data;
|
||||
if (!FEXCore::FileLoading::LoadFile(Data, Config)) {
|
||||
return;
|
||||
}
|
||||
|
||||
JsonAllocator Pool {
|
||||
.PoolObject = {
|
||||
.init = PoolInit,
|
||||
.alloc = PoolAlloc,
|
||||
},
|
||||
};
|
||||
|
||||
json_t const *json = json_createWithPool(&Data.at(0), &Pool.PoolObject);
|
||||
if (!json) {
|
||||
LogMan::Msg::EFmt("Couldn't create json");
|
||||
return;
|
||||
}
|
||||
|
||||
json_t const* ConfigList = json_getProperty(json, "Config");
|
||||
|
||||
if (!ConfigList) {
|
||||
// This is a non-error if the configuration file exists but no Config section
|
||||
return;
|
||||
}
|
||||
|
||||
for (json_t const* ConfigItem = json_getChild(ConfigList);
|
||||
ConfigItem != nullptr;
|
||||
ConfigItem = json_getSibling(ConfigItem)) {
|
||||
const char* ConfigName = json_getName(ConfigItem);
|
||||
const char* ConfigString = json_getValue(ConfigItem);
|
||||
|
||||
if (!ConfigName) {
|
||||
LogMan::Msg::EFmt("Couldn't get config name");
|
||||
return;
|
||||
}
|
||||
|
||||
if (!ConfigString) {
|
||||
LogMan::Msg::EFmt("Couldn't get ConfigString for '{}'", ConfigName);
|
||||
return;
|
||||
}
|
||||
|
||||
Func(ConfigName, ConfigString);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
enum Paths {
|
||||
PATH_DATA_DIR = 0,
|
||||
PATH_CONFIG_DIR_LOCAL,
|
||||
@@ -518,7 +452,7 @@ namespace JSON {
|
||||
auto Value = FEXCore::Config::Get(Option);
|
||||
|
||||
if (!FEXCore::StrConv::Conv(**Value, &Result)) {
|
||||
assert(0 && "Attempted to convert invalid value");
|
||||
LOGMAN_MSG_A_FMT("Attempted to convert invalid value");
|
||||
}
|
||||
return Result;
|
||||
}
|
||||
@@ -583,154 +517,5 @@ namespace JSON {
|
||||
}
|
||||
}
|
||||
template void Value<fextl::string>::GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string> *List);
|
||||
|
||||
// Application loaders
|
||||
class MainLoader final : public FEXCore::Config::OptionMapper {
|
||||
public:
|
||||
explicit MainLoader(FEXCore::Config::LayerType Type);
|
||||
explicit MainLoader(fextl::string ConfigFile);
|
||||
void Load() override;
|
||||
|
||||
private:
|
||||
fextl::string Config;
|
||||
};
|
||||
|
||||
class AppLoader final : public FEXCore::Config::OptionMapper {
|
||||
public:
|
||||
explicit AppLoader(const fextl::string& Filename, FEXCore::Config::LayerType Type);
|
||||
void Load();
|
||||
|
||||
private:
|
||||
fextl::string Config;
|
||||
};
|
||||
|
||||
class EnvLoader final : public FEXCore::Config::Layer {
|
||||
public:
|
||||
explicit EnvLoader(char *const _envp[]);
|
||||
void Load() override;
|
||||
|
||||
private:
|
||||
char *const *envp;
|
||||
};
|
||||
|
||||
static const fextl::map<fextl::string, FEXCore::Config::ConfigOption, std::less<>> ConfigLookup = {{
|
||||
#define OPT_BASE(type, group, enum, json, default) {#json, FEXCore::Config::ConfigOption::CONFIG_##enum},
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}};
|
||||
static const fextl::vector<std::pair<const char*, FEXCore::Config::ConfigOption>> EnvConfigLookup = {{
|
||||
#define OPT_BASE(type, group, enum, json, default) {"FEX_" #enum, FEXCore::Config::ConfigOption::CONFIG_##enum},
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}};
|
||||
|
||||
OptionMapper::OptionMapper(FEXCore::Config::LayerType Layer)
|
||||
: FEXCore::Config::Layer(Layer) {
|
||||
}
|
||||
|
||||
void OptionMapper::MapNameToOption(const char *ConfigName, const char *ConfigString) {
|
||||
auto it = ConfigLookup.find(ConfigName);
|
||||
if (it != ConfigLookup.end()) {
|
||||
Set(it->second, ConfigString);
|
||||
}
|
||||
}
|
||||
|
||||
MainLoader::MainLoader(FEXCore::Config::LayerType Type)
|
||||
: FEXCore::Config::OptionMapper(Type)
|
||||
, Config{FEXCore::Config::GetConfigFileLocation(Type == FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN)} {
|
||||
}
|
||||
|
||||
MainLoader::MainLoader(fextl::string ConfigFile)
|
||||
: FEXCore::Config::OptionMapper(FEXCore::Config::LayerType::LAYER_MAIN)
|
||||
, Config{std::move(ConfigFile)} {
|
||||
}
|
||||
|
||||
void MainLoader::Load() {
|
||||
JSON::LoadJSonConfig(Config, [this](const char *Name, const char *ConfigString) {
|
||||
MapNameToOption(Name, ConfigString);
|
||||
});
|
||||
}
|
||||
|
||||
AppLoader::AppLoader(const fextl::string& Filename, FEXCore::Config::LayerType Type)
|
||||
: FEXCore::Config::OptionMapper(Type) {
|
||||
const bool Global = Type == FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP ||
|
||||
Type == FEXCore::Config::LayerType::LAYER_GLOBAL_APP;
|
||||
Config = FEXCore::Config::GetApplicationConfig(Filename, Global);
|
||||
|
||||
// Immediately load so we can reload the meta layer
|
||||
Load();
|
||||
}
|
||||
|
||||
void AppLoader::Load() {
|
||||
JSON::LoadJSonConfig(Config, [this](const char *Name, const char *ConfigString) {
|
||||
MapNameToOption(Name, ConfigString);
|
||||
});
|
||||
}
|
||||
|
||||
EnvLoader::EnvLoader(char *const _envp[])
|
||||
: FEXCore::Config::Layer(FEXCore::Config::LayerType::LAYER_ENVIRONMENT)
|
||||
, envp {_envp} {
|
||||
}
|
||||
|
||||
void EnvLoader::Load() {
|
||||
using EnvMapType = fextl::unordered_map<std::string_view, std::string_view>;
|
||||
EnvMapType EnvMap;
|
||||
|
||||
for(const char *const *pvar=envp; pvar && *pvar; pvar++) {
|
||||
std::string_view Var(*pvar);
|
||||
size_t pos = Var.rfind('=');
|
||||
if (fextl::string::npos == pos)
|
||||
continue;
|
||||
|
||||
std::string_view Key = Var.substr(0,pos);
|
||||
std::string_view Value {Var.substr(pos+1)};
|
||||
|
||||
#define ENVLOADER
|
||||
#include <FEXCore/Config/ConfigOptions.inl>
|
||||
|
||||
EnvMap[Key] = Value;
|
||||
}
|
||||
|
||||
auto GetVar = [](EnvMapType &EnvMap, const std::string_view id) -> std::optional<std::string_view> {
|
||||
if (EnvMap.find(id) != EnvMap.end())
|
||||
return EnvMap.at(id);
|
||||
|
||||
// If envp[] was empty, search using std::getenv()
|
||||
const char* vs = std::getenv(id.data());
|
||||
if (vs) {
|
||||
return vs;
|
||||
}
|
||||
else {
|
||||
return std::nullopt;
|
||||
}
|
||||
};
|
||||
|
||||
std::optional<std::string_view> Value;
|
||||
|
||||
for (auto &it : EnvConfigLookup) {
|
||||
if ((Value = GetVar(EnvMap, it.first)).has_value()) {
|
||||
Set(it.second, fextl::string(*Value));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateGlobalMainLayer() {
|
||||
return fextl::make_unique<FEXCore::Config::MainLoader>(FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN);
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(fextl::string const *File) {
|
||||
if (File) {
|
||||
return fextl::make_unique<FEXCore::Config::MainLoader>(*File);
|
||||
}
|
||||
else {
|
||||
return fextl::make_unique<FEXCore::Config::MainLoader>(FEXCore::Config::LayerType::LAYER_MAIN);
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const fextl::string& Filename, FEXCore::Config::LayerType Type) {
|
||||
return fextl::make_unique<FEXCore::Config::AppLoader>(Filename, Type);
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateEnvironmentLayer(char *const _envp[]) {
|
||||
return fextl::make_unique<FEXCore::Config::EnvLoader>(_envp);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -251,6 +251,20 @@
|
||||
"Set this in an application configuration for injecting in to only specific applications.",
|
||||
"\tNote: If x86/x86_64 libSegFault.so isn't installed then this option won't work."
|
||||
]
|
||||
},
|
||||
"Disassemble": {
|
||||
"Type": "strenum",
|
||||
"Default": "FEXCore::Config::Disassemble::OFF",
|
||||
"Enums": {
|
||||
"DISPATCHER": "dispatcher",
|
||||
"BLOCKS": "blocks"
|
||||
},
|
||||
"Desc": [
|
||||
"Allows controlling of the vixl disassembler.",
|
||||
"\toff: No disassembly will be output",
|
||||
"\tdispatcher: Will enable disassembly of the JIT dispatcher loop",
|
||||
"\tblocks: Will enable disassembly of the translated instruction code blocks"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Logging": {
|
||||
|
||||
+3
-6
@@ -1,5 +1,4 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Core.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
@@ -27,7 +26,9 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
bool FEXCore::Context::ContextImpl::InitializeContext() {
|
||||
return FEXCore::CPU::CreateCPUCore(this);
|
||||
// This should be used for generating things that are shared between threads
|
||||
CPUID.Init(this);
|
||||
return true;
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetExitHandler(ExitHandler handler) {
|
||||
@@ -66,10 +67,6 @@ namespace FEXCore::Context {
|
||||
CustomCPUFactory = std::move(Factory);
|
||||
}
|
||||
|
||||
bool FEXCore::Context::ContextImpl::AddVirtualMemoryMapping([[maybe_unused]] uint64_t VirtualAddress, [[maybe_unused]] uint64_t PhysicalAddress, [[maybe_unused]] uint64_t Size) {
|
||||
return false;
|
||||
}
|
||||
|
||||
HostFeatures FEXCore::Context::ContextImpl::GetHostFeatures() const {
|
||||
return HostFeatures;
|
||||
}
|
||||
|
||||
+20
-18
@@ -100,8 +100,6 @@ namespace FEXCore::Context {
|
||||
|
||||
void SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) override;
|
||||
|
||||
bool AddVirtualMemoryMapping(uint64_t VirtualAddress, uint64_t PhysicalAddress, uint64_t Size) override;
|
||||
|
||||
HostFeatures GetHostFeatures() const override;
|
||||
|
||||
void HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) override;
|
||||
@@ -154,7 +152,11 @@ namespace FEXCore::Context {
|
||||
* @param Thread The internal FEX thread state object
|
||||
*/
|
||||
void DestroyThread(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
void CleanupAfterFork(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
|
||||
#ifndef _WIN32
|
||||
void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child) override;
|
||||
#endif
|
||||
void SetSignalDelegator(FEXCore::SignalDelegator *SignalDelegation) override;
|
||||
void SetSyscallHandler(FEXCore::HLE::SyscallHandler *Handler) override;
|
||||
|
||||
@@ -165,29 +167,29 @@ namespace FEXCore::Context {
|
||||
FEXCore::IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(const fextl::string& Name) override;
|
||||
void UnloadAOTIRCacheEntry(FEXCore::IR::AOTIRCacheEntry *Entry) override;
|
||||
|
||||
void SetAOTIRLoader(std::function<int(const fextl::string&)> CacheReader) override {
|
||||
IRCaptureCache.SetAOTIRLoader(CacheReader);
|
||||
void SetAOTIRLoader(AOTIRLoaderCBFn CacheReader) override {
|
||||
IRCaptureCache.SetAOTIRLoader(std::move(CacheReader));
|
||||
}
|
||||
void SetAOTIRWriter(std::function<fextl::unique_ptr<AOTIRWriter>(const fextl::string&)> CacheWriter) override {
|
||||
IRCaptureCache.SetAOTIRWriter(CacheWriter);
|
||||
void SetAOTIRWriter(AOTIRWriterCBFn CacheWriter) override {
|
||||
IRCaptureCache.SetAOTIRWriter(std::move(CacheWriter));
|
||||
}
|
||||
void SetAOTIRRenamer(std::function<void(const fextl::string&)> CacheRenamer) override {
|
||||
IRCaptureCache.SetAOTIRRenamer(CacheRenamer);
|
||||
void SetAOTIRRenamer(AOTIRRenamerCBFn CacheRenamer) override {
|
||||
IRCaptureCache.SetAOTIRRenamer(std::move(CacheRenamer));
|
||||
}
|
||||
|
||||
void FinalizeAOTIRCache() override {
|
||||
IRCaptureCache.FinalizeAOTIRCache();
|
||||
}
|
||||
void WriteFilesWithCode(std::function<void(const fextl::string& fileid, const fextl::string& filename)> Writer) override {
|
||||
void WriteFilesWithCode(AOTIRCodeFileWriterFn Writer) override {
|
||||
IRCaptureCache.WriteFilesWithCode(Writer);
|
||||
}
|
||||
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) override;
|
||||
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, std::function<void(uint64_t start, uint64_t Length)> callback) override;
|
||||
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, CodeRangeInvalidationFn callback) override;
|
||||
void MarkMemoryShared() override;
|
||||
|
||||
void ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, fextl::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) override;
|
||||
// returns false if a handler was already registered
|
||||
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)> Handler, void *Creator = nullptr, void *Data = nullptr) override;
|
||||
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void *Creator = nullptr, void *Data = nullptr) override;
|
||||
|
||||
void AppendThunkDefinitions(fextl::vector<FEXCore::IR::ThunkDefinition> const& Definitions) override;
|
||||
|
||||
@@ -254,7 +256,7 @@ namespace FEXCore::Context {
|
||||
Event PauseWait;
|
||||
bool Running{};
|
||||
|
||||
std::shared_mutex CodeInvalidationMutex;
|
||||
FEXCore::ForkableSharedMutex CodeInvalidationMutex;
|
||||
|
||||
FEXCore::CPUIDEmu CPUID;
|
||||
FEXCore::HLE::SyscallHandler *SyscallHandler{};
|
||||
@@ -291,7 +293,7 @@ namespace FEXCore::Context {
|
||||
template<auto Fn>
|
||||
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
|
||||
auto Thread = Frame->Thread;
|
||||
ScopedDeferredSignalWithSharedLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
ScopedDeferredSignalWithForkableSharedLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
return Fn(Frame, record);
|
||||
}
|
||||
@@ -302,8 +304,7 @@ namespace FEXCore::Context {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == FHU::Syscalls::gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), FHU::Syscalls::gettid());
|
||||
|
||||
ScopedDeferredSignalWithUniqueLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
ScopedDeferredSignalWithForkableUniqueLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
ThreadRemoveCodeEntry(Thread, GuestRIP);
|
||||
}
|
||||
@@ -380,6 +381,8 @@ namespace FEXCore::Context {
|
||||
|
||||
bool ExitOnHLTEnabled() const { return ExitOnHLT; }
|
||||
|
||||
FEXCore::CPU::CPUBackendFeatures BackendFeatures;
|
||||
|
||||
protected:
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
@@ -434,8 +437,7 @@ namespace FEXCore::Context {
|
||||
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
|
||||
|
||||
std::shared_mutex CustomIRMutex;
|
||||
fextl::unordered_map<uint64_t, std::tuple<std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)>, void *, void *>> CustomIRHandlers;
|
||||
FEXCore::CPU::CPUBackendFeatures BackendFeatures;
|
||||
fextl::unordered_map<uint64_t, std::tuple<CustomIREntrypointHandler, void *, void *>> CustomIRHandlers;
|
||||
FEXCore::CPU::DispatcherConfig DispatcherConfig;
|
||||
};
|
||||
|
||||
|
||||
+213
-76
@@ -20,6 +20,131 @@
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
// Register x18 is unused in the current configuration.
|
||||
// This is due to it being a platform register on wine platforms.
|
||||
// TODO: Allow x18 register allocation in the future to gain one more register.
|
||||
|
||||
namespace x64 {
|
||||
// All but x19 and x29 are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 16> SRA = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
|
||||
FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
|
||||
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 9> RA = {
|
||||
// All these callee saved
|
||||
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
|
||||
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
|
||||
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
|
||||
FEXCore::ARMEmitter::Reg::r30,
|
||||
};
|
||||
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 4> RAPair = {{
|
||||
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
|
||||
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
|
||||
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
|
||||
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
|
||||
}};
|
||||
|
||||
// All are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 16> SRAFPR = {
|
||||
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17,
|
||||
FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
|
||||
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21,
|
||||
FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
|
||||
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
|
||||
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
|
||||
};
|
||||
|
||||
// v8..v15 = (lower 64bits) Callee saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 12> RAFPR = {
|
||||
// v0 ~ v3 are used as temps.
|
||||
// FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
|
||||
// FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
|
||||
|
||||
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
|
||||
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
|
||||
};
|
||||
}
|
||||
|
||||
namespace x32 {
|
||||
// All but x19 and x29 are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 8> SRA = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
|
||||
FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 17> RA = {
|
||||
// All these callee saved
|
||||
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
|
||||
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
|
||||
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
|
||||
|
||||
// Registers only available on 32-bit
|
||||
// All these are caller saved (except for r19).
|
||||
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r29, FEXCore::ARMEmitter::Reg::r30,
|
||||
|
||||
FEXCore::ARMEmitter::Reg::r19,
|
||||
};
|
||||
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 8> RAPair = {{
|
||||
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
|
||||
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
|
||||
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
|
||||
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
|
||||
|
||||
{FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13},
|
||||
{FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15},
|
||||
{FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17},
|
||||
{FEXCore::ARMEmitter::Reg::r29, FEXCore::ARMEmitter::Reg::r30},
|
||||
}};
|
||||
|
||||
// All are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 8> SRAFPR = {
|
||||
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17,
|
||||
FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
|
||||
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21,
|
||||
FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
|
||||
};
|
||||
|
||||
// v8..v15 = (lower 64bits) Callee saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 20> RAFPR = {
|
||||
// v0 ~ v3 are used as temps.
|
||||
// FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
|
||||
// FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
|
||||
|
||||
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
|
||||
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
|
||||
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
|
||||
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
|
||||
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
|
||||
};
|
||||
}
|
||||
|
||||
// We want vixl to not allocate a default buffer. Jit and dispatcher will manually create one.
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, size_t size)
|
||||
@@ -29,22 +154,21 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, size_t size)
|
||||
|
||||
// Number of register available is dependent on what operating mode the proccess is in.
|
||||
if (EmitterCTX->Config.Is64BitMode()) {
|
||||
ConfiguredGPRs = NumGPRs64;
|
||||
ConfiguredSRAGPRs = NumSRAGPRs64;
|
||||
ConfiguredGPRPairs = NumGPRPairs64;
|
||||
ConfiguredFPRs = NumFPRs64;
|
||||
ConfiguredSRAFPRs = NumSRAFPRs64;
|
||||
ConfiguredDynamicGPRs = NumGPRs64 - NumGPRs64; // Will be zero, just to be consistent with 32-bit side
|
||||
ConfiguredDynamicRegisterBase = nullptr;
|
||||
StaticRegisters = x64::SRA;
|
||||
GeneralRegisters = x64::RA;
|
||||
GeneralPairRegisters = x64::RAPair;
|
||||
StaticFPRegisters = x64::SRAFPR;
|
||||
GeneralFPRegisters = x64::RAFPR;
|
||||
}
|
||||
else {
|
||||
ConfiguredGPRs = NumGPRs32;
|
||||
ConfiguredSRAGPRs = NumSRAGPRs32;
|
||||
ConfiguredGPRPairs = NumGPRPairs32;
|
||||
ConfiguredFPRs = NumFPRs32;
|
||||
ConfiguredSRAFPRs = NumSRAFPRs32;
|
||||
ConfiguredDynamicGPRs = NumGPRs32 - NumGPRs64; // Will be 8
|
||||
ConfiguredDynamicRegisterBase = &RA64[9];
|
||||
ConfiguredDynamicRegisterBase = std::span(x32::RA.begin() + 8, 8);
|
||||
|
||||
StaticRegisters = x32::SRA;
|
||||
GeneralRegisters = x32::RA;
|
||||
GeneralPairRegisters = x32::RAPair;
|
||||
|
||||
StaticFPRegisters = x32::SRAFPR;
|
||||
GeneralFPRegisters = x32::RAFPR;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -68,6 +192,23 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
|
||||
return;
|
||||
}
|
||||
|
||||
if ((Constant >> 32) == 0) {
|
||||
// If the upper 32-bits is all zero, we can now switch to a 32-bit move.
|
||||
s = ARMEmitter::Size::i32Bit;
|
||||
Is64Bit = false;
|
||||
Segments = 2;
|
||||
}
|
||||
|
||||
// If this can be loaded with a mov bitmask.
|
||||
const auto IsImm = vixl::aarch64::Assembler::IsImmLogical(Constant, RegSizeInBits(s));
|
||||
if (IsImm) {
|
||||
orr(s, Reg, ARMEmitter::Reg::zr, Constant);
|
||||
if (NOPPad) {
|
||||
nop(); nop(); nop();
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
int NumMoves = 1;
|
||||
int RequiredMoveSegments{};
|
||||
|
||||
@@ -224,9 +365,9 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
|
||||
return;
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < ConfiguredSRAGPRs; i+=2) {
|
||||
auto Reg1 = SRA64[i];
|
||||
auto Reg2 = SRA64[i+1];
|
||||
for (size_t i = 0; i < StaticRegisters.size(); i+=2) {
|
||||
auto Reg1 = StaticRegisters[i];
|
||||
auto Reg2 = StaticRegisters[i+1];
|
||||
if (((1U << Reg1.Idx()) & GPRSpillMask) &&
|
||||
((1U << Reg2.Idx()) & GPRSpillMask)) {
|
||||
stp<ARMEmitter::IndexType::OFFSET>(Reg1.X(), Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
|
||||
@@ -241,8 +382,8 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
|
||||
|
||||
if (FPRs) {
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i++) {
|
||||
const auto Reg = SRAFPR[i];
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i++) {
|
||||
const auto Reg = StaticFPRegisters[i];
|
||||
|
||||
if (((1U << Reg.Idx()) & FPRSpillMask) != 0) {
|
||||
mov(ARMEmitter::Size::i64Bit, TMP4.R(), offsetof(Core::CpuStateFrame, State.xmm.avx.data[i][0]));
|
||||
@@ -254,18 +395,18 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
|
||||
// Optimize the common case where we can spill four registers per instruction
|
||||
// Load the sse offset in to the temporary register
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[0][0]));
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i += 4) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
const auto Reg3 = SRAFPR[i + 2];
|
||||
const auto Reg4 = SRAFPR[i + 3];
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
const auto Reg3 = StaticFPRegisters[i + 2];
|
||||
const auto Reg4 = StaticFPRegisters[i + 3];
|
||||
st1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
|
||||
}
|
||||
}
|
||||
else {
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i += 2) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 2) {
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
|
||||
if (((1U << Reg1.Idx()) & FPRSpillMask) &&
|
||||
((1U << Reg2.Idx()) & FPRSpillMask)) {
|
||||
@@ -297,8 +438,8 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
ptrue<ARMEmitter::SubRegSize::i8Bit>(PRED_TMP_16B, ARMEmitter::PredicatePattern::SVE_VL16);
|
||||
ptrue<ARMEmitter::SubRegSize::i8Bit>(PRED_TMP_32B, ARMEmitter::PredicatePattern::SVE_VL32);
|
||||
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i++) {
|
||||
const auto Reg = SRAFPR[i];
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i++) {
|
||||
const auto Reg = StaticFPRegisters[i];
|
||||
if (((1U << Reg.Idx()) & FPRFillMask) != 0) {
|
||||
mov(ARMEmitter::Size::i64Bit, TMP4.R(), offsetof(Core::CpuStateFrame, State.xmm.avx.data[i][0]));
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Reg.Z(), PRED_TMP_32B.Zeroing(), STATE.R(), TMP4.R());
|
||||
@@ -308,22 +449,22 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
if (GPRFillMask && FPRFillMask == ~0U) {
|
||||
// Optimize the common case where we can fill four registers per instruction.
|
||||
// Use one of the filling static registers before we fill it.
|
||||
auto TmpReg = SRA64[FindFirstSetBit(GPRFillMask)];
|
||||
auto TmpReg = StaticRegisters[FindFirstSetBit(GPRFillMask)];
|
||||
|
||||
// Load the sse offset in to the temporary register
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[0][0]));
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i += 4) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
const auto Reg3 = SRAFPR[i + 2];
|
||||
const auto Reg4 = SRAFPR[i + 3];
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
const auto Reg3 = StaticFPRegisters[i + 2];
|
||||
const auto Reg4 = StaticFPRegisters[i + 3];
|
||||
ld1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
|
||||
}
|
||||
}
|
||||
else {
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i += 2) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 2) {
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
|
||||
if (((1U << Reg1.Idx()) & FPRFillMask) &&
|
||||
((1U << Reg2.Idx()) & FPRFillMask)) {
|
||||
@@ -340,9 +481,9 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
}
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < ConfiguredSRAGPRs; i+=2) {
|
||||
auto Reg1 = SRA64[i];
|
||||
auto Reg2 = SRA64[i+1];
|
||||
for (size_t i = 0; i < StaticRegisters.size(); i+=2) {
|
||||
auto Reg1 = StaticRegisters[i];
|
||||
auto Reg2 = StaticRegisters[i+1];
|
||||
if (((1U << Reg1.Idx()) & GPRFillMask) &&
|
||||
((1U << Reg2.Idx()) & GPRFillMask)) {
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(Reg1.X(), Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
|
||||
@@ -358,10 +499,10 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
|
||||
void Arm64Emitter::PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg) {
|
||||
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
|
||||
const auto GPRSize = (ConfiguredDynamicGPRs + 1) * Core::CPUState::GPR_REG_SIZE;
|
||||
const auto GPRSize = (ConfiguredDynamicRegisterBase.size() + 1) * Core::CPUState::GPR_REG_SIZE;
|
||||
const auto FPRRegSize = CanUseSVE ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto FPRSize = ConfiguredFPRs * FPRRegSize;
|
||||
const auto FPRSize = GeneralFPRegisters.size() * FPRRegSize;
|
||||
const uint64_t SPOffset = AlignUp(GPRSize + FPRSize, 16);
|
||||
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
|
||||
@@ -370,31 +511,29 @@ void Arm64Emitter::PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg) {
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, ARMEmitter::Reg::rsp, 0);
|
||||
|
||||
if (CanUseSVE) {
|
||||
for (size_t i = 0; i < ConfiguredFPRs; i += 4) {
|
||||
const auto Reg1 = RAFPR[i];
|
||||
const auto Reg2 = RAFPR[i + 1];
|
||||
const auto Reg3 = RAFPR[i + 2];
|
||||
const auto Reg4 = RAFPR[i + 3];
|
||||
for (size_t i = 0; i < GeneralFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = GeneralFPRegisters[i];
|
||||
const auto Reg2 = GeneralFPRegisters[i + 1];
|
||||
const auto Reg3 = GeneralFPRegisters[i + 2];
|
||||
const auto Reg4 = GeneralFPRegisters[i + 3];
|
||||
st4b(Reg1.Z(), Reg2.Z(), Reg3.Z(), Reg4.Z(), PRED_TMP_32B, TmpReg, 0);
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, TmpReg, 32 * 4);
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(ConfiguredFPRs % 4 == 0, "Needs to have multiple of 4 FPRs for RA");
|
||||
for (size_t i = 0; i < ConfiguredFPRs; i += 4) {
|
||||
const auto Reg1 = RAFPR[i];
|
||||
const auto Reg2 = RAFPR[i + 1];
|
||||
const auto Reg3 = RAFPR[i + 2];
|
||||
const auto Reg4 = RAFPR[i + 3];
|
||||
LOGMAN_THROW_A_FMT(GeneralFPRegisters.size() % 4 == 0, "Needs to have multiple of 4 FPRs for RA");
|
||||
for (size_t i = 0; i < GeneralFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = GeneralFPRegisters[i];
|
||||
const auto Reg2 = GeneralFPRegisters[i + 1];
|
||||
const auto Reg3 = GeneralFPRegisters[i + 2];
|
||||
const auto Reg4 = GeneralFPRegisters[i + 3];
|
||||
st1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
|
||||
}
|
||||
}
|
||||
|
||||
if (ConfiguredDynamicRegisterBase) {
|
||||
for (size_t i = 0; i < ConfiguredDynamicGPRs; i += 2) {
|
||||
const auto Reg1 = ConfiguredDynamicRegisterBase[i];
|
||||
const auto Reg2 = ConfiguredDynamicRegisterBase[i + 1];
|
||||
stp<ARMEmitter::IndexType::POST>(Reg1.X(), Reg2.X(), TmpReg, 16);
|
||||
}
|
||||
for (size_t i = 0; i < ConfiguredDynamicRegisterBase.size(); i += 2) {
|
||||
const auto Reg1 = ConfiguredDynamicRegisterBase[i];
|
||||
const auto Reg2 = ConfiguredDynamicRegisterBase[i + 1];
|
||||
stp<ARMEmitter::IndexType::POST>(Reg1.X(), Reg2.X(), TmpReg, 16);
|
||||
}
|
||||
|
||||
str(ARMEmitter::XReg::lr, TmpReg, 0);
|
||||
@@ -404,30 +543,28 @@ void Arm64Emitter::PopDynamicRegsAndLR() {
|
||||
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
|
||||
|
||||
if (CanUseSVE) {
|
||||
for (size_t i = 0; i < ConfiguredFPRs; i += 4) {
|
||||
const auto Reg1 = RAFPR[i];
|
||||
const auto Reg2 = RAFPR[i + 1];
|
||||
const auto Reg3 = RAFPR[i + 2];
|
||||
const auto Reg4 = RAFPR[i + 3];
|
||||
for (size_t i = 0; i < GeneralFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = GeneralFPRegisters[i];
|
||||
const auto Reg2 = GeneralFPRegisters[i + 1];
|
||||
const auto Reg3 = GeneralFPRegisters[i + 2];
|
||||
const auto Reg4 = GeneralFPRegisters[i + 3];
|
||||
ld4b(Reg1.Z(), Reg2.Z(), Reg3.Z(), Reg4.Z(), PRED_TMP_32B.Zeroing(), ARMEmitter::Reg::rsp);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 32 * 4);
|
||||
}
|
||||
} else {
|
||||
for (size_t i = 0; i < ConfiguredFPRs; i += 4) {
|
||||
const auto Reg1 = RAFPR[i];
|
||||
const auto Reg2 = RAFPR[i + 1];
|
||||
const auto Reg3 = RAFPR[i + 2];
|
||||
const auto Reg4 = RAFPR[i + 3];
|
||||
for (size_t i = 0; i < GeneralFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = GeneralFPRegisters[i];
|
||||
const auto Reg2 = GeneralFPRegisters[i + 1];
|
||||
const auto Reg3 = GeneralFPRegisters[i + 2];
|
||||
const auto Reg4 = GeneralFPRegisters[i + 3];
|
||||
ld1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), ARMEmitter::Reg::rsp, 64);
|
||||
}
|
||||
}
|
||||
|
||||
if (ConfiguredDynamicRegisterBase) {
|
||||
for (size_t i = 0; i < ConfiguredDynamicGPRs; i += 2) {
|
||||
const auto Reg1 = ConfiguredDynamicRegisterBase[i];
|
||||
const auto Reg2 = ConfiguredDynamicRegisterBase[i + 1];
|
||||
ldp<ARMEmitter::IndexType::POST>(Reg1.X(), Reg2.X(), ARMEmitter::Reg::rsp, 16);
|
||||
}
|
||||
for (size_t i = 0; i < ConfiguredDynamicRegisterBase.size(); i += 2) {
|
||||
const auto Reg1 = ConfiguredDynamicRegisterBase[i];
|
||||
const auto Reg2 = ConfiguredDynamicRegisterBase[i + 1];
|
||||
ldp<ARMEmitter::IndexType::POST>(Reg1.X(), Reg2.X(), ARMEmitter::Reg::rsp, 16);
|
||||
}
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
|
||||
@@ -26,87 +26,9 @@
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <utility>
|
||||
#include <span>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
// Register x18 is unused in the current configuration.
|
||||
// This is due to it being a platform register on wine platforms.
|
||||
// TODO: Allow x18 register allocation in the future to gain one more register.
|
||||
|
||||
// All but x19 and x29 are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 16> SRA64 = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
|
||||
FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
|
||||
// Registers that don't exist on 32-bit
|
||||
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 9 + 8> RA64 = {
|
||||
// All these callee saved
|
||||
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
|
||||
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
|
||||
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
|
||||
FEXCore::ARMEmitter::Reg::r30,
|
||||
|
||||
// Registers only available on 32-bit
|
||||
// All these are caller saved (except for r19).
|
||||
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29
|
||||
};
|
||||
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 4 + 3> RA64Pair = {{
|
||||
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
|
||||
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
|
||||
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
|
||||
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
|
||||
|
||||
// Registers only available on 32-bit
|
||||
{FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13},
|
||||
{FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15},
|
||||
{FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17}
|
||||
}};
|
||||
|
||||
// All are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 16> SRAFPR = {
|
||||
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17,
|
||||
FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
|
||||
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21,
|
||||
FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
|
||||
|
||||
// Registers that don't exist on 32-bit
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
|
||||
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
|
||||
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
|
||||
};
|
||||
|
||||
// v8..v15 = (lower 64bits) Callee saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 12 + 8> RAFPR = {
|
||||
// v0 ~ v3 are used as temps.
|
||||
// FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
|
||||
// FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
|
||||
|
||||
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
|
||||
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
|
||||
|
||||
// Registers only available on 32-bit
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
|
||||
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
|
||||
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
|
||||
};
|
||||
|
||||
// Contains the address to the currently available CPU state
|
||||
constexpr auto STATE = FEXCore::ARMEmitter::XReg::x28;
|
||||
|
||||
@@ -139,32 +61,16 @@ protected:
|
||||
FEXCore::Context::ContextImpl *EmitterCTX;
|
||||
vixl::aarch64::CPU CPU;
|
||||
|
||||
uint32_t ConfiguredGPRs;
|
||||
uint32_t ConfiguredSRAGPRs;
|
||||
uint32_t ConfiguredGPRPairs;
|
||||
uint32_t ConfiguredFPRs;
|
||||
uint32_t ConfiguredSRAFPRs;
|
||||
uint32_t ConfiguredDynamicGPRs;
|
||||
const FEXCore::ARMEmitter::Register *ConfiguredDynamicRegisterBase{};
|
||||
std::span<const FEXCore::ARMEmitter::Register> ConfiguredDynamicRegisterBase{};
|
||||
std::span<const FEXCore::ARMEmitter::Register> StaticRegisters{};
|
||||
std::span<const FEXCore::ARMEmitter::Register> GeneralRegisters{};
|
||||
std::span<const std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>> GeneralPairRegisters{};
|
||||
std::span<const FEXCore::ARMEmitter::VRegister> StaticFPRegisters{};
|
||||
std::span<const FEXCore::ARMEmitter::VRegister> GeneralFPRegisters{};
|
||||
|
||||
/**
|
||||
* @name Register Allocation
|
||||
* @{ */
|
||||
// 64-bit gets removal of additional pairs
|
||||
constexpr static uint32_t NumGPRs64 = RA64.size() - 8;
|
||||
constexpr static uint32_t NumSRAGPRs64 = SRA64.size();
|
||||
constexpr static uint32_t NumFPRs64 = RAFPR.size() - 8;
|
||||
constexpr static uint32_t NumSRAFPRs64 = SRAFPR.size();
|
||||
constexpr static uint32_t NumGPRPairs64 = RA64Pair.size() - 3;
|
||||
|
||||
// 32-bit gets full array of GPR registers
|
||||
// SRA registers remove the additional 8
|
||||
constexpr static uint32_t NumGPRs32 = RA64.size();
|
||||
constexpr static uint32_t NumSRAGPRs32 = SRA64.size() - 8;
|
||||
constexpr static uint32_t NumFPRs32 = RAFPR.size();
|
||||
constexpr static uint32_t NumSRAFPRs32 = SRAFPR.size() - 8;
|
||||
constexpr static uint32_t NumGPRPairs32 = RA64Pair.size();
|
||||
|
||||
constexpr static uint32_t RegisterClasses = 6;
|
||||
|
||||
constexpr static uint64_t GPRBase = (0ULL << 32);
|
||||
@@ -269,6 +175,7 @@ protected:
|
||||
#endif
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
vixl::aarch64::PrintDisassembler Disasm {stderr};
|
||||
FEX_CONFIG_OPT(Disassemble, DISASSEMBLE);
|
||||
#endif
|
||||
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
|
||||
};
|
||||
|
||||
+164
-4
@@ -129,7 +129,9 @@ public:
|
||||
constexpr uint32_t Op = 0b0011'0001'0 << 23;
|
||||
DataProcessing_AddSub_Imm(Op, s, rd, rn, Imm, LSL12);
|
||||
}
|
||||
|
||||
void cmn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, uint32_t Imm, bool LSL12 = false) {
|
||||
adds(s, FEXCore::ARMEmitter::Reg::zr, rn, Imm, LSL12);
|
||||
}
|
||||
void sub(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t Imm, bool LSL12 = false) {
|
||||
constexpr uint32_t Op = 0b0101'0001'0 << 23;
|
||||
DataProcessing_AddSub_Imm(Op, s, rd, rn, Imm, LSL12);
|
||||
@@ -145,6 +147,27 @@ public:
|
||||
DataProcessing_AddSub_Imm(Op, s, rd, rn, Imm, LSL12);
|
||||
}
|
||||
|
||||
// Min/max immediate
|
||||
void smax(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, int64_t Imm) {
|
||||
LOGMAN_THROW_A_FMT(Imm >= -128 && Imm <= 127, "{} Immediate too large", __func__);
|
||||
MinMaxImmediate(0b0000, s, rd, rn, Imm);
|
||||
}
|
||||
|
||||
void umax(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
|
||||
LOGMAN_THROW_A_FMT(Imm <= 255, "{} Immediate too large", __func__);
|
||||
MinMaxImmediate(0b0001, s, rd, rn, Imm);
|
||||
}
|
||||
|
||||
void smin(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, int64_t Imm) {
|
||||
LOGMAN_THROW_A_FMT(Imm >= -128 && Imm <= 127, "{} Immediate too large", __func__);
|
||||
MinMaxImmediate(0b0010, s, rd, rn, Imm);
|
||||
}
|
||||
|
||||
void umin(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
|
||||
LOGMAN_THROW_A_FMT(Imm <= 255, "{} Immediate too large", __func__);
|
||||
MinMaxImmediate(0b0011, s, rd, rn, Imm);
|
||||
}
|
||||
|
||||
// Logical immediate
|
||||
void and_(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
|
||||
uint32_t n, immr, imms;
|
||||
@@ -198,6 +221,10 @@ public:
|
||||
eor(s, rd, rn, n, immr, imms);
|
||||
}
|
||||
|
||||
void tst(ARMEmitter::Size s, Register rn, uint64_t imm) {
|
||||
ands(s, Reg::zr, rn, imm);
|
||||
}
|
||||
|
||||
// Move wide immediate
|
||||
void movn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, uint32_t Imm, uint32_t Offset = 0) {
|
||||
LOGMAN_THROW_A_FMT((Imm & 0xFFFF0000U) == 0, "Upper bits of move wide not valid");
|
||||
@@ -265,6 +292,9 @@ public:
|
||||
LOGMAN_THROW_A_FMT((lsb + width) <= RegSizeInBits(s), "Tried to sbfx a region larger than the register");
|
||||
sbfm(s, rd, rn, lsb, lsb + width - 1);
|
||||
}
|
||||
void sbfiz(ARMEmitter::Size s, Register rd, Register rn, uint32_t lsb, uint32_t width) {
|
||||
xbfiz_helper(true, s, rd, rn, lsb, width);
|
||||
}
|
||||
void asr(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t shift) {
|
||||
LOGMAN_THROW_A_FMT(shift <= RegSizeInBits(s), "Tried to asr a region larger than the register");
|
||||
sbfm(s, rd, rn, shift, RegSizeInBits(s) - 1);
|
||||
@@ -285,6 +315,10 @@ public:
|
||||
DataProcessing_Logical_Imm(Op, s, rd, rn, s == ARMEmitter::Size::i64Bit, immr, imms);
|
||||
}
|
||||
|
||||
void ubfiz(ARMEmitter::Size s, Register rd, Register rn, uint32_t lsb, uint32_t width) {
|
||||
xbfiz_helper(false, s, rd, rn, lsb, width);
|
||||
}
|
||||
|
||||
void lsl(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t shift) {
|
||||
const auto RegSize = RegSizeInBits(s);
|
||||
LOGMAN_THROW_A_FMT(shift < RegSize, "Tried to lsl a region larger than the register");
|
||||
@@ -303,10 +337,23 @@ public:
|
||||
|
||||
void bfi(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t lsb, uint32_t width) {
|
||||
const auto RegSize = RegSizeInBits(s);
|
||||
LOGMAN_THROW_A_FMT(width > 0, "bfi needs width > 0");
|
||||
LOGMAN_THROW_A_FMT((lsb + width) <= RegSize, "Tried to bfi a region larger than the register");
|
||||
LOGMAN_THROW_A_FMT(width > 0, "bfc/bfi needs width > 0");
|
||||
LOGMAN_THROW_A_FMT((lsb + width) <= RegSize, "Tried to bfc/bfi a region larger than the register");
|
||||
bfm(s, rd, rn, (RegSize - lsb) & (RegSize - 1), width - 1);
|
||||
}
|
||||
void bfc(ARMEmitter::Size s, Register rd, uint32_t lsb, uint32_t width) {
|
||||
bfi(s, rd, Reg::zr, lsb, width);
|
||||
}
|
||||
void bfxil(ARMEmitter::Size s, Register rd, Register rn, uint32_t lsb, uint32_t width) {
|
||||
[[maybe_unused]] const auto reg_size_bits = RegSizeInBits(s);
|
||||
const auto lsb_p_width = lsb + width;
|
||||
|
||||
LOGMAN_THROW_A_FMT(width >= 1, "bfxil needs width >= 1");
|
||||
LOGMAN_THROW_A_FMT(lsb_p_width <= reg_size_bits, "bfxil lsb + width ({}) must be <= {}. lsb={}, width={}",
|
||||
lsb_p_width, reg_size_bits, lsb, width);
|
||||
|
||||
bfm(s, rd, rn, lsb, lsb_p_width - 1);
|
||||
}
|
||||
|
||||
// Extract
|
||||
void extr(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, uint32_t Imm) {
|
||||
@@ -381,6 +428,26 @@ public:
|
||||
(0b0101'10U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
|
||||
}
|
||||
void smax(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0110'00U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void umax(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0110'01U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void smin(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0110'10U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void umin(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0110'11U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void subp(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0000'00U << 10);
|
||||
@@ -467,7 +534,24 @@ public:
|
||||
(s == ARMEmitter::Size::i64Bit ? (1U << 10) : 0);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
|
||||
void ctz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0001'10U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void cnt(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0001'11U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void abs(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0010'00U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
|
||||
// TODO: PAUTH
|
||||
|
||||
@@ -506,6 +590,9 @@ public:
|
||||
constexpr uint32_t Op = 0b010'1010'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void tst(ARMEmitter::Size s, Register rn, Register rm, ShiftType shift = ShiftType::LSL, uint32_t amt = 0) {
|
||||
ands(s, Reg::zr, rn, rm, shift, amt);
|
||||
}
|
||||
|
||||
void orn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
constexpr uint32_t Op = 0b010'1010'001U << 21;
|
||||
@@ -527,6 +614,9 @@ public:
|
||||
void adds(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
adds(ARMEmitter::Size::i64Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void cmn(FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
adds(ARMEmitter::Size::i64Bit, FEXCore::ARMEmitter::XReg::zr, rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void sub(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
sub(ARMEmitter::Size::i64Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
@@ -549,6 +639,9 @@ public:
|
||||
void adds(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
adds(ARMEmitter::Size::i32Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void cmn(FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
adds(ARMEmitter::Size::i32Bit, FEXCore::ARMEmitter::WReg::zr, rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void sub(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
sub(ARMEmitter::Size::i32Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
@@ -575,6 +668,9 @@ public:
|
||||
constexpr uint32_t Op = 0b010'1011'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void cmn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
adds(s, FEXCore::ARMEmitter::Reg::zr, rn, rm, Shift, amt);
|
||||
}
|
||||
void sub(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
LOGMAN_THROW_AA_FMT(Shift != FEXCore::ARMEmitter::ShiftType::ROR, "Doesn't support ROR");
|
||||
constexpr uint32_t Op = 0b100'1011'000U << 21;
|
||||
@@ -606,6 +702,9 @@ public:
|
||||
constexpr uint32_t Op = 0b010'1011'001U << 21;
|
||||
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, Option, Shift);
|
||||
}
|
||||
void cmn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ExtendedType Option, uint32_t Shift = 0) {
|
||||
adds(s, FEXCore::ARMEmitter::Reg::zr, rn, rm, Option, Shift);
|
||||
}
|
||||
void sub(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ExtendedType Option, uint32_t Shift = 0) {
|
||||
constexpr uint32_t Op = 0b100'1011'001U << 21;
|
||||
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, Option, Shift);
|
||||
@@ -636,6 +735,12 @@ public:
|
||||
constexpr uint32_t Op = 0b0111'1010'000U << 21;
|
||||
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, FEXCore::ARMEmitter::ExtendedType::UXTB, 0);
|
||||
}
|
||||
void ngc(ARMEmitter::Size s, Register rd, Register rm) {
|
||||
sbc(s, rd, Reg::zr, rm);
|
||||
}
|
||||
void ngcs(ARMEmitter::Size s, Register rd, Register rm) {
|
||||
sbcs(s, rd, Reg::zr, rm);
|
||||
}
|
||||
|
||||
// Rotate right into flags
|
||||
void rmif(XRegister rn, uint32_t shift, uint32_t mask) {
|
||||
@@ -703,6 +808,18 @@ public:
|
||||
constexpr uint32_t Op = 0b0001'1010'100 << 21;
|
||||
ConditionalCompare(Op, 1, 0b01, s, rd, rn, rm, Cond);
|
||||
}
|
||||
void cneg(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Condition Cond) {
|
||||
csneg(s, rd, rn, rn, InvertCondition(Cond));
|
||||
}
|
||||
void cinc(ARMEmitter::Size s, Register rd, Register rn, Condition cond) {
|
||||
csinc(s, rd, rn, rn, InvertCondition(cond));
|
||||
}
|
||||
void cinv(ARMEmitter::Size s, Register rd, Register rn, Condition cond) {
|
||||
csinv(s, rd, rn, rn, InvertCondition(cond));
|
||||
}
|
||||
void csetm(ARMEmitter::Size s, Register rd, Condition cond) {
|
||||
csinv(s, rd, Reg::zr, Reg::zr, InvertCondition(cond));
|
||||
}
|
||||
|
||||
// Data processing - 3 source
|
||||
void madd(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::Register ra) {
|
||||
@@ -757,6 +874,13 @@ public:
|
||||
}
|
||||
|
||||
private:
|
||||
static constexpr Condition InvertCondition(Condition cond) {
|
||||
// These behave as always, so it makes no sense to allow inverting these.
|
||||
LOGMAN_THROW_AA_FMT(cond != Condition::CC_AL && cond != Condition::CC_NV,
|
||||
"Cannot invert CC_AL or CC_NV");
|
||||
return static_cast<Condition>(FEXCore::ToUnderlying(cond) ^ 1);
|
||||
}
|
||||
|
||||
void and_(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t n, uint32_t immr, uint32_t imms) {
|
||||
constexpr uint32_t Op = 0b001'0010'00 << 22;
|
||||
DataProcessing_Logical_Imm(Op, s, rd, rn, n, immr, imms);
|
||||
@@ -819,6 +943,21 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Min/max immediate
|
||||
void MinMaxImmediate(uint32_t opc, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = 0b1'0001'11U << 22;
|
||||
|
||||
Instr |= SF;
|
||||
Instr |= opc << 18;
|
||||
Instr |= (Imm & 0xFF) << 10;
|
||||
Instr |= Encode_rn(rn);
|
||||
Instr |= Encode_rd(rd);
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Move Wide
|
||||
void DataProcessing_MoveWide(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, uint32_t Imm, uint32_t Offset) {
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
@@ -849,6 +988,24 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
void xbfiz_helper(bool is_signed, ARMEmitter::Size s, Register rd, Register rn, uint32_t lsb, uint32_t width) {
|
||||
[[maybe_unused]] const auto lsb_p_width = lsb + width;
|
||||
const auto reg_size_bits = RegSizeInBits(s);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(lsb_p_width <= reg_size_bits, "lsb + width ({}) must be <= {}. lsb={}, width={}",
|
||||
lsb_p_width, reg_size_bits, lsb, width);
|
||||
LOGMAN_THROW_AA_FMT(width >= 1, "xbfiz width must be >= 1");
|
||||
|
||||
const auto immr = (reg_size_bits - lsb) & (reg_size_bits - 1);
|
||||
const auto imms = width - 1;
|
||||
|
||||
if (is_signed) {
|
||||
sbfm(s, rd, rn, immr, imms);
|
||||
} else {
|
||||
ubfm(s, rd, rn, immr, imms);
|
||||
}
|
||||
}
|
||||
|
||||
void DataProcessing_Extract(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, uint32_t Imm) {
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
@@ -899,6 +1056,9 @@ private:
|
||||
// AddSub - shifted register
|
||||
void DataProcessing_Shifted_Reg(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift, uint32_t amt) {
|
||||
LOGMAN_THROW_AA_FMT((amt & ~0b11'1111U) == 0, "Shift amount too large");
|
||||
if (s == FEXCore::ARMEmitter::Size::i32Bit) {
|
||||
LOGMAN_THROW_AA_FMT(amt < 32, "Shift amount for 32-bit must be below 32");
|
||||
}
|
||||
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
|
||||
@@ -6,6 +6,7 @@
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
@@ -207,58 +208,94 @@ namespace FEXCore::ARMEmitter {
|
||||
*/
|
||||
class SVEMemOperand final {
|
||||
public:
|
||||
// Used for scalar + vector variants to determine
|
||||
// extension behavior on the index values.
|
||||
enum class ModType : uint8_t {
|
||||
MOD_UXTW,
|
||||
MOD_SXTW,
|
||||
MOD_LSL,
|
||||
MOD_NONE,
|
||||
};
|
||||
|
||||
enum class Type {
|
||||
ScalarPlusScalar,
|
||||
ScalarPlusImm,
|
||||
ScalarPlusVector,
|
||||
VectorPlusImm,
|
||||
};
|
||||
|
||||
SVEMemOperand(XRegister rn, XRegister rm = XReg::zr)
|
||||
: rn {rn}
|
||||
, MemType{Type::ScalarPlusScalar}
|
||||
, MetaType {
|
||||
.ScalarScalarType {
|
||||
.Header = { .MemType = TYPE_SCALAR_SCALAR },
|
||||
.rm = rm,
|
||||
}
|
||||
} {}
|
||||
SVEMemOperand(XRegister rn, int32_t imm = 0)
|
||||
: rn {rn}
|
||||
, MemType{Type::ScalarPlusImm}
|
||||
, MetaType {
|
||||
.ScalarImmType {
|
||||
.Header = { .MemType = TYPE_SCALAR_IMM },
|
||||
.Imm = imm,
|
||||
}
|
||||
} {}
|
||||
SVEMemOperand(XRegister rn, ZRegister zm, ModType mod = ModType::MOD_NONE, uint8_t scale = 0)
|
||||
: rn{rn}
|
||||
, MemType{Type::ScalarPlusVector}
|
||||
, MetaType {
|
||||
.ScalarVectorType {
|
||||
.zm = zm,
|
||||
.mod = mod,
|
||||
.scale = scale,
|
||||
}
|
||||
} {}
|
||||
SVEMemOperand(ZRegister zn, uint32_t imm)
|
||||
: rn{Register{zn.Idx()}}
|
||||
, MemType{Type::VectorPlusImm}
|
||||
, MetaType {
|
||||
.VectorImmType{
|
||||
.Imm = imm,
|
||||
}
|
||||
} {}
|
||||
|
||||
Register rn;
|
||||
enum Type {
|
||||
TYPE_SCALAR_SCALAR,
|
||||
TYPE_SCALAR_IMM,
|
||||
TYPE_SCALAR_VECTOR,
|
||||
TYPE_VECTOR_IMM,
|
||||
};
|
||||
struct HeaderStruct {
|
||||
Type MemType;
|
||||
};
|
||||
[[nodiscard]] bool IsScalarPlusScalar() const {
|
||||
return MemType == Type::ScalarPlusScalar;
|
||||
}
|
||||
[[nodiscard]] bool IsScalarPlusImm() const {
|
||||
return MemType == Type::ScalarPlusImm;
|
||||
}
|
||||
[[nodiscard]] bool IsScalarPlusVector() const {
|
||||
return MemType == Type::ScalarPlusVector;
|
||||
}
|
||||
[[nodiscard]] bool IsVectorPlusImm() const {
|
||||
return MemType == Type::VectorPlusImm;
|
||||
}
|
||||
|
||||
union {
|
||||
HeaderStruct Header;
|
||||
union Data {
|
||||
struct {
|
||||
HeaderStruct Header;
|
||||
Register rm;
|
||||
} ScalarScalarType;
|
||||
|
||||
struct {
|
||||
HeaderStruct Header;
|
||||
int32_t Imm;
|
||||
} ScalarImmType;
|
||||
|
||||
struct {
|
||||
HeaderStruct Header;
|
||||
ZRegister zm;
|
||||
// TODO: Implement support for modifier
|
||||
ModType mod;
|
||||
uint8_t scale;
|
||||
} ScalarVectorType;
|
||||
|
||||
struct {
|
||||
HeaderStruct Header;
|
||||
// rn will be a ZRegister
|
||||
int32_t Imm;
|
||||
uint32_t Imm;
|
||||
} VectorImmType;
|
||||
} MetaType;
|
||||
};
|
||||
|
||||
Register rn;
|
||||
Type MemType;
|
||||
Data MetaType;
|
||||
};
|
||||
|
||||
/* This `ExtendedMemOperand` class is used for the helper load-store instructions.
|
||||
@@ -475,6 +512,20 @@ namespace FEXCore::ARMEmitter {
|
||||
SVE_ALL = 0b11111,
|
||||
};
|
||||
|
||||
// Used with SVE FP immediate arithmetic instructions
|
||||
enum class SVEFAddSubImm : uint32_t {
|
||||
_0_5,
|
||||
_1_0,
|
||||
};
|
||||
enum class SVEFMulImm : uint32_t {
|
||||
_0_5,
|
||||
_2_0,
|
||||
};
|
||||
enum class SVEFMaxMinImm : uint32_t {
|
||||
_0_0,
|
||||
_1_0,
|
||||
};
|
||||
|
||||
/* This `BackwardLabel` struct used for retaining a location for PC-Relative instructions.
|
||||
* This is specifically a label for a target that is logically `below` an instruction that uses it.
|
||||
* Which means that a branch would jump backwards.
|
||||
|
||||
+436
-546
File diff suppressed because it is too large.
Load diff
+1356
-1238
File diff suppressed because it is too large.
Load diff
+3
-5
@@ -395,8 +395,6 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
uint32_t CoreCount = Cores();
|
||||
// XXX: Enable once the rest of the SSE4.2 instructions are emulated
|
||||
uint32_t SupportsSSE42 = CTX->HostFeatures.SupportsCRC && false ? 1 : 0;
|
||||
|
||||
// Hypervisor bit is normally set but some applications have issues with it.
|
||||
uint32_t Hypervisor = HideHypervisorBit() ? 0 : 1;
|
||||
@@ -429,14 +427,14 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
(0 << 17) | // Process-context identifiers
|
||||
(0 << 18) | // Prefetching from memory mapped device
|
||||
(1 << 19) | // SSE4.1
|
||||
(SupportsSSE42 << 20) | // SSE4.2
|
||||
(CTX->HostFeatures.SupportsCRC << 20) | // SSE4.2
|
||||
(0 << 21) | // X2APIC
|
||||
(1 << 22) | // MOVBE
|
||||
(1 << 23) | // POPCNT
|
||||
(0 << 24) | // APIC TSC-Deadline
|
||||
(CTX->HostFeatures.SupportsAES << 25) | // AES
|
||||
(0 << 26) | // XSAVE
|
||||
(0 << 27) | // OSXSAVE
|
||||
(SupportsAVX() << 26) | // XSAVE
|
||||
(SupportsAVX() << 27) | // OSXSAVE
|
||||
(SupportsAVX() << 28) | // AVX
|
||||
(0 << 29) | // F16C
|
||||
(CTX->HostFeatures.SupportsRAND << 30) | // RDRAND
|
||||
|
||||
+55
-27
@@ -8,9 +8,9 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include <cstdint>
|
||||
#include "FEXCore/Utils/DeferredSignalMutex.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/Core.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/GdbServer.h"
|
||||
@@ -24,6 +24,7 @@ $end_info$
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
#include "Interface/IR/Passes.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "Utils/Allocator.h"
|
||||
#include "Utils/Allocator/HostAllocator.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
@@ -75,22 +76,11 @@ $end_info$
|
||||
#include <utility>
|
||||
#include <xxhash.h>
|
||||
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
bool CreateCPUCore(Context::ContextImpl *CTX) {
|
||||
// This should be used for generating things that are shared between threads
|
||||
CTX->CPUID.Init(CTX);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct ThreadLocalData {
|
||||
FEXCore::Core::InternalThreadState* Thread;
|
||||
};
|
||||
|
||||
thread_local ThreadLocalData ThreadData{};
|
||||
|
||||
constexpr std::array<std::string_view const, 22> FlagNames = {
|
||||
"CF",
|
||||
"",
|
||||
@@ -195,17 +185,35 @@ namespace FEXCore::Context {
|
||||
uint64_t ContextImpl::RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState *Thread, uint64_t HostPC) {
|
||||
const auto Frame = Thread->CurrentFrame;
|
||||
const uint64_t BlockBegin = Frame->State.InlineJITBlockHeader;
|
||||
const CPU::CPUBackend::JITCodeHeader *InlineHeader = reinterpret_cast<const CPU::CPUBackend::JITCodeHeader *>(BlockBegin);
|
||||
auto InlineHeader = reinterpret_cast<const CPU::CPUBackend::JITCodeHeader *>(BlockBegin);
|
||||
|
||||
if (InlineHeader) {
|
||||
const CPU::CPUBackend::JITCodeTail *InlineTail = reinterpret_cast<const CPU::CPUBackend::JITCodeTail *>(Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail);
|
||||
auto InlineTail = reinterpret_cast<const CPU::CPUBackend::JITCodeTail *>(Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail);
|
||||
auto RIPEntries = reinterpret_cast<const CPU::CPUBackend::JITRIPReconstructEntries *>(Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail + InlineTail->OffsetToRIPEntries);
|
||||
|
||||
// Check if the host PC is currently within a code block.
|
||||
// If it is then RIP can be reconstructed from the beginning of the code block.
|
||||
// This is currently as close as FEX can get RIP reconstructions.
|
||||
if (HostPC >= reinterpret_cast<uint64_t>(BlockBegin) &&
|
||||
HostPC < reinterpret_cast<uint64_t>(BlockBegin + InlineTail->Size)) {
|
||||
return InlineTail->RIP;
|
||||
|
||||
// Reconstruct RIP from JIT entries for this block.
|
||||
uint64_t StartingHostPC = BlockBegin;
|
||||
uint64_t StartingGuestRIP = InlineTail->RIP;
|
||||
|
||||
for (uint32_t i = 0; i < InlineTail->NumberOfRIPEntries; ++i) {
|
||||
const auto &RIPEntry = RIPEntries[i];
|
||||
if (HostPC >= (StartingHostPC + RIPEntry.HostPCOffset)) {
|
||||
// We are beyond this entry, keep going forward.
|
||||
StartingHostPC += RIPEntry.HostPCOffset;
|
||||
StartingGuestRIP += RIPEntry.GuestRIPOffset;
|
||||
}
|
||||
else {
|
||||
// Passed where the Host PC is at. Break now.
|
||||
break;
|
||||
}
|
||||
}
|
||||
return StartingGuestRIP;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -645,7 +653,18 @@ namespace FEXCore::Context {
|
||||
delete Thread;
|
||||
}
|
||||
|
||||
void ContextImpl::CleanupAfterFork(FEXCore::Core::InternalThreadState *LiveThread) {
|
||||
#ifndef _WIN32
|
||||
void ContextImpl::UnlockAfterFork(FEXCore::Core::InternalThreadState *LiveThread, bool Child) {
|
||||
Allocator::UnlockAfterFork(LiveThread, Child);
|
||||
|
||||
if (Child) {
|
||||
CodeInvalidationMutex.StealAndDropActiveLocks();
|
||||
}
|
||||
else {
|
||||
CodeInvalidationMutex.unlock();
|
||||
return;
|
||||
}
|
||||
|
||||
// This function is called after fork
|
||||
// We need to cleanup some of the thread data that is dead
|
||||
for (auto &DeadThread : Threads) {
|
||||
@@ -684,6 +703,12 @@ namespace FEXCore::Context {
|
||||
FEXCore::Threads::Thread::CleanupAfterFork();
|
||||
}
|
||||
|
||||
void ContextImpl::LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) {
|
||||
CodeInvalidationMutex.lock();
|
||||
Allocator::LockBeforeFork(Thread);
|
||||
}
|
||||
#endif
|
||||
|
||||
void ContextImpl::AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr) {
|
||||
Thread->LookupCache->AddBlockMapping(Address, Ptr);
|
||||
}
|
||||
@@ -811,7 +836,7 @@ namespace FEXCore::Context {
|
||||
DecodedInfo = &Block.DecodedInstructions[i];
|
||||
bool IsLocked = DecodedInfo->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK;
|
||||
|
||||
if (ExtendedDebugInfo) {
|
||||
if (ExtendedDebugInfo || Thread->OpDispatcher->CanHaveSideEffects(TableInfo, DecodedInfo)) {
|
||||
Thread->OpDispatcher->_GuestOpcode(Block.Entry + BlockInstructionsLength - GuestRIP);
|
||||
}
|
||||
|
||||
@@ -838,14 +863,14 @@ namespace FEXCore::Context {
|
||||
|
||||
if (TableInfo && TableInfo->OpcodeDispatcher) {
|
||||
auto Fn = TableInfo->OpcodeDispatcher;
|
||||
Thread->OpDispatcher->HandledLock = false;
|
||||
Thread->OpDispatcher->ResetHandledLock();
|
||||
Thread->OpDispatcher->ResetDecodeFailure();
|
||||
std::invoke(Fn, Thread->OpDispatcher, DecodedInfo);
|
||||
if (Thread->OpDispatcher->HadDecodeFailure()) {
|
||||
HadDispatchError = true;
|
||||
}
|
||||
else {
|
||||
if (Thread->OpDispatcher->HandledLock != IsLocked) {
|
||||
if (Thread->OpDispatcher->HasHandledLock() != IsLocked) {
|
||||
HadDispatchError = true;
|
||||
LogMan::Msg::EFmt("Missing LOCK HANDLER at 0x{:x}{{'{}'}}", Block.Entry + BlockInstructionsLength, TableInfo->Name ?: "UND");
|
||||
}
|
||||
@@ -895,7 +920,7 @@ namespace FEXCore::Context {
|
||||
|
||||
IR::IREmitter *IREmitter = Thread->OpDispatcher.get();
|
||||
|
||||
auto ShouldDump = static_cast<ContextImpl*>(Thread->CTX)->Config.DumpIR() != "no" || Thread->OpDispatcher->ShouldDump;
|
||||
auto ShouldDump = static_cast<ContextImpl*>(Thread->CTX)->Config.DumpIR() != "no" || Thread->OpDispatcher->ShouldDumpIR();
|
||||
// Debug
|
||||
{
|
||||
if (ShouldDump) {
|
||||
@@ -1030,7 +1055,7 @@ namespace FEXCore::Context {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
|
||||
std::shared_lock lk(CodeInvalidationMutex);
|
||||
ScopedDeferredSignalWithForkableSharedLock lk(CodeInvalidationMutex, Thread);
|
||||
|
||||
// Is the code in the cache?
|
||||
// The backends only check L1 and L2, not L3
|
||||
@@ -1126,11 +1151,12 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
void ContextImpl::ExecutionThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
Core::ThreadData.Thread = Thread;
|
||||
Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_WAITING;
|
||||
|
||||
InitializeThreadTLSData(Thread);
|
||||
#ifndef _WIN32
|
||||
Alloc::OSAllocator::RegisterTLSData(Thread);
|
||||
#endif
|
||||
|
||||
++IdleWaitRefCount;
|
||||
|
||||
@@ -1175,7 +1201,9 @@ namespace FEXCore::Context {
|
||||
--IdleWaitRefCount;
|
||||
IdleWaitCV.notify_all();
|
||||
|
||||
#ifndef _WIN32
|
||||
Alloc::OSAllocator::UninstallTLSData(Thread);
|
||||
#endif
|
||||
SignalDelegation->UninstallTLSState(Thread);
|
||||
|
||||
// If the parent thread is waiting to join, then we can't destroy our thread object
|
||||
@@ -1210,16 +1238,16 @@ namespace FEXCore::Context {
|
||||
// Potential deferred since Thread might not be valid.
|
||||
// Thread object isn't valid very early in frontend's initialization.
|
||||
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
|
||||
ScopedPotentialDeferredSignalWithUniqueLock CodeInvalidationLock(CodeInvalidationMutex, Thread);
|
||||
ScopedPotentialDeferredSignalWithForkableUniqueLock lk(CodeInvalidationMutex, Thread);
|
||||
|
||||
InvalidateGuestCodeRangeInternal(this, Start, Length);
|
||||
}
|
||||
|
||||
void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, std::function<void(uint64_t start, uint64_t Length)> CallAfter) {
|
||||
void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, CodeRangeInvalidationFn CallAfter) {
|
||||
// Potential deferred since Thread might not be valid.
|
||||
// Thread object isn't valid very early in frontend's initialization.
|
||||
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
|
||||
ScopedPotentialDeferredSignalWithUniqueLock CodeInvalidationLock(CodeInvalidationMutex, Thread);
|
||||
ScopedPotentialDeferredSignalWithForkableUniqueLock lk(CodeInvalidationMutex, Thread);
|
||||
|
||||
InvalidateGuestCodeRangeInternal(this, Start, Length);
|
||||
CallAfter(Start, Length);
|
||||
@@ -1247,7 +1275,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
void ContextImpl::ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker) {
|
||||
std::shared_lock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex);
|
||||
ScopedDeferredSignalWithForkableSharedLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
Thread->LookupCache->AddBlockLink(GuestDestination, HostLink, delinker);
|
||||
}
|
||||
@@ -1261,7 +1289,7 @@ namespace FEXCore::Context {
|
||||
Thread->LookupCache->Erase(GuestRIP);
|
||||
}
|
||||
|
||||
CustomIRResult ContextImpl::AddCustomIREntrypoint(uintptr_t Entrypoint, std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)> Handler, void *Creator, void *Data) {
|
||||
CustomIRResult ContextImpl::AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void *Creator, void *Data) {
|
||||
LOGMAN_THROW_A_FMT(Config.Is64BitMode || !(Entrypoint >> 32), "64-bit Entrypoint in 32-bit mode {:x}", Entrypoint);
|
||||
|
||||
std::unique_lock lk(CustomIRMutex);
|
||||
|
||||
-23
@@ -1,23 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
namespace FEXCore {
|
||||
class CodeLoader;
|
||||
}
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
/**
|
||||
* @brief Create the CPU core backend for the context passed in
|
||||
*
|
||||
* @param CTX
|
||||
*
|
||||
* @return true if core was able to be create
|
||||
*/
|
||||
bool CreateCPUCore(FEXCore::Context::ContextImpl *CTX);
|
||||
|
||||
bool LoadCode(FEXCore::Context::ContextImpl *CTX, FEXCore::CodeLoader *Loader);
|
||||
}
|
||||
@@ -503,8 +503,10 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
}
|
||||
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
const auto DisasmEnd = GetCursorAddress<const vixl::aarch64::Instruction*>();
|
||||
Disasm.DisassembleBuffer(DisasmBegin, DisasmEnd);
|
||||
if (Disassemble() & FEXCore::Config::Disassemble::DISPATCHER) {
|
||||
const auto DisasmEnd = GetCursorAddress<const vixl::aarch64::Instruction*>();
|
||||
Disasm.DisassembleBuffer(DisasmBegin, DisasmEnd);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
|
||||
@@ -31,22 +31,22 @@ class Arm64Dispatcher final : public Dispatcher, public Arm64Emitter {
|
||||
void EmitDispatcher();
|
||||
|
||||
uint16_t GetSRAGPRCount() const override {
|
||||
return SRA64.size();
|
||||
return StaticRegisters.size();
|
||||
}
|
||||
|
||||
uint16_t GetSRAFPRCount() const override {
|
||||
return SRAFPR.size();
|
||||
return StaticFPRegisters.size();
|
||||
}
|
||||
|
||||
void GetSRAGPRMapping(uint8_t Mapping[16]) const override {
|
||||
for (size_t i = 0; i < SRA64.size(); ++i) {
|
||||
Mapping[i] = SRA64[i].Idx();
|
||||
for (size_t i = 0; i < StaticRegisters.size(); ++i) {
|
||||
Mapping[i] = StaticRegisters[i].Idx();
|
||||
}
|
||||
}
|
||||
|
||||
void GetSRAFPRMapping(uint8_t Mapping[16]) const override {
|
||||
for (size_t i = 0; i < SRAFPR.size(); ++i) {
|
||||
Mapping[i] = SRAFPR[i].Idx();
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); ++i) {
|
||||
Mapping[i] = StaticFPRegisters[i].Idx();
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+18
-13
@@ -10,7 +10,6 @@ $end_info$
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <array>
|
||||
#include <assert.h>
|
||||
#include <algorithm>
|
||||
#include <cstring>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
@@ -239,8 +238,19 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModR
|
||||
Operand->Data.SIB.Scale = 1 << SIB.scale;
|
||||
|
||||
// The invalid encoding types are described at Table 1-12. "promoted nsigned is always non-zero"
|
||||
Operand->Data.SIB.Index = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_X ? 1 : 0, SIB.index, false, false, false, false, 0b100);
|
||||
Operand->Data.SIB.Base = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, SIB.base, false, false, false, false, ModRM.mod == 0 ? 0b101 : 16);
|
||||
{
|
||||
// If we have a VSIB byte (as opposed to SIB), then the index register is a vector.
|
||||
const bool IsIndexVector = (DecodeInst->TableInfo->Flags & InstFlags::FLAGS_VEX_VSIB) != 0;
|
||||
if (IsIndexVector) {
|
||||
DecodeInst->Flags |= X86Tables::DecodeFlags::FLAG_VSIB_BYTE;
|
||||
}
|
||||
|
||||
const uint8_t IndexREX = (DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_X) != 0 ? 1 : 0;
|
||||
const uint8_t BaseREX = (DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B) != 0 ? 1 : 0;
|
||||
|
||||
Operand->Data.SIB.Index = MapModRMToReg(IndexREX, SIB.index, false, false, IsIndexVector, false, 0b100);
|
||||
Operand->Data.SIB.Base = MapModRMToReg(BaseREX, SIB.base, false, false, false, false, ModRM.mod == 0 ? 0b101 : 16);
|
||||
}
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Displacement <= 4, "Number of bytes should be <= 4 for literal src");
|
||||
|
||||
@@ -671,9 +681,6 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
DecodeInst->ModRM = ModRMByte;
|
||||
DecodeInst->DecodedModRM = true;
|
||||
|
||||
FEXCore::X86Tables::ModRMDecoded ModRM;
|
||||
ModRM.Hex = DecodeInst->ModRM;
|
||||
|
||||
uint16_t X87Op = ((Op - 0xD8) << 8) | ModRMByte;
|
||||
return NormalOp(&X87Ops[X87Op], X87Op);
|
||||
}
|
||||
@@ -957,7 +964,7 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
|
||||
}
|
||||
|
||||
if (DecodeInst->Dest.IsGPR()) {
|
||||
assert(DecodeInst->Dest.Data.GPR.GPR != FEXCore::X86State::REG_INVALID);
|
||||
LOGMAN_THROW_A_FMT(DecodeInst->Dest.Data.GPR.GPR != FEXCore::X86State::REG_INVALID, "Destination GPR was invalid");
|
||||
}
|
||||
|
||||
return true;
|
||||
@@ -1015,15 +1022,13 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
|
||||
// If we are conditional then a target can be the instruction past the conditional instruction
|
||||
uint64_t FallthroughRIP = DecodeInst->PC + DecodeInst->InstSize;
|
||||
if (HasBlocks.find(FallthroughRIP) == HasBlocks.end() &&
|
||||
BlocksToDecode.find(FallthroughRIP) == BlocksToDecode.end()) {
|
||||
BlocksToDecode.emplace(FallthroughRIP);
|
||||
if (!HasBlocks.contains(FallthroughRIP)) {
|
||||
BlocksToDecode.insert(FallthroughRIP);
|
||||
}
|
||||
}
|
||||
|
||||
if (HasBlocks.find(TargetRIP) == HasBlocks.end() &&
|
||||
BlocksToDecode.find(TargetRIP) == BlocksToDecode.end()) {
|
||||
BlocksToDecode.emplace(TargetRIP);
|
||||
if (!HasBlocks.contains(TargetRIP)) {
|
||||
BlocksToDecode.insert(TargetRIP);
|
||||
}
|
||||
} else {
|
||||
if (ExternalBranches) {
|
||||
|
||||
@@ -71,6 +71,7 @@ HostFeatures::HostFeatures() {
|
||||
SupportsRCPC = Features.Has(vixl::CPUFeatures::Feature::kRCpc);
|
||||
SupportsTSOImm9 = Features.Has(vixl::CPUFeatures::Feature::kRCpcImm);
|
||||
SupportsPMULL_128Bit = Features.Has(vixl::CPUFeatures::Feature::kPmull1Q);
|
||||
SupportsCSSC = Features.Has(vixl::CPUFeatures::Feature::kCSSC);
|
||||
|
||||
Supports3DNow = true;
|
||||
SupportsSSE4A = true;
|
||||
|
||||
@@ -113,6 +113,22 @@ DEF_OP(Neg) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Abs) {
|
||||
auto Op = IROp->C<IR::IROp_Abs>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
const int64_t Src = *GetSrc<int64_t*>(Data->SSAData, Op->Src);
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
GD = std::abs(static_cast<int32_t>(Src));
|
||||
break;
|
||||
case 8:
|
||||
GD = std::abs(static_cast<int64_t>(Src));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Abs Size: {}\n", OpSize); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Mul) {
|
||||
auto Op = IROp->C<IR::IROp_Mul>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
@@ -869,12 +885,8 @@ DEF_OP(Select) {
|
||||
auto Op = IROp->C<IR::IROp_Select>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp2);
|
||||
|
||||
uint64_t ArgTrue;
|
||||
uint64_t ArgFalse;
|
||||
|
||||
if (OpSize == 4) {
|
||||
ArgTrue = *GetSrc<uint32_t*>(Data->SSAData, Op->TrueVal);
|
||||
ArgFalse = *GetSrc<uint32_t*>(Data->SSAData, Op->FalseVal);
|
||||
@@ -885,10 +897,25 @@ DEF_OP(Select) {
|
||||
|
||||
bool CompResult;
|
||||
|
||||
if (Op->CompareSize == 4)
|
||||
if (Op->CompareSize == 4) {
|
||||
const auto Src1 = *GetSrc<uint32_t*>(Data->SSAData, Op->Cmp1);
|
||||
const auto Src2 = *GetSrc<uint32_t*>(Data->SSAData, Op->Cmp2);
|
||||
CompResult = IsConditionTrue<uint32_t, int32_t, float>(Op->Cond.Val, Src1, Src2);
|
||||
else
|
||||
}
|
||||
else if (Op->CompareSize == 8) {
|
||||
const auto Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp1);
|
||||
const auto Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp2);
|
||||
CompResult = IsConditionTrue<uint64_t, int64_t, double>(Op->Cond.Val, Src1, Src2);
|
||||
}
|
||||
else if (Op->CompareSize == 16) {
|
||||
const auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Cmp1);
|
||||
const auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Cmp2);
|
||||
CompResult = IsConditionTrue<__uint128_t, __int128_t, double>(Op->Cond.Val, Src1, Src2);
|
||||
}
|
||||
else {
|
||||
LOGMAN_MSG_A_FMT("Unknown select size: {}", Op->CompareSize);
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
GD = CompResult ? ArgTrue : ArgFalse;
|
||||
}
|
||||
|
||||
@@ -211,7 +211,7 @@ DEF_OP(Vector_FToF) {
|
||||
// Little bit tricky here
|
||||
// Sometimes is used to convert from a 128bit vector register
|
||||
// in to a 64bit vector register with different sized elements
|
||||
// eg: %ssa5 i32v2 = Vector_FToF %ssa4 i128, #0x8
|
||||
// eg: %5 i32v2 = Vector_FToF %4 i128, #0x8
|
||||
uint8_t Elements = OpSize == 8 ? 2 : OpSize / Op->SrcElementSize;
|
||||
DO_VECTOR_1SRC_2TYPE_OP_NOSIZE(float, double, Func, 0, 0)
|
||||
break;
|
||||
|
||||
+1
-1
@@ -3,7 +3,7 @@
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
enum IROps : uint8_t;
|
||||
enum IROps : uint16_t;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
+1
-1
@@ -46,7 +46,7 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
|
||||
// Main PCMPXSTRX algorithm body. Allows for reuse with both implicit and explicit length variants.
|
||||
static uint32_t MainBody(const __uint128_t& lhs, int valid_lhs, const __uint128_t& rhs, int valid_rhs, uint16_t control) {
|
||||
const uint32_t aggregation = PerformAggregation(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
const uint32_t upper_limit = (16U >> (control & 1)) - 1;
|
||||
const int32_t upper_limit = (16 >> (control & 1)) - 1;
|
||||
|
||||
// Bits are arranged as:
|
||||
// Bit #: 3 2 1 0
|
||||
|
||||
@@ -52,6 +52,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(ADD, Add);
|
||||
REGISTER_OP(SUB, Sub);
|
||||
REGISTER_OP(NEG, Neg);
|
||||
REGISTER_OP(ABS, Abs);
|
||||
REGISTER_OP(MUL, Mul);
|
||||
REGISTER_OP(UMUL, UMul);
|
||||
REGISTER_OP(DIV, Div);
|
||||
@@ -266,6 +267,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(VUMULL2, VUMull2);
|
||||
REGISTER_OP(VSMULL2, VSMull2);
|
||||
REGISTER_OP(VUABDL, VUABDL);
|
||||
REGISTER_OP(VUABDL2, VUABDL2);
|
||||
REGISTER_OP(VTBL1, VTBL1);
|
||||
REGISTER_OP(VREV64, VRev64);
|
||||
REGISTER_OP(VPCMPESTRX, VPCMPESTRX);
|
||||
|
||||
@@ -85,6 +85,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(Add);
|
||||
DEF_OP(Sub);
|
||||
DEF_OP(Neg);
|
||||
DEF_OP(Abs);
|
||||
DEF_OP(Mul);
|
||||
DEF_OP(UMul);
|
||||
DEF_OP(Div);
|
||||
@@ -292,6 +293,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(VUMull2);
|
||||
DEF_OP(VSMull2);
|
||||
DEF_OP(VUABDL);
|
||||
DEF_OP(VUABDL2);
|
||||
DEF_OP(VTBL1);
|
||||
DEF_OP(VRev64);
|
||||
DEF_OP(VPCMPESTRX);
|
||||
@@ -348,6 +350,15 @@ namespace FEXCore::CPU {
|
||||
template<typename unsigned_type, typename signed_type, typename float_type>
|
||||
[[nodiscard]] static bool IsConditionTrue(uint8_t Cond, uint64_t Src1, uint64_t Src2) {
|
||||
bool CompResult = false;
|
||||
if constexpr (sizeof(unsigned_type) == 16) {
|
||||
LOGMAN_THROW_A_FMT(Cond != FEXCore::IR::COND_FLU &&
|
||||
Cond != FEXCore::IR::COND_FGE &&
|
||||
Cond != FEXCore::IR::COND_FLEU &&
|
||||
Cond != FEXCore::IR::COND_FGT &&
|
||||
Cond != FEXCore::IR::COND_FU &&
|
||||
Cond != FEXCore::IR::COND_FNU, "Unsupported comparison for 128-bit floats");
|
||||
}
|
||||
|
||||
switch (Cond) {
|
||||
case FEXCore::IR::COND_EQ:
|
||||
CompResult = static_cast<unsigned_type>(Src1) == static_cast<unsigned_type>(Src2);
|
||||
|
||||
@@ -190,19 +190,31 @@ DEF_OP(LoadFlag) {
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += offsetof(FEXCore::Core::CPUState, flags[0]);
|
||||
ContextPtr += Op->Flag;
|
||||
uint8_t const *MemData = reinterpret_cast<uint8_t const*>(ContextPtr);
|
||||
GD = *MemData;
|
||||
|
||||
if (Op->Flag == 24 /* NZCV */) {
|
||||
uint32_t const *MemData = reinterpret_cast<uint32_t const*>(ContextPtr);
|
||||
GD = *MemData;
|
||||
} else {
|
||||
uint8_t const *MemData = reinterpret_cast<uint8_t const*>(ContextPtr);
|
||||
GD = *MemData;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreFlag) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
uint8_t Arg = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
uint32_t Arg = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += offsetof(FEXCore::Core::CPUState, flags[0]);
|
||||
ContextPtr += Op->Flag;
|
||||
uint8_t *MemData = reinterpret_cast<uint8_t*>(ContextPtr);
|
||||
*MemData = Arg;
|
||||
|
||||
if (Op->Flag == 24 /* NZCV */) {
|
||||
uint32_t *MemData = reinterpret_cast<uint32_t*>(ContextPtr);
|
||||
*MemData = Arg;
|
||||
} else {
|
||||
uint8_t *MemData = reinterpret_cast<uint8_t*>(ContextPtr);
|
||||
*MemData = Arg;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadMem) {
|
||||
|
||||
@@ -2227,6 +2227,33 @@ DEF_OP(VUABDL) {
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VUABDL2) {
|
||||
const auto Op = IROp->C<IR::IROp_VUABDL2>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Vector1);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Vector2);
|
||||
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE];
|
||||
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
|
||||
const auto Func8 = [](auto a, auto b) { return std::abs((int16_t)a - (int16_t)b); };
|
||||
const auto Func16 = [](auto a, auto b) { return std::abs((int32_t)a - (int32_t)b); };
|
||||
const auto Func32 = [](auto a, auto b) { return std::abs((int64_t)a - (int64_t)b); };
|
||||
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_2SRC_2TYPE_OP_TOP_SRC(2, uint16_t, uint8_t, Func8)
|
||||
DO_VECTOR_2SRC_2TYPE_OP_TOP_SRC(4, uint32_t, uint16_t, Func16)
|
||||
DO_VECTOR_2SRC_2TYPE_OP_TOP_SRC(8, uint64_t, uint32_t, Func32)
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VTBL1) {
|
||||
const auto Op = IROp->C<IR::IROp_VTBL1>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
@@ -2280,16 +2307,13 @@ DEF_OP(VRev64) {
|
||||
|
||||
DEF_OP(VPCMPESTRX) {
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPESTRX>();
|
||||
const auto Is64Bit = Op->GPRSize == 8;
|
||||
const auto Control = Op->Control;
|
||||
|
||||
const auto RAX = *GetSrc<uint64_t*>(Data->SSAData, Op->RAX);
|
||||
const auto RDX = *GetSrc<uint64_t*>(Data->SSAData, Op->RDX);
|
||||
const auto LHS = *GetSrc<__uint128_t*>(Data->SSAData, Op->LHS);
|
||||
const auto RHS = *GetSrc<__uint128_t*>(Data->SSAData, Op->RHS);
|
||||
|
||||
// We can be cheeky and encode the size at bit 8 to save a parameter
|
||||
const auto Control = Op->Control | (uint16_t(Is64Bit) << 8);
|
||||
|
||||
const auto Result = OpHandlers<IR::OP_VPCMPESTRX>::handle(RAX, RDX, LHS, RHS, Control);
|
||||
|
||||
memset(GDP, 0, sizeof(uint64_t));
|
||||
|
||||
+111
-12
@@ -4,6 +4,7 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
@@ -83,6 +84,21 @@ DEF_OP(Add) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(TestNZ) {
|
||||
auto Op = IROp->C<IR::IROp_TestNZ>();
|
||||
const uint8_t OpSize = Op->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src1.ID());
|
||||
tst(EmitSize, Src, Src);
|
||||
|
||||
// TODO: Optimize this out
|
||||
mrs(Dst, ARMEmitter::SystemRegister::NZCV);
|
||||
}
|
||||
|
||||
DEF_OP(Sub) {
|
||||
auto Op = IROp->C<IR::IROp_Sub>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
@@ -108,6 +124,26 @@ DEF_OP(Neg) {
|
||||
neg(EmitSize, GetReg(Node), GetReg(Op->Src.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(Abs) {
|
||||
auto Op = IROp->C<IR::IROp_Abs>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsCSSC) {
|
||||
// On CSSC supporting processors, this turns in to one instruction and doesn't modify flags.
|
||||
abs(EmitSize, Dst, Src);
|
||||
}
|
||||
else {
|
||||
cmp(EmitSize, Src, 0);
|
||||
cneg(EmitSize, Dst, Src, ARMEmitter::Condition::CC_MI);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Mul) {
|
||||
auto Op = IROp->C<IR::IROp_Mul>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
@@ -310,6 +346,40 @@ DEF_OP(Or) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Orlshl) {
|
||||
auto Op = IROp->C<IR::IROp_Orlshl>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
orr(EmitSize, Dst, Src1, Const << Op->BitShift);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
orr(EmitSize, Dst, Src1, Src2, ARMEmitter::ShiftType::LSL, Op->BitShift);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Orlshr) {
|
||||
auto Op = IROp->C<IR::IROp_Orlshr>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
orr(EmitSize, Dst, Src1, Const >> Op->BitShift);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
orr(EmitSize, Dst, Src1, Src2, ARMEmitter::ShiftType::LSR, Op->BitShift);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(And) {
|
||||
auto Op = IROp->C<IR::IROp_And>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
@@ -477,9 +547,9 @@ DEF_OP(PDep) {
|
||||
const auto IndexReg = TMP4.R();
|
||||
const auto ZeroReg = ARMEmitter::Reg::zr;
|
||||
|
||||
const auto InputReg = SRA64[0];
|
||||
const auto MaskReg = SRA64[1];
|
||||
const auto DestReg = SRA64[2];
|
||||
const auto InputReg = StaticRegisters[0];
|
||||
const auto MaskReg = StaticRegisters[1];
|
||||
const auto DestReg = StaticRegisters[2];
|
||||
|
||||
const auto SpillCode = 1U << InputReg.Idx() |
|
||||
1U << MaskReg.Idx() |
|
||||
@@ -633,7 +703,10 @@ DEF_OP(LDiv) {
|
||||
mov(EmitSize, ARMEmitter::Reg::r2, Divisor);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LDIVHandler));
|
||||
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
|
||||
// Move result to its destination register
|
||||
mov(EmitSize, Dst, ARMEmitter::Reg::r0);
|
||||
@@ -697,7 +770,11 @@ DEF_OP(LUDiv) {
|
||||
mov(EmitSize, ARMEmitter::Reg::r2, Divisor);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LUDIVHandler));
|
||||
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
|
||||
// Move result to its destination register
|
||||
mov(EmitSize, Dst, ARMEmitter::Reg::r0);
|
||||
|
||||
@@ -768,7 +845,11 @@ DEF_OP(LRem) {
|
||||
mov(EmitSize, ARMEmitter::Reg::r2, Divisor);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LREMHandler));
|
||||
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
|
||||
// Move result to its destination register
|
||||
mov(EmitSize, Dst, ARMEmitter::Reg::r0);
|
||||
|
||||
@@ -833,7 +914,11 @@ DEF_OP(LURem) {
|
||||
mov(EmitSize, ARMEmitter::Reg::r2, Divisor);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LUREMHandler));
|
||||
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
|
||||
// Move result to its destination register
|
||||
mov(EmitSize, Dst, ARMEmitter::Reg::r0);
|
||||
|
||||
@@ -1020,14 +1105,22 @@ DEF_OP(Bfi) {
|
||||
const auto SrcDst = GetReg(Op->Dest.ID());
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
mov(EmitSize, TMP1, SrcDst);
|
||||
bfi(EmitSize, TMP1, Src, Op->lsb, Op->Width);
|
||||
|
||||
if (OpSize == 8) {
|
||||
mov(EmitSize, Dst, TMP1.R());
|
||||
if (Dst == SrcDst) {
|
||||
// If Dst and SrcDst match then this turns in to a simple BFI instruction.
|
||||
bfi(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
}
|
||||
else {
|
||||
ubfx(EmitSize, Dst, TMP1, 0, OpSize * 8);
|
||||
// Destination didn't match the dst source register.
|
||||
// TODO: Inefficient until FEX can have RA constraints here.
|
||||
mov(EmitSize, TMP1, SrcDst);
|
||||
bfi(EmitSize, TMP1, Src, Op->lsb, Op->Width);
|
||||
|
||||
if (OpSize == 8) {
|
||||
mov(EmitSize, Dst, TMP1.R());
|
||||
}
|
||||
else {
|
||||
ubfx(EmitSize, Dst, TMP1, 0, OpSize * 8);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1085,6 +1178,7 @@ DEF_OP(Select) {
|
||||
const auto CompareEmitSize = Op->CompareSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
uint64_t Const;
|
||||
auto cc = MapSelectCC(Op->Cond);
|
||||
|
||||
if (IsGPR(Op->Cmp1.ID())) {
|
||||
const auto Src1 = GetReg(Op->Cmp1.ID());
|
||||
@@ -1095,7 +1189,14 @@ DEF_OP(Select) {
|
||||
const auto Src2 = GetReg(Op->Cmp2.ID());
|
||||
cmp(CompareEmitSize, Src1, Src2);
|
||||
}
|
||||
} else if (IsFPR(Op->Cmp1.ID())) {
|
||||
}
|
||||
else if (IsGPRPair(Op->Cmp1.ID())) {
|
||||
const auto Src1 = GetRegPair(Op->Cmp1.ID());
|
||||
const auto Src2 = GetRegPair(Op->Cmp2.ID());
|
||||
cmp(EmitSize, Src1.first, Src2.first);
|
||||
ccmp(EmitSize, Src1.second, Src2.second, ARMEmitter::StatusFlags::None, cc);
|
||||
}
|
||||
else if (IsFPR(Op->Cmp1.ID())) {
|
||||
const auto Src1 = GetVReg(Op->Cmp1.ID());
|
||||
const auto Src2 = GetVReg(Op->Cmp2.ID());
|
||||
fcmp(Op->CompareSize == 8 ? ARMEmitter::ScalarRegSize::i64Bit : ARMEmitter::ScalarRegSize::i32Bit, Src1, Src2);
|
||||
@@ -1103,8 +1204,6 @@ DEF_OP(Select) {
|
||||
LOGMAN_MSG_A_FMT("Select: Expected GPR or FPR");
|
||||
}
|
||||
|
||||
auto cc = MapSelectCC(Op->Cond);
|
||||
|
||||
uint64_t const_true, const_false;
|
||||
bool is_const_true = IsInlineConstant(Op->TrueVal, &const_true);
|
||||
bool is_const_false = IsInlineConstant(Op->FalseVal, &const_false);
|
||||
|
||||
+52
-16
@@ -450,16 +450,13 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPESTRX>();
|
||||
const auto Is64Bit = Op->GPRSize == 8;
|
||||
const auto Control = Op->Control;
|
||||
|
||||
const auto Src1 = GetVReg(Op->LHS.ID());
|
||||
const auto Src2 = GetVReg(Op->RHS.ID());
|
||||
const auto SrcRAX = GetReg(Op->RAX.ID());
|
||||
const auto SrcRDX = GetReg(Op->RDX.ID());
|
||||
|
||||
// We can be cheeky and encode the size at bit 8 to save a parameter
|
||||
const auto Control = Op->Control | (uint16_t(Is64Bit) << 8);
|
||||
|
||||
mov(ARMEmitter::XReg::x0, SrcRAX.X());
|
||||
mov(ARMEmitter::XReg::x1, SrcRDX.X());
|
||||
|
||||
@@ -552,7 +549,7 @@ static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Fram
|
||||
FEXCore::ARMEmitter::Emitter::ClearICache((void*)branch, 24);
|
||||
|
||||
// Add de-linking handler
|
||||
Context::ContextImpl::ThreadAddBlockLink(Thread, GuestRip, (uintptr_t)record, [branch, LinkerAddress]{
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [branch, LinkerAddress]{
|
||||
FEXCore::ARMEmitter::Emitter emit((uint8_t*)(branch), 24);
|
||||
FEXCore::ARMEmitter::ForwardLabel l_BranchHost;
|
||||
emit.ldr(FEXCore::ARMEmitter::XReg::x0, &l_BranchHost);
|
||||
@@ -566,7 +563,7 @@ static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Fram
|
||||
record[0] = HostCode;
|
||||
|
||||
// Add de-linking handler
|
||||
Context::ContextImpl::ThreadAddBlockLink(Thread, GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
record[0] = LinkerAddress;
|
||||
});
|
||||
}
|
||||
@@ -587,14 +584,14 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::In
|
||||
|
||||
RAPass->AllocateRegisterSet(RegisterClasses);
|
||||
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRClass, ConfiguredGPRs);
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRFixedClass, ConfiguredSRAGPRs);
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRClass, ConfiguredFPRs);
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRFixedClass, ConfiguredSRAFPRs);
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRPairClass, ConfiguredGPRPairs);
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRClass, GeneralRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRFixedClass, StaticRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRClass, GeneralFPRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRFixedClass, StaticFPRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRPairClass, GeneralPairRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::ComplexClass, 1);
|
||||
|
||||
for (uint32_t i = 0; i < ConfiguredGPRPairs; ++i) {
|
||||
for (uint32_t i = 0; i < GeneralPairRegisters.size(); ++i) {
|
||||
RAPass->AddRegisterConflict(FEXCore::IR::GPRClass, i * 2, FEXCore::IR::GPRPairClass, i);
|
||||
RAPass->AddRegisterConflict(FEXCore::IR::GPRClass, i * 2 + 1, FEXCore::IR::GPRPairClass, i);
|
||||
}
|
||||
@@ -727,6 +724,12 @@ bool Arm64JITCore::IsGPR(IR::NodeID Node) const {
|
||||
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
|
||||
}
|
||||
|
||||
bool Arm64JITCore::IsGPRPair(IR::NodeID Node) const {
|
||||
auto Class = GetRegClass(Node);
|
||||
|
||||
return Class == IR::GPRPairClass;
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
@@ -848,8 +851,10 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
REGISTER_OP(INLINEENTRYPOINTOFFSET, InlineEntrypointOffset);
|
||||
REGISTER_OP(CYCLECOUNTER, CycleCounter);
|
||||
REGISTER_OP(ADD, Add);
|
||||
REGISTER_OP(TESTNZ, TestNZ);
|
||||
REGISTER_OP(SUB, Sub);
|
||||
REGISTER_OP(NEG, Neg);
|
||||
REGISTER_OP(ABS, Abs);
|
||||
REGISTER_OP(MUL, Mul);
|
||||
REGISTER_OP(UMUL, UMul);
|
||||
REGISTER_OP(DIV, Div);
|
||||
@@ -859,6 +864,8 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
REGISTER_OP(MULH, MulH);
|
||||
REGISTER_OP(UMULH, UMulH);
|
||||
REGISTER_OP(OR, Or);
|
||||
REGISTER_OP(ORLSHL, Orlshl);
|
||||
REGISTER_OP(ORLSHR, Orlshr);
|
||||
REGISTER_OP(AND, And);
|
||||
REGISTER_OP(ANDN, Andn);
|
||||
REGISTER_OP(XOR, Xor);
|
||||
@@ -1075,6 +1082,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
REGISTER_OP(VUMULL2, VUMull2);
|
||||
REGISTER_OP(VSMULL2, VSMull2);
|
||||
REGISTER_OP(VUABDL, VUABDL);
|
||||
REGISTER_OP(VUABDL2, VUABDL2);
|
||||
REGISTER_OP(VTBL1, VTBL1);
|
||||
REGISTER_OP(VREV64, VRev64);
|
||||
#undef REGISTER_OP
|
||||
@@ -1100,28 +1108,54 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
}
|
||||
PendingTargetLabel = nullptr;
|
||||
|
||||
// CodeSize not including the tail data.
|
||||
const uint64_t CodeOnlySize = GetCursorAddress<uint8_t *>() - CodeData.BlockBegin;
|
||||
|
||||
// Add the JitCodeTail
|
||||
auto JITBlockTailLocation = GetCursorAddress<uint8_t *>();
|
||||
auto JITBlockTail = GetCursorAddress<JITCodeTail*>();
|
||||
CursorIncrement(sizeof(JITCodeTail));
|
||||
|
||||
auto JITRIPEntriesLocation = GetCursorAddress<uint8_t *>();
|
||||
auto JITRIPEntries = GetCursorAddress<JITRIPReconstructEntries*>();
|
||||
|
||||
CursorIncrement(sizeof(JITRIPReconstructEntries) * DebugData->GuestOpcodes.size());
|
||||
|
||||
// Put the block's RIP entry in the tail.
|
||||
// This will be used for RIP reconstruction in the future.
|
||||
// TODO: This needs to be a data RIP relocation once code caching works.
|
||||
// Current relocation code doesn't support this feature yet.
|
||||
JITBlockTail->RIP = Entry;
|
||||
|
||||
{
|
||||
// Store the RIP entries.
|
||||
JITBlockTail->NumberOfRIPEntries = DebugData->GuestOpcodes.size();
|
||||
JITBlockTail->OffsetToRIPEntries = JITRIPEntriesLocation - JITBlockTailLocation;
|
||||
uintptr_t CurrentRIPOffset = 0;
|
||||
uint64_t CurrentPCOffset = 0;
|
||||
for (size_t i = 0; i < DebugData->GuestOpcodes.size(); i++) {
|
||||
const auto &GuestOpcode = DebugData->GuestOpcodes[i];
|
||||
auto &RIPEntry = JITRIPEntries[i];
|
||||
RIPEntry.HostPCOffset = GuestOpcode.HostEntryOffset - CurrentPCOffset;
|
||||
RIPEntry.GuestRIPOffset = GuestOpcode.GuestEntryOffset - CurrentRIPOffset;
|
||||
CurrentPCOffset = GuestOpcode.HostEntryOffset;
|
||||
CurrentRIPOffset = GuestOpcode.GuestEntryOffset;
|
||||
}
|
||||
}
|
||||
|
||||
CodeHeader->OffsetToBlockTail = JITBlockTailLocation - CodeData.BlockBegin;
|
||||
|
||||
CodeData.Size = GetCursorAddress<uint8_t *>() - CodeData.BlockBegin;
|
||||
|
||||
JITBlockTail->Size = CodeData.Size;
|
||||
|
||||
ClearICache(CodeData.BlockBegin, CodeData.Size);
|
||||
ClearICache(CodeData.BlockBegin, CodeOnlySize);
|
||||
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
const auto DisasmEnd = GetCursorAddress<const vixl::aarch64::Instruction*>();
|
||||
Disasm.DisassembleBuffer(DisasmBegin, DisasmEnd);
|
||||
if (Disassemble() & FEXCore::Config::Disassemble::BLOCKS) {
|
||||
const auto DisasmEnd = reinterpret_cast<const vixl::aarch64::Instruction*>(JITBlockTailLocation);
|
||||
Disasm.DisassembleBuffer(DisasmBegin, DisasmEnd);
|
||||
}
|
||||
#endif
|
||||
|
||||
if (DebugData) {
|
||||
@@ -1156,7 +1190,9 @@ fextl::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl *
|
||||
|
||||
CPUBackendFeatures GetArm64JITBackendFeatures() {
|
||||
return CPUBackendFeatures {
|
||||
.SupportsStaticRegisterAllocation = true
|
||||
.SupportsStaticRegisterAllocation = true,
|
||||
.SupportsShiftedBitwise = true,
|
||||
.SupportsFlags = true,
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
+11
-5
@@ -70,9 +70,9 @@ private:
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
return SRA64[Reg.Reg];
|
||||
return StaticRegisters[Reg.Reg];
|
||||
} else if (Reg.Class == IR::GPRClass.Val) {
|
||||
return RA64[Reg.Reg];
|
||||
return GeneralRegisters[Reg.Reg];
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
@@ -84,9 +84,9 @@ private:
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
return SRAFPR[Reg.Reg];
|
||||
return StaticFPRegisters[Reg.Reg];
|
||||
} else if (Reg.Class == IR::FPRClass.Val) {
|
||||
return RAFPR[Reg.Reg];
|
||||
return GeneralFPRegisters[Reg.Reg];
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
@@ -97,7 +97,7 @@ private:
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRPairClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
return RA64Pair[Reg.Reg];
|
||||
return GeneralPairRegisters[Reg.Reg];
|
||||
}
|
||||
|
||||
[[nodiscard]] FEXCore::IR::RegisterClassType GetRegClass(IR::NodeID Node) const;
|
||||
@@ -112,6 +112,7 @@ private:
|
||||
|
||||
[[nodiscard]] bool IsFPR(IR::NodeID Node) const;
|
||||
[[nodiscard]] bool IsGPR(IR::NodeID Node) const;
|
||||
[[nodiscard]] bool IsGPRPair(IR::NodeID Node) const;
|
||||
|
||||
[[nodiscard]] FEXCore::ARMEmitter::ExtendedMemOperand GenerateMemOperand(uint8_t AccessSize,
|
||||
FEXCore::ARMEmitter::Register Base,
|
||||
@@ -236,8 +237,10 @@ private:
|
||||
DEF_OP(InlineEntrypointOffset);
|
||||
DEF_OP(CycleCounter);
|
||||
DEF_OP(Add);
|
||||
DEF_OP(TestNZ);
|
||||
DEF_OP(Sub);
|
||||
DEF_OP(Neg);
|
||||
DEF_OP(Abs);
|
||||
DEF_OP(Mul);
|
||||
DEF_OP(UMul);
|
||||
DEF_OP(Div);
|
||||
@@ -247,6 +250,8 @@ private:
|
||||
DEF_OP(MulH);
|
||||
DEF_OP(UMulH);
|
||||
DEF_OP(Or);
|
||||
DEF_OP(Orlshl);
|
||||
DEF_OP(Orlshr);
|
||||
DEF_OP(And);
|
||||
DEF_OP(Andn);
|
||||
DEF_OP(Xor);
|
||||
@@ -452,6 +457,7 @@ private:
|
||||
DEF_OP(VUMull2);
|
||||
DEF_OP(VSMull2);
|
||||
DEF_OP(VUABDL);
|
||||
DEF_OP(VUABDL2);
|
||||
DEF_OP(VTBL1);
|
||||
DEF_OP(VRev64);
|
||||
|
||||
|
||||
+145
-81
@@ -10,6 +10,7 @@ $end_info$
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
@@ -132,7 +133,7 @@ DEF_OP(LoadRegister) {
|
||||
[[maybe_unused]] const auto regId = (Op->Offset / Core::CPUState::GPR_REG_SIZE) - 1;
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRA64.size(), "out of range regId");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
@@ -166,7 +167,7 @@ DEF_OP(LoadRegister) {
|
||||
[[maybe_unused]] const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE, "Unsupported code path!");
|
||||
LOGMAN_THROW_A_FMT(regId < SRAFPR.size(), "out of range regId");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
|
||||
const auto host = GetVReg(Node);
|
||||
|
||||
@@ -214,7 +215,7 @@ DEF_OP(StoreRegister) {
|
||||
[[maybe_unused]] const auto regId = (Op->Offset / Core::CPUState::GPR_REG_SIZE) - 1;
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRA64.size(), "out of range regId");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
@@ -248,7 +249,7 @@ DEF_OP(StoreRegister) {
|
||||
[[maybe_unused]] const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE, "Unsupported code path!");
|
||||
LOGMAN_THROW_A_FMT(regId < SRAFPR.size(), "regId out of range");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "regId out of range");
|
||||
|
||||
const auto host = GetVReg(Op->Value.ID());
|
||||
|
||||
@@ -296,9 +297,9 @@ DEF_OP(LoadRegisterSRA) {
|
||||
const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.gregs[0])) / Core::CPUState::GPR_REG_SIZE;
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRA64.size(), "out of range regId");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
|
||||
const auto reg = SRA64[regId];
|
||||
const auto reg = StaticRegisters[regId];
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
@@ -334,9 +335,9 @@ DEF_OP(LoadRegisterSRA) {
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRAFPR.size(), "out of range regId");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
|
||||
const auto guest = SRAFPR[regId];
|
||||
const auto guest = StaticFPRegisters[regId];
|
||||
const auto host = GetVReg(Node);
|
||||
|
||||
if (HostSupportsSVE) {
|
||||
@@ -484,9 +485,9 @@ DEF_OP(StoreRegisterSRA) {
|
||||
const auto regId = (Op->Offset / Core::CPUState::GPR_REG_SIZE) - 1;
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRA64.size(), "out of range regId");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
|
||||
const auto reg = SRA64[regId];
|
||||
const auto reg = StaticRegisters[regId];
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
switch (OpSize) {
|
||||
@@ -520,9 +521,9 @@ DEF_OP(StoreRegisterSRA) {
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRAFPR.size(), "regId out of range");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "regId out of range");
|
||||
|
||||
const auto guest = SRAFPR[regId];
|
||||
const auto guest = StaticFPRegisters[regId];
|
||||
const auto host = GetVReg(Op->Value.ID());
|
||||
|
||||
if (HostSupportsSVE) {
|
||||
@@ -676,10 +677,7 @@ DEF_OP(LoadContextIndexed) {
|
||||
case 2:
|
||||
case 4:
|
||||
case 8: {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Op->Stride);
|
||||
mul(ARMEmitter::Size::i64Bit, TMP1, Index, TMP1);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, TMP1.R());
|
||||
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, Index, FEXCore::ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Op->Stride));
|
||||
const auto Dst = GetReg(Node);
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
@@ -716,10 +714,7 @@ DEF_OP(LoadContextIndexed) {
|
||||
case 8:
|
||||
case 16:
|
||||
case 32: {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Op->Stride);
|
||||
mul(ARMEmitter::Size::i64Bit, TMP1, Index, TMP1);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, TMP1.R());
|
||||
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, Index, FEXCore::ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Op->Stride));
|
||||
const auto Dst = GetVReg(Node);
|
||||
|
||||
switch (OpSize) {
|
||||
@@ -774,9 +769,7 @@ DEF_OP(StoreContextIndexed) {
|
||||
case 2:
|
||||
case 4:
|
||||
case 8: {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Op->Stride);
|
||||
mul(ARMEmitter::Size::i64Bit, TMP1, Index, TMP1);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, TMP1.R());
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, Index, FEXCore::ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Op->Stride));
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
@@ -815,9 +808,7 @@ DEF_OP(StoreContextIndexed) {
|
||||
case 8:
|
||||
case 16:
|
||||
case 32: {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Op->Stride);
|
||||
mul(ARMEmitter::Size::i64Bit, TMP1, Index, TMP1);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, TMP1.R());
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, Index, FEXCore::ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Op->Stride));
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
@@ -1060,12 +1051,20 @@ DEF_OP(FillRegister) {
|
||||
DEF_OP(LoadFlag) {
|
||||
auto Op = IROp->C<IR::IROp_LoadFlag>();
|
||||
auto Dst = GetReg(Node);
|
||||
ldrb(Dst, STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
|
||||
if (Op->Flag == 24 /* NZCV */)
|
||||
ldr(Dst.W(), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
else
|
||||
ldrb(Dst, STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
}
|
||||
|
||||
DEF_OP(StoreFlag) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
strb(GetReg(Op->Value.ID()), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
|
||||
if (Op->Flag == 24 /* NZCV */)
|
||||
str(GetReg(Op->Value.ID()).W(), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
else
|
||||
strb(GetReg(Op->Value.ID()), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
}
|
||||
|
||||
FEXCore::ARMEmitter::ExtendedMemOperand Arm64JITCore::GenerateMemOperand(uint8_t AccessSize,
|
||||
@@ -1085,9 +1084,9 @@ FEXCore::ARMEmitter::ExtendedMemOperand Arm64JITCore::GenerateMemOperand(uint8_t
|
||||
} else {
|
||||
auto RegOffset = GetReg(Offset.ID());
|
||||
switch(OffsetType.Val) {
|
||||
case IR::MEM_OFFSET_SXTX.Val: return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::SXTX, (int)std::log2(OffsetScale) );
|
||||
case IR::MEM_OFFSET_UXTW.Val: return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::UXTW, (int)std::log2(OffsetScale) );
|
||||
case IR::MEM_OFFSET_SXTW.Val: return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::SXTW, (int)std::log2(OffsetScale) );
|
||||
case IR::MEM_OFFSET_SXTX.Val: return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::SXTX, FEXCore::ilog2(OffsetScale) );
|
||||
case IR::MEM_OFFSET_UXTW.Val: return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::UXTW, FEXCore::ilog2(OffsetScale) );
|
||||
case IR::MEM_OFFSET_SXTW.Val: return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale) );
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled GenerateMemOperand OffsetType: {}", OffsetType.Val); break;
|
||||
}
|
||||
}
|
||||
@@ -1240,8 +1239,6 @@ DEF_OP(LoadMemTSO) {
|
||||
ldapurb(Dst, MemReg, Offset);
|
||||
}
|
||||
else {
|
||||
// Aligned
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
ldapurh(Dst, MemReg, Offset);
|
||||
@@ -1256,19 +1253,17 @@ DEF_OP(LoadMemTSO) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
}
|
||||
else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
if (OpSize == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
ldaprb(Dst.W(), MemReg);
|
||||
}
|
||||
else {
|
||||
// Aligned
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
ldaprh(Dst.W(), MemReg);
|
||||
@@ -1283,6 +1278,7 @@ DEF_OP(LoadMemTSO) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
}
|
||||
@@ -1293,8 +1289,6 @@ DEF_OP(LoadMemTSO) {
|
||||
ldarb(Dst, MemReg);
|
||||
}
|
||||
else {
|
||||
// Aligned
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
ldarh(Dst, MemReg);
|
||||
@@ -1309,11 +1303,11 @@ DEF_OP(LoadMemTSO) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
}
|
||||
else {
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
switch (OpSize) {
|
||||
@@ -1341,7 +1335,8 @@ DEF_OP(LoadMemTSO) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
// Half-barrier.
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISHLD);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1526,6 +1521,7 @@ DEF_OP(StoreMemTSO) {
|
||||
stlurb(Src, MemReg, Offset);
|
||||
}
|
||||
else {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
@@ -1541,7 +1537,6 @@ DEF_OP(StoreMemTSO) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
nop();
|
||||
}
|
||||
}
|
||||
else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
@@ -1552,6 +1547,7 @@ DEF_OP(StoreMemTSO) {
|
||||
stlrb(Src, MemReg);
|
||||
}
|
||||
else {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
@@ -1567,10 +1563,10 @@ DEF_OP(StoreMemTSO) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
nop();
|
||||
}
|
||||
}
|
||||
else {
|
||||
// Half-Barrier.
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
const auto Src = GetVReg(Op->Value.ID());
|
||||
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
@@ -1599,7 +1595,6 @@ DEF_OP(StoreMemTSO) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2023,32 +2018,78 @@ DEF_OP(MemCpy) {
|
||||
// Destination already set to the final pointer.
|
||||
}
|
||||
|
||||
|
||||
|
||||
DEF_OP(ParanoidLoadMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Addr = GetReg(Op->Addr.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
LOGMAN_MSG_A_FMT("ParanoidLoadMemTSO: No offset allowed");
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
(void)IsInlineConstant(Op->Offset, &Offset);
|
||||
}
|
||||
|
||||
if (OpSize == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
const auto Dst = GetReg(Node);
|
||||
ldapurb(Dst, MemReg, Offset);
|
||||
}
|
||||
else {
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
ldapurh(Dst, MemReg, Offset);
|
||||
break;
|
||||
case 4:
|
||||
ldapur(Dst.W(), MemReg, Offset);
|
||||
break;
|
||||
case 8:
|
||||
ldapur(Dst.X(), MemReg, Offset);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
if (OpSize == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
ldaprb(Dst.W(), MemReg);
|
||||
}
|
||||
else {
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
ldaprh(Dst.W(), MemReg);
|
||||
break;
|
||||
case 4:
|
||||
ldapr(Dst.W(), MemReg);
|
||||
break;
|
||||
case 8:
|
||||
ldapr(Dst.X(), MemReg);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
ldarb(Dst, Addr);
|
||||
ldarb(Dst, MemReg);
|
||||
break;
|
||||
case 2:
|
||||
ldarh(Dst, Addr);
|
||||
ldarh(Dst, MemReg);
|
||||
break;
|
||||
case 4:
|
||||
ldar(Dst.W(), Addr);
|
||||
ldar(Dst.W(), MemReg);
|
||||
break;
|
||||
case 8:
|
||||
ldar(Dst.X(), Addr);
|
||||
ldar(Dst.X(), MemReg);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize);
|
||||
@@ -2059,31 +2100,30 @@ DEF_OP(ParanoidLoadMemTSO) {
|
||||
const auto Dst = GetVReg(Node);
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
ldarb(TMP1, Addr);
|
||||
ins(ARMEmitter::SubRegSize::i8Bit, Dst, 0, TMP1);
|
||||
ldarb(TMP1, MemReg);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
|
||||
break;
|
||||
case 2:
|
||||
ldarh(TMP1, Addr);
|
||||
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 0, TMP1);
|
||||
ldarh(TMP1, MemReg);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
|
||||
break;
|
||||
case 4:
|
||||
ldar(TMP1.W(), Addr);
|
||||
ins(ARMEmitter::SubRegSize::i32Bit, Dst, 0, TMP1);
|
||||
ldar(TMP1.W(), MemReg);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
|
||||
break;
|
||||
case 8:
|
||||
ldar(TMP1, Addr);
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, TMP1);
|
||||
ldar(TMP1, MemReg);
|
||||
fmov(ARMEmitter::Size::i64Bit, Dst.D(), TMP1);
|
||||
break;
|
||||
case 16:
|
||||
nop();
|
||||
ldaxp(ARMEmitter::Size::i64Bit, TMP1, TMP2, Addr);
|
||||
ldaxp(ARMEmitter::Size::i64Bit, TMP1, TMP2, MemReg);
|
||||
clrex();
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, TMP1);
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 1, TMP2);
|
||||
break;
|
||||
case 32:
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), PRED_TMP_32B.Zeroing(), Addr);
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), PRED_TMP_32B.Zeroing(), MemReg);
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
break;
|
||||
default:
|
||||
@@ -2097,26 +2137,50 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Addr = GetReg(Op->Addr.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
LOGMAN_MSG_A_FMT("ParanoidStoreMemTSO: No offset allowed");
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
(void)IsInlineConstant(Op->Offset, &Offset);
|
||||
}
|
||||
|
||||
if (OpSize == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
stlurb(Src, MemReg, Offset);
|
||||
}
|
||||
else {
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
stlurh(Src, MemReg, Offset);
|
||||
break;
|
||||
case 4:
|
||||
stlur(Src.W(), MemReg, Offset);
|
||||
break;
|
||||
case 8:
|
||||
stlur(Src.X(), MemReg, Offset);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
stlrb(Src, Addr);
|
||||
stlrb(Src, MemReg);
|
||||
break;
|
||||
case 2:
|
||||
stlrh(Src, Addr);
|
||||
stlrh(Src, MemReg);
|
||||
break;
|
||||
case 4:
|
||||
stlr(Src.W(), Addr);
|
||||
stlr(Src.W(), MemReg);
|
||||
break;
|
||||
case 8:
|
||||
stlr(Src.X(), Addr);
|
||||
stlr(Src.X(), MemReg);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize);
|
||||
@@ -2129,19 +2193,19 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
umov<ARMEmitter::SubRegSize::i8Bit>(TMP1, Src, 0);
|
||||
stlrb(TMP1, Addr);
|
||||
stlrb(TMP1, MemReg);
|
||||
break;
|
||||
case 2:
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(TMP1, Src, 0);
|
||||
stlrh(TMP1, Addr);
|
||||
stlrh(TMP1, MemReg);
|
||||
break;
|
||||
case 4:
|
||||
umov<ARMEmitter::SubRegSize::i32Bit>(TMP1, Src, 0);
|
||||
stlr(TMP1.W(), Addr);
|
||||
stlr(TMP1.W(), MemReg);
|
||||
break;
|
||||
case 8:
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(TMP1, Src, 0);
|
||||
stlr(TMP1, Addr);
|
||||
stlr(TMP1, MemReg);
|
||||
break;
|
||||
case 16: {
|
||||
// Move vector to GPRs
|
||||
@@ -2151,14 +2215,14 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
Bind(&B);
|
||||
|
||||
// ldaxp must not have both the destination registers be the same
|
||||
ldaxp(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::zr, TMP3, Addr); // <- Can hit SIGBUS. Overwritten with DMB
|
||||
stlxp(ARMEmitter::Size::i64Bit, TMP3, TMP1, TMP2, Addr); // <- Can also hit SIGBUS
|
||||
ldaxp(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::zr, TMP3, MemReg); // <- Can hit SIGBUS. Overwritten with DMB
|
||||
stlxp(ARMEmitter::Size::i64Bit, TMP3, TMP1, TMP2, MemReg); // <- Can also hit SIGBUS
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP3, &B); // < Overwritten with DMB
|
||||
break;
|
||||
}
|
||||
case 32: {
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(Src.Z(), PRED_TMP_32B, Addr, 0);
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(Src.Z(), PRED_TMP_32B, MemReg, 0);
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
break;
|
||||
}
|
||||
|
||||
+59
-26
@@ -8,6 +8,8 @@ $end_info$
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
DEF_OP(VectorZero) {
|
||||
@@ -2293,40 +2295,47 @@ DEF_OP(VInsElement) {
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const uint32_t ElementSize = Op->Header.ElementSize;
|
||||
|
||||
const auto DestIdx = Op->DestIdx;
|
||||
const auto SrcIdx = Op->SrcIdx;
|
||||
const uint32_t DestIdx = Op->DestIdx;
|
||||
const uint32_t SrcIdx = Op->SrcIdx;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto SrcVector = GetVReg(Op->SrcVector.ID());
|
||||
auto Reg = GetVReg(Op->DestVector.ID());
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 1 || ElementSize == 2 || ElementSize == 4 || ElementSize == 8 || ElementSize == 16, "Invalid size");
|
||||
const auto SubRegSize =
|
||||
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i128Bit;
|
||||
|
||||
// We're going to use this to create our predicate register literal.
|
||||
// On an SVE 256-bit capable system, the predicate register will be
|
||||
// 32-bit in size. We want to set up only the element corresponding
|
||||
// to the destination index, since we're going to copy over the equivalent
|
||||
// indexed element from the source vector.
|
||||
auto Data = [ElementSize, DestIdx]() -> uint32_t {
|
||||
const auto Data = [ElementSize, DestIdx]() -> uint32_t {
|
||||
const auto Log2ElementSize = FEXCore::ilog2(ElementSize);
|
||||
|
||||
[[maybe_unused]] const auto MaxIndex = (32U >> Log2ElementSize) - 1;
|
||||
LOGMAN_THROW_AA_FMT(DestIdx <= MaxIndex, "DestIdx ({}) out of range. Must be within [0, {}]",
|
||||
DestIdx, MaxIndex);
|
||||
|
||||
const auto ShiftAmount = DestIdx << Log2ElementSize;
|
||||
|
||||
switch (ElementSize) {
|
||||
case 1:
|
||||
LOGMAN_THROW_AA_FMT(DestIdx <= 31, "DestIdx out of range: {}", DestIdx);
|
||||
return 1U << DestIdx;
|
||||
case 2:
|
||||
LOGMAN_THROW_AA_FMT(DestIdx <= 15, "DestIdx out of range: {}", DestIdx);
|
||||
return 1U << (DestIdx * 2);
|
||||
case 4:
|
||||
LOGMAN_THROW_AA_FMT(DestIdx <= 7, "DestIdx out of range: {}", DestIdx);
|
||||
return 1U << (DestIdx * 4);
|
||||
case 8:
|
||||
LOGMAN_THROW_AA_FMT(DestIdx <= 3, "DestIdx out of range: {}", DestIdx);
|
||||
return 1U << (DestIdx * 8);
|
||||
return 1U << ShiftAmount;
|
||||
case 16:
|
||||
LOGMAN_THROW_AA_FMT(DestIdx <= 1, "DestIdx out of range: {}", DestIdx);
|
||||
// Predicates can't be subdivided into the Q format, so we can just set up
|
||||
// the predicate to select the two adjacent doublewords.
|
||||
return 0x101U << (DestIdx * 16);
|
||||
return 0x101U << ShiftAmount;
|
||||
default:
|
||||
FEX_UNREACHABLE;
|
||||
return UINT32_MAX;
|
||||
@@ -2339,13 +2348,6 @@ DEF_OP(VInsElement) {
|
||||
adr(TMP1, &DataLocation);
|
||||
ldr(Predicate, TMP1);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 1 || ElementSize == 2 || ElementSize == 4 || ElementSize == 8 || ElementSize == 16, "Invalid size");
|
||||
const auto SubRegSize =
|
||||
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i128Bit;
|
||||
|
||||
// Broadcast our source value across a temporary,
|
||||
// then combine with the destination.
|
||||
dup(SubRegSize, VTMP2.Z(), SrcVector.Z(), SrcIdx);
|
||||
@@ -2365,11 +2367,6 @@ DEF_OP(VInsElement) {
|
||||
Bind(&PastConstant);
|
||||
}
|
||||
else {
|
||||
if (Dst.Idx() != Reg.Idx()) {
|
||||
mov(VTMP1.Q(), Reg.Q());
|
||||
Reg = VTMP1;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 1 || ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize =
|
||||
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
@@ -2377,6 +2374,11 @@ DEF_OP(VInsElement) {
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (Dst.Idx() != Reg.Idx()) {
|
||||
mov(VTMP1.Q(), Reg.Q());
|
||||
Reg = VTMP1;
|
||||
}
|
||||
|
||||
ins(SubRegSize, Reg.Q(), DestIdx, SrcVector.Q(), SrcIdx);
|
||||
|
||||
if (Dst.Idx() != Reg.Idx()) {
|
||||
@@ -3025,6 +3027,37 @@ DEF_OP(VUABDL) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VUABDL2) {
|
||||
const auto Op = IROp->C<IR::IROp_VUABDL2>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize =
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
// To mimic the behavior of AdvSIMD UABDL, we need to get the
|
||||
// absolute difference of the even elements (UADBLB), get the
|
||||
// absolute difference of the odd elemenets (UABDLT), then
|
||||
// interleave the results in both vectors together.
|
||||
|
||||
uabdlb(SubRegSize, VTMP1.Z(), Vector1.Z(), Vector2.Z());
|
||||
uabdlt(SubRegSize, VTMP2.Z(), Vector1.Z(), Vector2.Z());
|
||||
zip2(SubRegSize, Dst.Z(), VTMP1.Z(), VTMP2.Z());
|
||||
} else {
|
||||
uabdl2(SubRegSize, Dst.Q(), Vector1.Q(), Vector2.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VTBL1) {
|
||||
const auto Op = IROp->C<IR::IROp_VTBL1>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -161,6 +161,30 @@ DEF_OP(Neg) {
|
||||
neg(Dst);
|
||||
}
|
||||
|
||||
DEF_OP(Abs) {
|
||||
auto Op = IROp->C<IR::IROp_Abs>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
Xbyak::Reg Src;
|
||||
Xbyak::Reg Dst;
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
Src = GetSrc<RA_32>(Op->Src.ID());
|
||||
Dst = GetDst<RA_32>(Node);
|
||||
break;
|
||||
case 8:
|
||||
Src = GetSrc<RA_64>(Op->Src.ID());
|
||||
Dst = GetDst<RA_64>(Node);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Abs size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
mov(Dst, Src);
|
||||
mov(TMP1, Src);
|
||||
neg(Dst);
|
||||
cmovs(Dst, TMP1);
|
||||
}
|
||||
|
||||
DEF_OP(Mul) {
|
||||
auto Op = IROp->C<IR::IROp_Mul>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
@@ -1116,7 +1140,28 @@ DEF_OP(Select) {
|
||||
} else {
|
||||
cmp(GRCMP(Op->Cmp1.ID()), GRCMP(Op->Cmp2.ID()));
|
||||
}
|
||||
} else if (IsFPR(Op->Cmp1.ID())) {
|
||||
}
|
||||
else if (IsGPRPair(Op->Cmp1.ID())) {
|
||||
if (Op->CompareSize == 4) {
|
||||
const auto Src1 = GetSrcPair<RA_32>(Op->Cmp1.ID());
|
||||
const auto Src2 = GetSrcPair<RA_32>(Op->Cmp2.ID());
|
||||
mov (TMP1.cvt32(), Src1.first);
|
||||
mov (TMP2.cvt32(), Src1.second);
|
||||
xor_(TMP1.cvt32(), Src2.first);
|
||||
xor_(TMP2.cvt32(), Src2.second);
|
||||
or_(TMP1.cvt32(), TMP2.cvt32());
|
||||
}
|
||||
else {
|
||||
const auto Src1 = GetSrcPair<RA_64>(Op->Cmp1.ID());
|
||||
const auto Src2 = GetSrcPair<RA_64>(Op->Cmp2.ID());
|
||||
mov (TMP1, Src1.first);
|
||||
mov (TMP2, Src1.second);
|
||||
xor_(TMP1, Src2.first);
|
||||
xor_(TMP2, Src2.second);
|
||||
or_(TMP1, TMP2);
|
||||
}
|
||||
}
|
||||
else if (IsFPR(Op->Cmp1.ID())) {
|
||||
if (Op->CompareSize == 4)
|
||||
ucomiss(GetSrc(Op->Cmp1.ID()), GetSrc(Op->Cmp2.ID()));
|
||||
else
|
||||
@@ -1298,6 +1343,7 @@ void X86JITCore::RegisterALUHandlers() {
|
||||
REGISTER_OP(ADD, Add);
|
||||
REGISTER_OP(SUB, Sub);
|
||||
REGISTER_OP(NEG, Neg);
|
||||
REGISTER_OP(ABS, Abs);
|
||||
REGISTER_OP(MUL, Mul);
|
||||
REGISTER_OP(UMUL, UMul);
|
||||
REGISTER_OP(DIV, Div);
|
||||
|
||||
+29
-7
@@ -308,16 +308,13 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushRegs();
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPESTRX>();
|
||||
const auto Is64Bit = Op->GPRSize == 8;
|
||||
const auto Control = Op->Control;
|
||||
|
||||
const auto LHS = GetSrc(Op->LHS.ID());
|
||||
const auto RHS = GetSrc(Op->RHS.ID());
|
||||
const auto SrcRAX = GetSrc<RA_64>(Op->RAX.ID());
|
||||
const auto SrcRDX = GetSrc<RA_64>(Op->RDX.ID());
|
||||
|
||||
// Encode the size check into the 8th bit to save a parameter
|
||||
const auto Control = Op->Control | (uint16_t(Is64Bit) << 8);
|
||||
|
||||
mov(rdi, SrcRAX);
|
||||
mov(rsi, SrcRDX);
|
||||
|
||||
@@ -387,7 +384,7 @@ static uint64_t X86JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame,
|
||||
}
|
||||
|
||||
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
|
||||
Context::ContextImpl::ThreadAddBlockLink(Thread, GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
// undo the link
|
||||
record[0] = LinkerAddress;
|
||||
});
|
||||
@@ -487,11 +484,15 @@ IR::PhysicalRegister X86JITCore::GetPhys(IR::NodeID Node) const {
|
||||
}
|
||||
|
||||
bool X86JITCore::IsFPR(IR::NodeID Node) const {
|
||||
return RAData->GetNodeRegister(Node).Class == IR::FPRClass.Val;
|
||||
return RAData->GetNodeRegister(Node).Class == IR::FPRClass;
|
||||
}
|
||||
|
||||
bool X86JITCore::IsGPR(IR::NodeID Node) const {
|
||||
return RAData->GetNodeRegister(Node).Class == IR::GPRClass.Val;
|
||||
return RAData->GetNodeRegister(Node).Class == IR::GPRClass;
|
||||
}
|
||||
|
||||
bool X86JITCore::IsGPRPair(IR::NodeID Node) const {
|
||||
return RAData->GetNodeRegister(Node).Class == IR::GPRPairClass;
|
||||
}
|
||||
|
||||
template<uint8_t RAType>
|
||||
@@ -820,12 +821,33 @@ CPUBackend::CompiledCode X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]
|
||||
auto JITBlockTail = getCurr<JITCodeTail*>();
|
||||
setSize(getSize() + sizeof(JITCodeTail));
|
||||
|
||||
auto JITRIPEntriesLocation = getCurr<uint8_t *>();
|
||||
auto JITRIPEntries = getCurr<JITRIPReconstructEntries*>();
|
||||
|
||||
setSize(getSize() + sizeof(JITRIPReconstructEntries) * DebugData->GuestOpcodes.size());
|
||||
|
||||
// Put the block's RIP entry in the tail.
|
||||
// This will be used for RIP reconstruction in the future.
|
||||
// TODO: This needs to be a data RIP relocation once code caching works.
|
||||
// Current relocation code doesn't support this feature yet.
|
||||
JITBlockTail->RIP = Entry;
|
||||
|
||||
{
|
||||
// Store the RIP entries.
|
||||
JITBlockTail->NumberOfRIPEntries = DebugData->GuestOpcodes.size();
|
||||
JITBlockTail->OffsetToRIPEntries = JITRIPEntriesLocation - JITBlockTailLocation;
|
||||
uintptr_t CurrentRIPOffset = 0;
|
||||
uint64_t CurrentPCOffset = 0;
|
||||
for (size_t i = 0; i < DebugData->GuestOpcodes.size(); i++) {
|
||||
const auto &GuestOpcode = DebugData->GuestOpcodes[i];
|
||||
auto &RIPEntry = JITRIPEntries[i];
|
||||
RIPEntry.HostPCOffset = GuestOpcode.HostEntryOffset - CurrentPCOffset;
|
||||
RIPEntry.GuestRIPOffset = GuestOpcode.GuestEntryOffset - CurrentRIPOffset;
|
||||
CurrentPCOffset = GuestOpcode.HostEntryOffset;
|
||||
CurrentRIPOffset = GuestOpcode.GuestEntryOffset;
|
||||
}
|
||||
}
|
||||
|
||||
CodeHeader->OffsetToBlockTail = JITBlockTailLocation - CodeData.BlockBegin;
|
||||
|
||||
CodeData.Size = getCurr<uint8_t*>() - CodeData.BlockBegin;
|
||||
|
||||
@@ -169,6 +169,7 @@ private:
|
||||
|
||||
[[nodiscard]] bool IsFPR(IR::NodeID Node) const;
|
||||
[[nodiscard]] bool IsGPR(IR::NodeID Node) const;
|
||||
[[nodiscard]] bool IsGPRPair(IR::NodeID Node) const;
|
||||
|
||||
template<uint8_t RAType>
|
||||
[[nodiscard]] Xbyak::Reg GetSrc(IR::NodeID Node) const;
|
||||
@@ -245,6 +246,7 @@ private:
|
||||
DEF_OP(Add);
|
||||
DEF_OP(Sub);
|
||||
DEF_OP(Neg);
|
||||
DEF_OP(Abs);
|
||||
DEF_OP(Mul);
|
||||
DEF_OP(UMul);
|
||||
DEF_OP(Div);
|
||||
@@ -453,6 +455,7 @@ private:
|
||||
DEF_OP(VUMull2);
|
||||
DEF_OP(VSMull2);
|
||||
DEF_OP(VUABDL);
|
||||
DEF_OP(VUABDL2);
|
||||
DEF_OP(VTBL1);
|
||||
DEF_OP(VRev64);
|
||||
|
||||
|
||||
@@ -601,14 +601,22 @@ DEF_OP(LoadFlag) {
|
||||
auto Op = IROp->C<IR::IROp_LoadFlag>();
|
||||
|
||||
auto Dst = GetDst<RA_64>(Node);
|
||||
movzx(Dst, byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)]);
|
||||
|
||||
if (Op->Flag == 24 /* NZCV */)
|
||||
mov(Dst.cvt32(), dword [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)]);
|
||||
else
|
||||
movzx(Dst, byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)]);
|
||||
}
|
||||
|
||||
DEF_OP(StoreFlag) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
|
||||
mov (rax, GetSrc<RA_64>(Op->Value.ID()));
|
||||
mov(byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)], al);
|
||||
|
||||
if (Op->Flag == 24 /* NZCV */)
|
||||
mov(dword [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)], eax);
|
||||
else
|
||||
mov(byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)], al);
|
||||
}
|
||||
|
||||
Xbyak::RegExp X86JITCore::GenerateModRM(Xbyak::Reg Base, IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale) const {
|
||||
|
||||
@@ -4367,6 +4367,93 @@ DEF_OP(VUABDL) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VUABDL2) {
|
||||
const auto Op = IROp->C<IR::IROp_VUABDL2>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector1 = GetSrc(Op->Vector1.ID());
|
||||
const auto Vector2 = GetSrc(Op->Vector2.ID());
|
||||
|
||||
switch (ElementSize) {
|
||||
case 2: {
|
||||
vpxor(xmm15, xmm15, xmm15);
|
||||
vpxor(xmm14, xmm14, xmm14);
|
||||
|
||||
if (Is256Bit) {
|
||||
vextracti128(xmm15, ToYMM(Vector1), 1);
|
||||
vextracti128(xmm14, ToYMM(Vector2), 1);
|
||||
|
||||
vpxor(xmm12, xmm12, xmm12);
|
||||
vpxor(xmm13, xmm13, xmm13);
|
||||
vpxor(Dst, Dst, Dst);
|
||||
|
||||
// Bottom half
|
||||
vpunpcklbw(xmm13, xmm15, xmm13);
|
||||
vpunpcklbw(xmm12, xmm14, xmm12);
|
||||
|
||||
// Top half
|
||||
vpunpckhbw(xmm15, xmm15, Dst);
|
||||
vpunpckhbw(xmm14, xmm14, Dst);
|
||||
|
||||
// Reinsert
|
||||
vinserti128(ymm13, ymm13, xmm15, 1);
|
||||
vinserti128(ymm12, ymm12, xmm14, 1);
|
||||
|
||||
vpsubw(ToYMM(Dst), ymm12, ymm13);
|
||||
vpabsw(ToYMM(Dst), ToYMM(Dst));
|
||||
} else {
|
||||
vpunpckhbw(xmm15, Vector1, xmm15);
|
||||
vpunpckhbw(xmm14, Vector2, xmm14);
|
||||
vpsubw(Dst, xmm14, xmm15);
|
||||
vpabsw(Dst, Dst);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
vpxor(xmm15, xmm15, xmm15);
|
||||
vpxor(xmm14, xmm14, xmm14);
|
||||
|
||||
if (Is256Bit) {
|
||||
vextracti128(xmm15, ToYMM(Vector1), 1);
|
||||
vextracti128(xmm14, ToYMM(Vector2), 1);
|
||||
|
||||
vpxor(xmm12, xmm12, xmm12);
|
||||
vpxor(xmm13, xmm13, xmm13);
|
||||
vpxor(Dst, Dst, Dst);
|
||||
|
||||
// Bottom half
|
||||
vpunpcklwd(xmm13, xmm15, xmm13);
|
||||
vpunpcklwd(xmm12, xmm14, xmm12);
|
||||
|
||||
// Top half
|
||||
vpunpckhwd(xmm15, xmm15, Dst);
|
||||
vpunpckhwd(xmm14, xmm14, Dst);
|
||||
|
||||
// Reinsert
|
||||
vinserti128(ymm13, ymm13, xmm15, 1);
|
||||
vinserti128(ymm12, ymm12, xmm14, 1);
|
||||
|
||||
vpsubd(ToYMM(Dst), ymm12, ymm13);
|
||||
vpabsd(ToYMM(Dst), ToYMM(Dst));
|
||||
} else {
|
||||
vpunpckhwd(xmm15, Vector1, xmm15);
|
||||
vpunpckhwd(xmm14, Vector2, xmm14);
|
||||
vpsubd(Dst, xmm14, xmm15);
|
||||
vpabsd(Dst, Dst);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
DEF_OP(VTBL1) {
|
||||
const auto Op = IROp->C<IR::IROp_VTBL1>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -4609,6 +4696,7 @@ void X86JITCore::RegisterVectorHandlers() {
|
||||
REGISTER_OP(VUMULL2, VUMull2);
|
||||
REGISTER_OP(VSMULL2, VSMull2);
|
||||
REGISTER_OP(VUABDL, VUABDL);
|
||||
REGISTER_OP(VUABDL2, VUABDL2);
|
||||
REGISTER_OP(VTBL1, VTBL1);
|
||||
REGISTER_OP(VREV64, VRev64);
|
||||
#undef REGISTER_OP
|
||||
|
||||
+204
-146
@@ -145,6 +145,7 @@ void OpDispatchBuilder::SyscallOp(OpcodeArgs) {
|
||||
if (Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_BLOCK_END) {
|
||||
// RIP could have been updated after coming back from the Syscall.
|
||||
NewRIP = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, rip));
|
||||
CalculateDeferredFlags();
|
||||
_ExitFunction(NewRIP);
|
||||
}
|
||||
}
|
||||
@@ -178,6 +179,7 @@ void OpDispatchBuilder::ThunkOp(OpcodeArgs) {
|
||||
|
||||
// Store the new stack pointer
|
||||
StoreGPRRegister(X86State::REG_RSP, NewSP);
|
||||
CalculateDeferredFlags();
|
||||
|
||||
// Store the new RIP
|
||||
_ExitFunction(NewRIP);
|
||||
@@ -240,6 +242,7 @@ void OpDispatchBuilder::RETOp(OpcodeArgs) {
|
||||
|
||||
// Store the new stack pointer
|
||||
StoreGPRRegister(X86State::REG_RSP, NewSP);
|
||||
CalculateDeferredFlags();
|
||||
|
||||
// Store the new RIP
|
||||
_ExitFunction(NewRIP);
|
||||
@@ -304,6 +307,7 @@ void OpDispatchBuilder::IRETOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RSP, SP);
|
||||
}
|
||||
|
||||
CalculateDeferredFlags();
|
||||
_ExitFunction(NewRIP);
|
||||
BlockSetRIP = true;
|
||||
}
|
||||
@@ -313,6 +317,7 @@ void OpDispatchBuilder::CallbackReturnOp(OpcodeArgs) {
|
||||
// Store the new RIP
|
||||
_CallbackReturn();
|
||||
auto NewRIP = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, rip));
|
||||
CalculateDeferredFlags();
|
||||
// This ExitFunction won't actually get hit but needs to exist
|
||||
_ExitFunction(NewRIP);
|
||||
BlockSetRIP = true;
|
||||
@@ -669,7 +674,6 @@ void OpDispatchBuilder::POPOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::POPAOp(OpcodeArgs) {
|
||||
// 32bit only
|
||||
const uint8_t Size = GetSrcSize(Op);
|
||||
const uint8_t GPRSize = 4;
|
||||
|
||||
auto Constant = _Constant(Size);
|
||||
auto OldSP = LoadGPRRegister(X86State::REG_RSP);
|
||||
@@ -687,32 +691,32 @@ void OpDispatchBuilder::POPAOp(OpcodeArgs) {
|
||||
OrderedNode *Src{};
|
||||
OrderedNode *NewSP = OldSP;
|
||||
Src = _LoadMem(GPRClass, Size, NewSP, Size);
|
||||
StoreGPRRegister(X86State::REG_RDI, Src, GPRSize);
|
||||
StoreGPRRegister(X86State::REG_RDI, Src, Size);
|
||||
NewSP = _Add(NewSP, Constant);
|
||||
|
||||
Src = _LoadMem(GPRClass, Size, NewSP, Size);
|
||||
StoreGPRRegister(X86State::REG_RSI, Src, GPRSize);
|
||||
StoreGPRRegister(X86State::REG_RSI, Src, Size);
|
||||
NewSP = _Add(NewSP, Constant);
|
||||
|
||||
Src = _LoadMem(GPRClass, Size, NewSP, Size);
|
||||
StoreGPRRegister(X86State::REG_RBP, Src, GPRSize);
|
||||
StoreGPRRegister(X86State::REG_RBP, Src, Size);
|
||||
NewSP = _Add(NewSP, _Constant(Size * 2));
|
||||
|
||||
// Skip SP loading
|
||||
Src = _LoadMem(GPRClass, Size, NewSP, Size);
|
||||
StoreGPRRegister(X86State::REG_RBX, Src, GPRSize);
|
||||
StoreGPRRegister(X86State::REG_RBX, Src, Size);
|
||||
NewSP = _Add(NewSP, Constant);
|
||||
|
||||
Src = _LoadMem(GPRClass, Size, NewSP, Size);
|
||||
StoreGPRRegister(X86State::REG_RDX, Src, GPRSize);
|
||||
StoreGPRRegister(X86State::REG_RDX, Src, Size);
|
||||
NewSP = _Add(NewSP, Constant);
|
||||
|
||||
Src = _LoadMem(GPRClass, Size, NewSP, Size);
|
||||
StoreGPRRegister(X86State::REG_RCX, Src, GPRSize);
|
||||
StoreGPRRegister(X86State::REG_RCX, Src, Size);
|
||||
NewSP = _Add(NewSP, Constant);
|
||||
|
||||
Src = _LoadMem(GPRClass, Size, NewSP, Size);
|
||||
StoreGPRRegister(X86State::REG_RAX, Src, GPRSize);
|
||||
StoreGPRRegister(X86State::REG_RAX, Src, Size);
|
||||
NewSP = _Add(NewSP, Constant);
|
||||
|
||||
// Store the new stack pointer
|
||||
@@ -810,6 +814,7 @@ void OpDispatchBuilder::CALLOp(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[0].IsLiteral(), "Had wrong operand type");
|
||||
const uint64_t TargetRIP = Op->PC + Op->InstSize + Op->Src[0].Data.Literal.Value;
|
||||
|
||||
CalculateDeferredFlags();
|
||||
if (NextRIP != TargetRIP) {
|
||||
// Store the RIP
|
||||
_ExitFunction(NewRIP); // If we get here then leave the function now
|
||||
@@ -840,6 +845,7 @@ void OpDispatchBuilder::CALLAbsoluteOp(OpcodeArgs) {
|
||||
_StoreMem(GPRClass, Size, NewSP, ConstantPCReturn, Size);
|
||||
|
||||
// Store the RIP
|
||||
CalculateDeferredFlags();
|
||||
_ExitFunction(JMPPCOffset); // If we get here then leave the function now
|
||||
}
|
||||
|
||||
@@ -915,15 +921,11 @@ OrderedNode *OpDispatchBuilder::SelectCC(uint8_t OP, OrderedNode *TrueValue, Ord
|
||||
break;
|
||||
}
|
||||
case 0xA: { // JP - Jump if PF == 1
|
||||
auto Flag = GetRFLAG(FEXCore::X86State::RFLAG_PF_LOC);
|
||||
SrcCond = _Select(FEXCore::IR::COND_NEQ,
|
||||
Flag, ZeroConst, TrueValue, FalseValue);
|
||||
SrcCond = _Select(FEXCore::IR::COND_NEQ, LoadPF(), ZeroConst, TrueValue, FalseValue);
|
||||
break;
|
||||
}
|
||||
case 0xB: { // JNP - Jump if PF == 0
|
||||
auto Flag = GetRFLAG(FEXCore::X86State::RFLAG_PF_LOC);
|
||||
SrcCond = _Select(FEXCore::IR::COND_EQ,
|
||||
Flag, ZeroConst, TrueValue, FalseValue);
|
||||
SrcCond = _Select(FEXCore::IR::COND_EQ, LoadPF(), ZeroConst, TrueValue, FalseValue);
|
||||
break;
|
||||
}
|
||||
case 0xC: { // SF <> OF
|
||||
@@ -1128,6 +1130,7 @@ void OpDispatchBuilder::CondJUMPOp(OpcodeArgs) {
|
||||
Target &= 0xFFFFFFFFU;
|
||||
}
|
||||
|
||||
CalculateDeferredFlags();
|
||||
auto TrueBlock = JumpTargets.find(Target);
|
||||
auto FalseBlock = JumpTargets.find(Op->PC + Op->InstSize);
|
||||
|
||||
@@ -1273,6 +1276,7 @@ void OpDispatchBuilder::LoopOp(OpcodeArgs) {
|
||||
SrcCond = _And(SrcCond, ZF);
|
||||
}
|
||||
|
||||
CalculateDeferredFlags();
|
||||
auto TrueBlock = JumpTargets.find(Target);
|
||||
auto FalseBlock = JumpTargets.find(Op->PC + Op->InstSize);
|
||||
|
||||
@@ -1341,6 +1345,7 @@ void OpDispatchBuilder::JUMPOp(OpcodeArgs) {
|
||||
TargetRIP &= 0xFFFFFFFFU;
|
||||
}
|
||||
|
||||
CalculateDeferredFlags();
|
||||
// This is just an unconditional relative literal jump
|
||||
if (Multiblock) {
|
||||
auto JumpBlock = JumpTargets.find(TargetRIP);
|
||||
@@ -1379,6 +1384,7 @@ void OpDispatchBuilder::JUMPAbsoluteOp(OpcodeArgs) {
|
||||
// This uses ModRM to determine its location
|
||||
// No way to use this effectively in multiblock
|
||||
auto RIPOffset = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
CalculateDeferredFlags();
|
||||
|
||||
// Store the new RIP
|
||||
_ExitFunction(RIPOffset);
|
||||
@@ -1554,7 +1560,7 @@ void OpDispatchBuilder::SAHFOp(OpcodeArgs) {
|
||||
}
|
||||
void OpDispatchBuilder::LAHFOp(OpcodeArgs) {
|
||||
// Load the lower 8 bits of the Rflags register
|
||||
auto RFLAG = GetPackedRFLAG(true);
|
||||
auto RFLAG = GetPackedRFLAG(0xFF);
|
||||
|
||||
// Store the lower 8 bits of the rflags register in to AH
|
||||
StoreGPRRegister(X86State::REG_RAX, RFLAG, 1, 8);
|
||||
@@ -1804,15 +1810,7 @@ void OpDispatchBuilder::SHLOp(OpcodeArgs) {
|
||||
}
|
||||
const auto Size = GetSrcBitSize(Op);
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op
|
||||
if (Size == 64) {
|
||||
Src = _And(Src, _Constant(0x3F));
|
||||
}
|
||||
else {
|
||||
Src = _And(Src, _Constant(0x1F));
|
||||
}
|
||||
|
||||
OrderedNode *Result = _Lshl(Dest, Src);
|
||||
OrderedNode *Result = _Lshl(std::max<uint8_t>(4, GetSrcSize(Op)), Dest, Src);
|
||||
StoreResult(GPRClass, Op, Result, -1);
|
||||
|
||||
if (Size < 32) {
|
||||
@@ -1866,17 +1864,7 @@ void OpDispatchBuilder::SHROp(OpcodeArgs) {
|
||||
Src = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
}
|
||||
|
||||
const auto Size = GetSrcBitSize(Op);
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op
|
||||
if (Size == 64) {
|
||||
Src = _And(Src, _Constant(0x3F));
|
||||
}
|
||||
else {
|
||||
Src = _And(Src, _Constant(0x1F));
|
||||
}
|
||||
|
||||
auto ALUOp = _Lshr(Dest, Src);
|
||||
auto ALUOp = _Lshr(std::max<uint8_t>(4, GetSrcSize(Op)), Dest, Src);
|
||||
StoreResult(GPRClass, Op, ALUOp, -1);
|
||||
|
||||
if constexpr (SHR1Bit) {
|
||||
@@ -1985,20 +1973,23 @@ void OpDispatchBuilder::SHLDImmediateOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
if (Shift != 0) {
|
||||
OrderedNode *ShiftLeft = _Constant(Shift);
|
||||
auto ShiftRight = _Constant(Size - Shift);
|
||||
OrderedNode *Res{};
|
||||
if (Size < 32) {
|
||||
OrderedNode *ShiftLeft = _Constant(Shift);
|
||||
auto ShiftRight = _Constant(Size - Shift);
|
||||
|
||||
auto Tmp1 = _Lshl(Dest, ShiftLeft);
|
||||
Tmp1.first->Header.Size = 8;
|
||||
auto Tmp2 = _Lshr(Src, ShiftRight);
|
||||
auto Tmp1 = _Lshl(Dest, ShiftLeft);
|
||||
Tmp1.first->Header.Size = 8;
|
||||
auto Tmp2 = _Lshr(Src, ShiftRight);
|
||||
|
||||
OrderedNode *Res = _Or(Tmp1, Tmp2);
|
||||
Res = _Or(Tmp1, Tmp2);
|
||||
}
|
||||
else {
|
||||
// 32-bit and 64-bit SHLD behaves like an EXTR where the lower bits are filled from the source.
|
||||
Res = _Extr(Dest, Src, Size - Shift);
|
||||
}
|
||||
|
||||
StoreResult(GPRClass, Op, Res, -1);
|
||||
|
||||
if (Size != 64) {
|
||||
Res = _Bfe(Size, 0, Res);
|
||||
}
|
||||
GenerateFlags_ShiftLeftImmediate(Op, Res, Dest, Shift);
|
||||
}
|
||||
else if (Shift == 0 && Size == 32) {
|
||||
@@ -2082,21 +2073,25 @@ void OpDispatchBuilder::SHRDImmediateOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
if (Shift != 0) {
|
||||
OrderedNode *ShiftRight = _Constant(Shift);
|
||||
auto ShiftLeft = _Constant(Size - Shift);
|
||||
|
||||
auto Tmp1 = _Lshr(Dest, ShiftRight);
|
||||
auto Tmp2 = _Lshl(Src, ShiftLeft);
|
||||
Tmp2.first->Header.Size = 8;
|
||||
OrderedNode *Res{};
|
||||
if (Size < 32) {
|
||||
OrderedNode *ShiftRight = _Constant(Shift);
|
||||
auto ShiftLeft = _Constant(Size - Shift);
|
||||
|
||||
OrderedNode *Res = _Or(Tmp1, Tmp2);
|
||||
auto Tmp1 = _Lshr(Dest, ShiftRight);
|
||||
auto Tmp2 = _Lshl(Src, ShiftLeft);
|
||||
Tmp2.first->Header.Size = 8;
|
||||
|
||||
Res = _Or(Tmp1, Tmp2);
|
||||
}
|
||||
else {
|
||||
// 32-bit and 64-bit SHRD behaves like an EXTR where the upper bits are filled from the source.
|
||||
Res = _Extr(Src, Dest, Shift);
|
||||
}
|
||||
|
||||
StoreResult(GPRClass, Op, Res, -1);
|
||||
|
||||
if (Size != 64) {
|
||||
Res = _Bfe(Size, 0, Res);
|
||||
}
|
||||
GenerateFlags_ShiftRightImmediate(Op, Res, Dest, Shift);
|
||||
GenerateFlags_ShiftRightDoubleImmediate(Op, Res, Dest, Shift);
|
||||
}
|
||||
else if (Shift == 0 && Size == 32) {
|
||||
// Ensure Zext still occurs
|
||||
@@ -2117,18 +2112,11 @@ void OpDispatchBuilder::ASHROp(OpcodeArgs) {
|
||||
Src = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
}
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op
|
||||
if (Size == 64) {
|
||||
Src = _And(Src, _Constant(Size, 0x3F));
|
||||
} else {
|
||||
Src = _And(Src, _Constant(Size, 0x1F));
|
||||
}
|
||||
|
||||
if (Size < 32) {
|
||||
Dest = _Sbfe(Size, 0, Dest);
|
||||
}
|
||||
|
||||
OrderedNode *Result = _Ashr(Dest, Src);
|
||||
OrderedNode *Result = _Ashr(std::max<uint8_t>(4, GetSrcSize(Op)), Dest, Src);
|
||||
StoreResult(GPRClass, Op, Result, -1);
|
||||
|
||||
if constexpr (SHR1Bit) {
|
||||
@@ -2412,29 +2400,20 @@ void OpDispatchBuilder::BMI2Shift(OpcodeArgs) {
|
||||
|
||||
auto* Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
auto* Shift = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
const auto OperandSize = GetSrcBitSize(Op);
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op
|
||||
auto SanitizedShift = [&] {
|
||||
if (OperandSize == 64) {
|
||||
return _And(Shift, _Constant(0x3F));
|
||||
} else {
|
||||
return _And(Shift, _Constant(0x1F));
|
||||
}
|
||||
}();
|
||||
const auto Size = GetSrcSize(Op);
|
||||
|
||||
auto* Result = [&]() -> OrderedNode* {
|
||||
// SARX
|
||||
if (Op->OP == 0x6F7) {
|
||||
return _Ashr(Src, SanitizedShift);
|
||||
return _Ashr(Size, Src, Shift);
|
||||
}
|
||||
// SHLX
|
||||
if (Op->OP == 0x5F7) {
|
||||
return _Lshl(Src, SanitizedShift);
|
||||
return _Lshl(Size, Src, Shift);
|
||||
}
|
||||
|
||||
// SHRX
|
||||
return _Lshr(Src, SanitizedShift);
|
||||
return _Lshr(Size, Src, Shift);
|
||||
}();
|
||||
|
||||
StoreResult(GPRClass, Op, Result, -1);
|
||||
@@ -2631,19 +2610,12 @@ void OpDispatchBuilder::RCROp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC);
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op
|
||||
if (Size == 64) {
|
||||
Src = _And(Src, _Constant(Size, 0x3F));
|
||||
} else {
|
||||
Src = _And(Src, _Constant(Size, 0x1F));
|
||||
}
|
||||
|
||||
// Res = Src >> Shift
|
||||
OrderedNode *Res = _Lshr(Dest, Src);
|
||||
|
||||
// Res |= (Src << (Size - Shift + 1));
|
||||
OrderedNode *SrcShl = _Sub(_Constant(Size, Size + 1), Src);
|
||||
auto TmpHigher = _Lshl(Dest, SrcShl);
|
||||
auto TmpHigher = _Lshl(GetSrcSize(Op), Dest, SrcShl);
|
||||
|
||||
auto One = _Constant(Size, 1);
|
||||
auto Zero = _Constant(Size, 0);
|
||||
@@ -2669,7 +2641,7 @@ void OpDispatchBuilder::RCROp(OpcodeArgs) {
|
||||
|
||||
// CF only changes if we actually shifted
|
||||
// Our new CF will be bit (Shift - 1) of the source
|
||||
auto NewCF = _Lshr(Dest, _Sub(Src, One));
|
||||
auto NewCF = _Bfe(1, 0, _Lshr(Dest, _Sub(Src, One)));
|
||||
CompareResult = _Select(FEXCore::IR::COND_UGE,
|
||||
Src, One,
|
||||
NewCF, CF);
|
||||
@@ -2701,17 +2673,62 @@ void OpDispatchBuilder::RCRSmallerOp(OpcodeArgs) {
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op
|
||||
Src = _And(Src, _Constant(Size, 0x1F));
|
||||
|
||||
OrderedNode *Tmp = _Constant(64, 0);
|
||||
OrderedNode *Tmp{};
|
||||
|
||||
// Insert the incoming value across the temporary 64bit source
|
||||
// Make sure to insert at <BitSize> + 1 offsets
|
||||
// We need to cover 32bits plus the amount that could rotate in
|
||||
for (size_t i = 0; i < (32 + Size + 1); i += (Size + 1)) {
|
||||
// Insert incoming value
|
||||
Tmp = _Bfi(8, Size, i, Tmp, Dest);
|
||||
|
||||
// Insert CF
|
||||
Tmp = _Bfi(8, 1, i + Size, Tmp, CF);
|
||||
if (Size == 8) {
|
||||
// 8-bit optimal cascade
|
||||
// Cascade: 0
|
||||
// Data: -> [7:0]
|
||||
// CF: -> [8:8]
|
||||
// Cascade: 1
|
||||
// Data: -> [16:9]
|
||||
// CF: -> [17:17]
|
||||
// Cascade: 2
|
||||
// Data: -> [25:18]
|
||||
// CF: -> [26:26]
|
||||
// Cascade: 3
|
||||
// Data: -> [34:27]
|
||||
// CF: -> [35:35]
|
||||
// Cascade: 4
|
||||
// Data: -> [43:36]
|
||||
// CF: -> [44:44]
|
||||
|
||||
// Insert CF, Destination already at [7:0]
|
||||
Tmp = _Bfi(8, 1, 8, Dest, CF);
|
||||
|
||||
// First Cascade, copies 9 bits from itself.
|
||||
Tmp = _Bfi(8, 9, 9, Tmp, Tmp);
|
||||
|
||||
// Second cascade, copies 18 bits from itself.
|
||||
Tmp = _Bfi(8, 18, 18, Tmp, Tmp);
|
||||
|
||||
// Final cascade, copies 9 bits again from itself.
|
||||
Tmp = _Bfi(8, 9, 36, Tmp, Tmp);
|
||||
}
|
||||
else {
|
||||
// 16-bit optimal cascade
|
||||
// Cascade: 0
|
||||
// Data: -> [15:0]
|
||||
// CF: -> [16:16]
|
||||
// Cascade: 1
|
||||
// Data: -> [32:17]
|
||||
// CF: -> [33:33]
|
||||
// Cascade: 2
|
||||
// Data: -> [49:34]
|
||||
// CF: -> [50:50]
|
||||
|
||||
// Insert CF, Destination already at [15:0]
|
||||
Tmp = _Bfi(8, 1, 16, Dest, CF);
|
||||
|
||||
// First Cascade, copies 17 bits from itself.
|
||||
Tmp = _Bfi(8, 17, 17, Tmp, Tmp);
|
||||
|
||||
// Final Cascade, copies 17 bits from itself again.
|
||||
Tmp = _Bfi(8, 17, 34, Tmp, Tmp);
|
||||
}
|
||||
|
||||
// Entire bitfield has been setup
|
||||
@@ -2723,7 +2740,7 @@ void OpDispatchBuilder::RCRSmallerOp(OpcodeArgs) {
|
||||
// CF only changes if we actually shifted
|
||||
// Our new CF will be bit (Shift - 1) of the source
|
||||
auto One = _Constant(Size, 1);
|
||||
auto NewCF = _Lshr(Tmp, _Sub(Src, One));
|
||||
auto NewCF = _Bfe(1, 0, _Lshr(Tmp, _Sub(Src, One)));
|
||||
auto CompareResult = _Select(FEXCore::IR::COND_UGE,
|
||||
Src, One,
|
||||
NewCF, CF);
|
||||
@@ -2780,15 +2797,8 @@ void OpDispatchBuilder::RCLOp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC);
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op
|
||||
if (Size == 64) {
|
||||
Src = _And(Src, _Constant(Size, 0x3F));
|
||||
} else {
|
||||
Src = _And(Src, _Constant(Size, 0x1F));
|
||||
}
|
||||
|
||||
// Res = Src << Shift
|
||||
OrderedNode *Res = _Lshl(Dest, Src);
|
||||
OrderedNode *Res = _Lshl(GetSrcSize(Op), Dest, Src);
|
||||
|
||||
// Res |= (Src << (Size - Shift + 1));
|
||||
OrderedNode *SrcShl = _Sub(_Constant(Size, Size + 1), Src);
|
||||
@@ -2819,7 +2829,7 @@ void OpDispatchBuilder::RCLOp(OpcodeArgs) {
|
||||
{
|
||||
// CF only changes if we actually shifted
|
||||
// Our new CF will be bit (Shift - 1) of the source
|
||||
auto NewCF = _Lshr(Dest, _Sub(_Constant(Size, Size), Src));
|
||||
auto NewCF = _Bfe(1, 0, _Lshr(Dest, _Sub(_Constant(Size, Size), Src)));
|
||||
CompareResult = _Select(FEXCore::IR::COND_UGE,
|
||||
Src, One,
|
||||
NewCF, CF);
|
||||
@@ -2878,7 +2888,7 @@ void OpDispatchBuilder::RCLSmallerOp(OpcodeArgs) {
|
||||
// Our new CF is now at the bit position that we are shifting
|
||||
// Either 0 if CF hasn't changed (CF is living in bit 0)
|
||||
// or higher
|
||||
auto NewCF = _Ror(Tmp, _Sub(_Constant(63), Src));
|
||||
auto NewCF = _Bfe(1, 0, _Ror(Tmp, _Sub(_Constant(63), Src)));
|
||||
auto CompareResult = _Select(FEXCore::IR::COND_UGE,
|
||||
Src, _Constant(1),
|
||||
NewCF, CF);
|
||||
@@ -2955,7 +2965,7 @@ void OpDispatchBuilder::BTOp(OpcodeArgs) {
|
||||
// Now shift in to the correct bit location
|
||||
Result = _Lshr(Result, BitSelect);
|
||||
}
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(Result);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(1, 0, Result));
|
||||
}
|
||||
|
||||
template<uint32_t SrcIndex>
|
||||
@@ -3033,7 +3043,7 @@ void OpDispatchBuilder::BTROp(OpcodeArgs) {
|
||||
_StoreMemAutoTSO(GPRClass, 1, MemoryLocation, Value, 1);
|
||||
}
|
||||
}
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(Result);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(1, 0, Result));
|
||||
}
|
||||
|
||||
template<uint32_t SrcIndex>
|
||||
@@ -3107,7 +3117,7 @@ void OpDispatchBuilder::BTSOp(OpcodeArgs) {
|
||||
_StoreMemAutoTSO(GPRClass, 1, MemoryLocation, Value, 1);
|
||||
}
|
||||
}
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(Result);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(1, 0, Result));
|
||||
}
|
||||
|
||||
template<uint32_t SrcIndex>
|
||||
@@ -3181,7 +3191,7 @@ void OpDispatchBuilder::BTCOp(OpcodeArgs) {
|
||||
_StoreMemAutoTSO(GPRClass, 1, MemoryLocation, Value, 1);
|
||||
}
|
||||
}
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(Result);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(1, 0, Result));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::IMUL1SrcOp(OpcodeArgs) {
|
||||
@@ -3407,12 +3417,14 @@ void OpDispatchBuilder::DAAOp(OpcodeArgs) {
|
||||
auto AL = LoadGPRRegister(X86State::REG_RAX, 1);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(0));
|
||||
CalculateDeferredFlags();
|
||||
|
||||
auto Cond = _Or(AF, _Select(FEXCore::IR::COND_UGT, _And(AL, _Constant(0xF)), _Constant(9), _Constant(1), _Constant(0)));
|
||||
auto FalseBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
auto TrueBlock = CreateNewCodeBlockAfter(FalseBlock);
|
||||
auto EndBlock = CreateNewCodeBlockAfter(TrueBlock);
|
||||
|
||||
CalculateDeferredFlags();
|
||||
_CondJump(Cond, TrueBlock, FalseBlock);
|
||||
SetCurrentCodeBlock(FalseBlock);
|
||||
{
|
||||
@@ -3425,8 +3437,13 @@ void OpDispatchBuilder::DAAOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RAX, NewAL, 1);
|
||||
CalculateDeferredFlags();
|
||||
auto NewCF = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC);
|
||||
// XXX: I don't think this is correct. Needs Investigation.
|
||||
// The `CF` variable is the original CF from the start of the operation
|
||||
// The `NewCF` will be _Constant(0) stored aboved.
|
||||
// So Or(CF, _Constant(0)) ill mean CF gets updated to the old value in the true case?
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Or(CF, NewCF));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(1));
|
||||
CalculateDeferredFlags();
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(EndBlock);
|
||||
@@ -3439,6 +3456,7 @@ void OpDispatchBuilder::DAAOp(OpcodeArgs) {
|
||||
SetCurrentCodeBlock(FalseBlock);
|
||||
{
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(0));
|
||||
CalculateDeferredFlags();
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(TrueBlock);
|
||||
@@ -3448,6 +3466,7 @@ void OpDispatchBuilder::DAAOp(OpcodeArgs) {
|
||||
auto NewAL = _Add(AL, _Constant(0x60));
|
||||
StoreGPRRegister(X86State::REG_RAX, NewAL, 1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(1));
|
||||
CalculateDeferredFlags();
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(EndBlock);
|
||||
@@ -3456,10 +3475,7 @@ void OpDispatchBuilder::DAAOp(OpcodeArgs) {
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(_Select(FEXCore::IR::COND_UGE, _And(AL, _Constant(0x80)), _Constant(0), _Constant(1), _Constant(0)));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(_Select(FEXCore::IR::COND_EQ, _And(AL, _Constant(0xFF)), _Constant(0), _Constant(1), _Constant(0)));
|
||||
auto EightBitMask = _Constant(0xFF);
|
||||
auto PopCountOp = _Popcount(_And(AL, EightBitMask));
|
||||
auto XorOp = _Xor(PopCountOp, _Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(XorOp);
|
||||
CalculatePFUncheckedABI(AL);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::DASOp(OpcodeArgs) {
|
||||
@@ -3469,12 +3485,14 @@ void OpDispatchBuilder::DASOp(OpcodeArgs) {
|
||||
auto AL = LoadGPRRegister(X86State::REG_RAX, 1);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(0));
|
||||
CalculateDeferredFlags();
|
||||
|
||||
auto Cond = _Or(AF, _Select(FEXCore::IR::COND_UGT, _And(AL, _Constant(0xf)), _Constant(9), _Constant(1), _Constant(0)));
|
||||
auto FalseBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
auto TrueBlock = CreateNewCodeBlockAfter(FalseBlock);
|
||||
auto EndBlock = CreateNewCodeBlockAfter(TrueBlock);
|
||||
|
||||
CalculateDeferredFlags();
|
||||
_CondJump(Cond, TrueBlock, FalseBlock);
|
||||
SetCurrentCodeBlock(FalseBlock);
|
||||
{
|
||||
@@ -3487,8 +3505,13 @@ void OpDispatchBuilder::DASOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RAX, NewAL, 1);
|
||||
CalculateDeferredFlags();
|
||||
auto NewCF = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC);
|
||||
// XXX: I don't think this is correct. Needs Investigation.
|
||||
// The `CF` variable is the original CF from the start of the operation
|
||||
// The `NewCF` will be _Constant(0) stored aboved.
|
||||
// So Or(CF, _Constant(0)) ill mean CF gets updated to the old value in the true case?
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Or(CF, NewCF));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(1));
|
||||
CalculateDeferredFlags();
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(EndBlock);
|
||||
@@ -3501,6 +3524,7 @@ void OpDispatchBuilder::DASOp(OpcodeArgs) {
|
||||
SetCurrentCodeBlock(FalseBlock);
|
||||
{
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(0));
|
||||
CalculateDeferredFlags();
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(TrueBlock);
|
||||
@@ -3509,6 +3533,7 @@ void OpDispatchBuilder::DASOp(OpcodeArgs) {
|
||||
auto NewAL = _Sub(AL, _Constant(0x60));
|
||||
StoreGPRRegister(X86State::REG_RAX, NewAL, 1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(1));
|
||||
CalculateDeferredFlags();
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(EndBlock);
|
||||
@@ -3516,13 +3541,12 @@ void OpDispatchBuilder::DASOp(OpcodeArgs) {
|
||||
AL = LoadGPRRegister(X86State::REG_RAX, 1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(_Select(FEXCore::IR::COND_UGE, _And(AL, _Constant(0x80)), _Constant(0), _Constant(1), _Constant(0)));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(_Select(FEXCore::IR::COND_EQ, _And(AL, _Constant(0xFF)), _Constant(0), _Constant(1), _Constant(0)));
|
||||
auto EightBitMask = _Constant(0xFF);
|
||||
auto PopCountOp = _Popcount(_And(AL, EightBitMask));
|
||||
auto XorOp = _Xor(PopCountOp, _Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(XorOp);
|
||||
CalculatePFUncheckedABI(AL);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AAAOp(OpcodeArgs) {
|
||||
InvalidateDeferredFlags();
|
||||
|
||||
auto AF = GetRFLAG(FEXCore::X86State::RFLAG_AF_LOC);
|
||||
auto AL = LoadGPRRegister(X86State::REG_RAX, 1);
|
||||
auto AX = LoadGPRRegister(X86State::REG_RAX, 2);
|
||||
@@ -3539,6 +3563,7 @@ void OpDispatchBuilder::AAAOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RAX, NewAX, 2);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(0));
|
||||
CalculateDeferredFlags();
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(TrueBlock);
|
||||
@@ -3548,12 +3573,15 @@ void OpDispatchBuilder::AAAOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RAX, Result, 2);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(1));
|
||||
CalculateDeferredFlags();
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(EndBlock);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AASOp(OpcodeArgs) {
|
||||
InvalidateDeferredFlags();
|
||||
|
||||
auto AF = GetRFLAG(FEXCore::X86State::RFLAG_AF_LOC);
|
||||
auto AL = LoadGPRRegister(X86State::REG_RAX, 1);
|
||||
auto AX = LoadGPRRegister(X86State::REG_RAX, 2);
|
||||
@@ -3570,6 +3598,7 @@ void OpDispatchBuilder::AASOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RAX, NewAX, 2);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(0));
|
||||
CalculateDeferredFlags();
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(TrueBlock);
|
||||
@@ -3580,12 +3609,15 @@ void OpDispatchBuilder::AASOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RAX, Result, 2);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(1));
|
||||
CalculateDeferredFlags();
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(EndBlock);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AAMOp(OpcodeArgs) {
|
||||
InvalidateDeferredFlags();
|
||||
|
||||
auto AL = LoadGPRRegister(X86State::REG_RAX, 1);
|
||||
auto Imm8 = _Constant(Op->Src[0].Data.Literal.Value & 0xFF);
|
||||
auto UDivOp = _UDiv(AL, Imm8);
|
||||
@@ -3598,13 +3630,12 @@ void OpDispatchBuilder::AAMOp(OpcodeArgs) {
|
||||
AL = LoadGPRRegister(X86State::REG_RAX, 1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(_Select(FEXCore::IR::COND_UGE, _And(AL, _Constant(0x80)), _Constant(0), _Constant(1), _Constant(0)));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(_Select(FEXCore::IR::COND_EQ, _And(AL, _Constant(0xFF)), _Constant(0), _Constant(1), _Constant(0)));
|
||||
auto EightBitMask = _Constant(0xFF);
|
||||
auto PopCountOp = _Popcount(_And(AL, EightBitMask));
|
||||
auto XorOp = _Xor(PopCountOp, _Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(XorOp);
|
||||
CalculatePFUncheckedABI(AL);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AADOp(OpcodeArgs) {
|
||||
InvalidateDeferredFlags();
|
||||
|
||||
auto AL = LoadGPRRegister(X86State::REG_RAX, 1);
|
||||
auto AH = _Lshr(LoadGPRRegister(X86State::REG_RAX, 2), _Constant(8));
|
||||
auto Imm8 = _Constant(Op->Src[0].Data.Literal.Value & 0xFF);
|
||||
@@ -3616,10 +3647,7 @@ void OpDispatchBuilder::AADOp(OpcodeArgs) {
|
||||
AL = LoadGPRRegister(X86State::REG_RAX, 1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(_Select(FEXCore::IR::COND_UGE, _And(AL, _Constant(0x80)), _Constant(0), _Constant(1), _Constant(0)));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(_Select(FEXCore::IR::COND_EQ, _And(AL, _Constant(0xFF)), _Constant(0), _Constant(1), _Constant(0)));
|
||||
auto EightBitMask = _Constant(0xFF);
|
||||
auto PopCountOp = _Popcount(_And(AL, EightBitMask));
|
||||
auto XorOp = _Xor(PopCountOp, _Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(XorOp);
|
||||
CalculatePFUncheckedABI(AL);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::XLATOp(OpcodeArgs) {
|
||||
@@ -4020,6 +4048,7 @@ void OpDispatchBuilder::CMPSOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RSI, Dest_RSI);
|
||||
|
||||
OrderedNode *ZF = GetRFLAG(FEXCore::X86State::RFLAG_ZF_LOC);
|
||||
CalculateDeferredFlags();
|
||||
InternalCondJump = _CondJump(ZF, {REPE ? COND_NEQ : COND_EQ});
|
||||
|
||||
// Jump back to the start if we have more work to do
|
||||
@@ -4238,6 +4267,7 @@ void OpDispatchBuilder::SCASOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RDI, TailDest_RDI);
|
||||
|
||||
OrderedNode *ZF = GetRFLAG(FEXCore::X86State::RFLAG_ZF_LOC);
|
||||
CalculateDeferredFlags();
|
||||
InternalCondJump = _CondJump(ZF, {REPE ? COND_NEQ : COND_EQ});
|
||||
|
||||
// Jump back to the start if we have more work to do
|
||||
@@ -4269,7 +4299,7 @@ void OpDispatchBuilder::BSWAPOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PUSHFOp(OpcodeArgs) {
|
||||
const uint8_t Size = GetSrcSize(Op);
|
||||
|
||||
OrderedNode *Src = GetPackedRFLAG(false);
|
||||
OrderedNode *Src = GetPackedRFLAG();
|
||||
if (Size != 8) {
|
||||
Src = _Bfe(Size * 8, 0, Src);
|
||||
}
|
||||
@@ -4659,18 +4689,16 @@ void OpDispatchBuilder::CMPXCHGPairOp(OpcodeArgs) {
|
||||
OrderedNode *Result_Upper = _ExtractElementPair(CASResult, 1);
|
||||
|
||||
// Set ZF if memory result was expected
|
||||
OrderedNode *EOR_Lower = _Xor(Result_Lower, Expected_Lower);
|
||||
OrderedNode *EOR_Upper = _Xor(Result_Upper, Expected_Upper);
|
||||
OrderedNode *Orr_Result = _Or(EOR_Lower, EOR_Upper);
|
||||
|
||||
auto OneConst = _Constant(1);
|
||||
auto ZeroConst = _Constant(0);
|
||||
|
||||
OrderedNode *ZFResult = _Select(FEXCore::IR::COND_EQ,
|
||||
Orr_Result, ZeroConst,
|
||||
CASResult, Expected,
|
||||
OneConst, ZeroConst);
|
||||
|
||||
// Set ZF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(ZFResult);
|
||||
CalculateDeferredFlags();
|
||||
|
||||
auto CondJump = _CondJump(ZFResult);
|
||||
|
||||
@@ -4707,8 +4735,6 @@ void OpDispatchBuilder::CreateJumpBlocks(fextl::vector<FEXCore::Frontend::Decode
|
||||
void OpDispatchBuilder::BeginFunction(uint64_t RIP, fextl::vector<FEXCore::Frontend::Decoder::DecodedBlocks> const *Blocks) {
|
||||
Entry = RIP;
|
||||
auto IRHeader = _IRHeader(InvalidNode, 0);
|
||||
Current_Header = IRHeader.first;
|
||||
Current_HeaderNode = IRHeader;
|
||||
CreateJumpBlocks(Blocks);
|
||||
|
||||
auto Block = GetNewJumpBlock(RIP);
|
||||
@@ -4737,6 +4763,7 @@ void OpDispatchBuilder::Finalize() {
|
||||
|
||||
// We haven't emitted. Dump out to the dispatcher
|
||||
SetCurrentCodeBlock(Handler.second.BlockEntry);
|
||||
CalculateDeferredFlags();
|
||||
_ExitFunction(_EntrypointOffset(Handler.first - Entry, GPRSize));
|
||||
}
|
||||
}
|
||||
@@ -4942,8 +4969,19 @@ OrderedNode *OpDispatchBuilder::LoadSource_WithOpSize(FEXCore::IR::RegisterClass
|
||||
LoadableType = true;
|
||||
}
|
||||
else if (Operand.IsSIB()) {
|
||||
OrderedNode *Tmp {};
|
||||
if (Operand.Data.SIB.Index != FEXCore::X86State::REG_INVALID) {
|
||||
const bool IsVSIB = (Op->Flags & X86Tables::DecodeFlags::FLAG_VSIB_BYTE) != 0;
|
||||
OrderedNode *Tmp{};
|
||||
|
||||
// NOTE: VSIB cannot have the index * scale portion calculated ahead of time,
|
||||
// since the index in this case is a vector. So, we can't just apply the scale
|
||||
// to it, since this needs to be applied to each element in the index register
|
||||
// after said element has been sign extended. So, we pass this through for the
|
||||
// instruction implementation to handle.
|
||||
//
|
||||
// What we do handle though, is the applying the displacement value to
|
||||
// the base register (if a base register is provided), since this is a
|
||||
// part of the address calculation that can be done ahead of time.
|
||||
if (Operand.Data.SIB.Index != FEXCore::X86State::REG_INVALID && !IsVSIB) {
|
||||
Tmp = LoadGPRRegister(Operand.Data.SIB.Index, GPRSize);
|
||||
|
||||
if (Operand.Data.SIB.Scale != 1) {
|
||||
@@ -5106,7 +5144,6 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
if (OpSize != VectorSize) {
|
||||
// Partial writes can come from FPRs.
|
||||
// TODO: Fix the instructions doing partial writes rather than dealing with it here.
|
||||
auto SrcVector = LoadXMMRegister(gprIndex);
|
||||
|
||||
LOGMAN_THROW_A_FMT(Class != IR::GPRClass, "Partial writes from GPR not allowed. Instruction: {}",
|
||||
Op->TableInfo->Name);
|
||||
@@ -5116,6 +5153,7 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
if (VectorSize == Core::CPUState::XMM_AVX_REG_SIZE && OpSize == Core::CPUState::XMM_SSE_REG_SIZE) {
|
||||
Result = _VMov(OpSize, Src);
|
||||
} else {
|
||||
auto SrcVector = LoadXMMRegister(gprIndex);
|
||||
Result = _VInsElement(VectorSize, OpSize, 0, 0, SrcVector, Src);
|
||||
}
|
||||
}
|
||||
@@ -5315,11 +5353,25 @@ void OpDispatchBuilder::ALUOpImpl(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCor
|
||||
else {
|
||||
Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
|
||||
auto ALUOp = _Add(Dest, Src);
|
||||
// Overwrite our IR's op type
|
||||
ALUOp.first->Header.Op = ALUIROp;
|
||||
/* On x86, the canonical way to zero a register is XOR with itself...
|
||||
* because modern x86 detects this pattern in hardware. arm64 does not
|
||||
* detect this pattern, we should do it like the x86 hardware would. On
|
||||
* arm64, "mov x0, #0" is faster than "eor x0, x0, x0". Additionally this
|
||||
* lets more constant folding kick in for flags.
|
||||
*/
|
||||
if (ALUIROp == FEXCore::IR::IROps::OP_XOR &&
|
||||
Op->Dest.IsGPR() && Op->Src[0].IsGPR() &&
|
||||
Op->Dest.Data.GPR == Op->Src[0].Data.GPR) {
|
||||
|
||||
Result = _Constant(0);
|
||||
} else {
|
||||
auto ALUOp = _Add(Dest, Src);
|
||||
// Overwrite our IR's op type
|
||||
ALUOp.first->Header.Op = ALUIROp;
|
||||
|
||||
Result = ALUOp;
|
||||
}
|
||||
|
||||
Result = ALUOp;
|
||||
StoreResult(GPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
@@ -5432,6 +5484,7 @@ void OpDispatchBuilder::INTOp(OpcodeArgs) {
|
||||
|
||||
if (Op->OP == 0xCE) { // Conditional to only break if Overflow == 1
|
||||
auto Flag = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC);
|
||||
CalculateDeferredFlags();
|
||||
|
||||
// If condition doesn't hold then keep going
|
||||
auto CondJump = _CondJump(Flag, {COND_EQ});
|
||||
@@ -5477,11 +5530,6 @@ void OpDispatchBuilder::MOVBEOp(OpcodeArgs) {
|
||||
StoreResult(GPRClass, Op, Src, 1);
|
||||
}
|
||||
|
||||
template<uint8_t FenceType>
|
||||
void OpDispatchBuilder::FenceOp(OpcodeArgs) {
|
||||
_Fence({FenceType});
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CLWB(OpcodeArgs) {
|
||||
OrderedNode *DestMem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false);
|
||||
DestMem = AppendSegmentOffset(DestMem, Op->Flags);
|
||||
@@ -5494,6 +5542,15 @@ void OpDispatchBuilder::CLFLUSHOPT(OpcodeArgs) {
|
||||
_CacheLineClear(DestMem, false);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::LoadFenceOrXRSTOR(OpcodeArgs) {
|
||||
// 0xE8 signifies LFENCE
|
||||
if (Op->ModRM == 0xE8) {
|
||||
_Fence(IR::Fence_Load);
|
||||
} else {
|
||||
XRstorOpImpl(Op);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::MemFenceOrXSAVEOPT(OpcodeArgs) {
|
||||
if (Op->ModRM == 0xF0) {
|
||||
// 0xF0 is MFENCE
|
||||
@@ -6713,9 +6770,10 @@ constexpr uint16_t PF_F2 = 3;
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 1), 1, &OpDispatchBuilder::FXRStoreOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 2), 1, &OpDispatchBuilder::LDMXCSR},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 3), 1, &OpDispatchBuilder::STMXCSR},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 5), 1, &OpDispatchBuilder::FenceOp<FEXCore::IR::Fence_Load.Val>}, //LFENCE
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 6), 1, &OpDispatchBuilder::MemFenceOrXSAVEOPT}, //MFENCE
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 7), 1, &OpDispatchBuilder::StoreFenceOrCLFlush}, //SFENCE
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 4), 1, &OpDispatchBuilder::XSaveOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 5), 1, &OpDispatchBuilder::LoadFenceOrXRSTOR}, // LFENCE (or XRSTOR)
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 6), 1, &OpDispatchBuilder::MemFenceOrXSAVEOPT}, // MFENCE (or XSAVEOPT)
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 7), 1, &OpDispatchBuilder::StoreFenceOrCLFlush}, // SFENCE (or CLFLUSH)
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 5), 1, &OpDispatchBuilder::UnimplementedOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 6), 1, &OpDispatchBuilder::UnimplementedOp},
|
||||
|
||||
+258
-65
@@ -28,13 +28,6 @@ class OpDispatchBuilder final : public IREmitter {
|
||||
friend class FEXCore::IR::Pass;
|
||||
friend class FEXCore::IR::PassManager;
|
||||
|
||||
enum class SelectionFlag {
|
||||
Nothing, // must rely on x86 flags
|
||||
CMP, // flags were set by a CMP between flagsOpDest/flagsOpDestSigned and flagsOpSrc/flagsOpSrcSigned with flagsOpSize size
|
||||
AND, // flags were set by an AND/TEST, flagsOpDest contains the resulting value of flagsOpSize size
|
||||
FCMP, // flags were set by a ucomis* / comis*
|
||||
};
|
||||
|
||||
public:
|
||||
enum class FlagsGenerationType : uint8_t {
|
||||
TYPE_NONE,
|
||||
@@ -49,6 +42,7 @@ public:
|
||||
TYPE_LSHLI,
|
||||
TYPE_LSHR,
|
||||
TYPE_LSHRI,
|
||||
TYPE_LSHRDI,
|
||||
TYPE_ASHR,
|
||||
TYPE_ASHRI,
|
||||
TYPE_ROR,
|
||||
@@ -68,25 +62,6 @@ public:
|
||||
TYPE_RDRAND,
|
||||
};
|
||||
|
||||
SelectionFlag flagsOp{};
|
||||
uint8_t flagsOpSize{};
|
||||
OrderedNode* flagsOpDest{};
|
||||
OrderedNode* flagsOpSrc{};
|
||||
OrderedNode* flagsOpDestSigned{};
|
||||
OrderedNode* flagsOpSrcSigned{};
|
||||
|
||||
FEXCore::Context::ContextImpl *CTX{};
|
||||
|
||||
// Used during new op bringup
|
||||
bool ShouldDump {false};
|
||||
|
||||
struct JumpTargetInfo {
|
||||
OrderedNode* BlockEntry;
|
||||
bool HaveEmitted;
|
||||
};
|
||||
|
||||
fextl::map<uint64_t, JumpTargetInfo> JumpTargets;
|
||||
|
||||
OrderedNode* GetNewJumpBlock(uint64_t RIP) {
|
||||
auto it = JumpTargets.find(RIP);
|
||||
LOGMAN_THROW_A_FMT(it != JumpTargets.end(), "Couldn't find block generated for 0x{:x}", RIP);
|
||||
@@ -108,6 +83,10 @@ public:
|
||||
|
||||
void StartNewBlock() {
|
||||
flagsOp = SelectionFlag::Nothing;
|
||||
|
||||
// If we loaded flags but didn't change them, invalidate the cached copy and move on.
|
||||
// Changes get stored out by CalculateDeferredFlags.
|
||||
CachedNZCV = nullptr;
|
||||
}
|
||||
|
||||
bool FinishOp(uint64_t NextRIP, bool LastOp) {
|
||||
@@ -149,6 +128,31 @@ public:
|
||||
return false;
|
||||
}
|
||||
|
||||
static bool CanHaveSideEffects(FEXCore::X86Tables::X86InstInfo const* TableInfo, FEXCore::X86Tables::DecodedOp Op) {
|
||||
if (TableInfo && TableInfo->Flags & X86Tables::InstFlags::FLAGS_DEBUG_MEM_ACCESS) {
|
||||
// If it is marked as having memory access then always say it has a side-effect.
|
||||
// Not always true but better to be safe.
|
||||
return true;
|
||||
}
|
||||
|
||||
auto CanHaveSideEffects = false;
|
||||
|
||||
auto HasPotentialMemoryAccess = [](X86Tables::DecodedOperand const &Operand) -> bool {
|
||||
if (Operand.IsNone()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// This isn't guaranteed that all of these types will access memory, but be safe.
|
||||
return Operand.IsGPRDirect() || Operand.IsGPRIndirect() || Operand.IsRIPRelative() || Operand.IsSIB();
|
||||
};
|
||||
|
||||
CanHaveSideEffects |= HasPotentialMemoryAccess(Op->Dest);
|
||||
CanHaveSideEffects |= HasPotentialMemoryAccess(Op->Src[0]);
|
||||
CanHaveSideEffects |= HasPotentialMemoryAccess(Op->Src[1]);
|
||||
CanHaveSideEffects |= HasPotentialMemoryAccess(Op->Src[2]);
|
||||
return CanHaveSideEffects;
|
||||
}
|
||||
|
||||
OpDispatchBuilder(FEXCore::Context::ContextImpl *ctx);
|
||||
OpDispatchBuilder(FEXCore::Utils::IntrusivePooledAllocator &Allocator);
|
||||
|
||||
@@ -157,6 +161,12 @@ public:
|
||||
bool HadDecodeFailure() const { return DecodeFailure; }
|
||||
bool NeedsBlockEnder() const { return NeedsBlockEnd; }
|
||||
|
||||
void ResetHandledLock() { HandledLock = false; }
|
||||
bool HasHandledLock() const { return HandledLock; }
|
||||
|
||||
void SetDumpIR(bool DumpIR) { ShouldDump = DumpIR; }
|
||||
bool ShouldDumpIR() const { return ShouldDump; }
|
||||
|
||||
void BeginFunction(uint64_t RIP, fextl::vector<FEXCore::Frontend::Decoder::DecodedBlocks> const *Blocks);
|
||||
void Finalize();
|
||||
|
||||
@@ -699,6 +709,8 @@ public:
|
||||
void FXSaveOp(OpcodeArgs);
|
||||
void FXRStoreOp(OpcodeArgs);
|
||||
|
||||
void XSaveOp(OpcodeArgs);
|
||||
|
||||
void PAlignrOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void UCOMISxOp(OpcodeArgs);
|
||||
@@ -748,11 +760,9 @@ public:
|
||||
void PHADDS(OpcodeArgs);
|
||||
void PHSUBS(OpcodeArgs);
|
||||
|
||||
template<uint8_t FenceType>
|
||||
void FenceOp(OpcodeArgs);
|
||||
|
||||
void CLWB(OpcodeArgs);
|
||||
void CLFLUSHOPT(OpcodeArgs);
|
||||
void LoadFenceOrXRSTOR(OpcodeArgs);
|
||||
void MemFenceOrXSAVEOPT(OpcodeArgs);
|
||||
void StoreFenceOrCLFlush(OpcodeArgs);
|
||||
void CLZeroOp(OpcodeArgs);
|
||||
@@ -802,16 +812,54 @@ public:
|
||||
void InvalidOp(OpcodeArgs);
|
||||
|
||||
void SetPackedRFLAG(bool Lower8, OrderedNode *Src);
|
||||
OrderedNode *GetPackedRFLAG(bool Lower8);
|
||||
OrderedNode *GetPackedRFLAG(uint32_t FlagsMask = ~0U);
|
||||
|
||||
void SetMultiblock(bool _Multiblock) { Multiblock = _Multiblock; }
|
||||
|
||||
bool HandledLock = false;
|
||||
private:
|
||||
enum class SelectionFlag {
|
||||
Nothing, // must rely on x86 flags
|
||||
CMP, // flags were set by a CMP between flagsOpDest/flagsOpDestSigned and flagsOpSrc/flagsOpSrcSigned with flagsOpSize size
|
||||
AND, // flags were set by an AND/TEST, flagsOpDest contains the resulting value of flagsOpSize size
|
||||
FCMP, // flags were set by a ucomis* / comis*
|
||||
};
|
||||
|
||||
struct JumpTargetInfo {
|
||||
OrderedNode* BlockEntry;
|
||||
bool HaveEmitted;
|
||||
};
|
||||
|
||||
FEXCore::Context::ContextImpl *CTX{};
|
||||
|
||||
SelectionFlag flagsOp{};
|
||||
uint8_t flagsOpSize{};
|
||||
OrderedNode* flagsOpDest{};
|
||||
OrderedNode* flagsOpSrc{};
|
||||
OrderedNode* flagsOpDestSigned{};
|
||||
OrderedNode* flagsOpSrcSigned{};
|
||||
|
||||
static bool IsNZCV(unsigned BitOffset) {
|
||||
switch (BitOffset) {
|
||||
case FEXCore::X86State::RFLAG_CF_LOC:
|
||||
case FEXCore::X86State::RFLAG_ZF_LOC:
|
||||
case FEXCore::X86State::RFLAG_SF_LOC:
|
||||
case FEXCore::X86State::RFLAG_OF_LOC:
|
||||
return true;
|
||||
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode* CachedNZCV = {};
|
||||
uint32_t PossiblySetNZCVBits = 0;
|
||||
|
||||
fextl::map<uint64_t, JumpTargetInfo> JumpTargets;
|
||||
bool HandledLock{false};
|
||||
bool DecodeFailure{false};
|
||||
bool NeedsBlockEnd{false};
|
||||
FEXCore::IR::IROp_IRHeader *Current_Header{};
|
||||
OrderedNode *Current_HeaderNode{};
|
||||
// Used during new op bringup
|
||||
bool ShouldDump{false};
|
||||
|
||||
void ALUOpImpl(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp, bool RequiresMask);
|
||||
|
||||
@@ -953,6 +1001,23 @@ private:
|
||||
|
||||
OrderedNode* Vector_CVT_Int_To_FloatImpl(OpcodeArgs, size_t SrcElementSize, bool Widen);
|
||||
|
||||
void XSaveOpImpl(OpcodeArgs);
|
||||
void SaveX87State(OpcodeArgs, OrderedNode *MemBase);
|
||||
void SaveSSEState(OrderedNode *MemBase);
|
||||
void SaveMXCSRState(OrderedNode *MemBase);
|
||||
void SaveAVXState(OrderedNode *MemBase);
|
||||
|
||||
void XRstorOpImpl(OpcodeArgs);
|
||||
void RestoreX87State(OrderedNode *MemBase);
|
||||
void RestoreSSEState(OrderedNode *MemBase);
|
||||
void RestoreMXCSRState(OrderedNode *MXCSR);
|
||||
void RestoreAVXState(OrderedNode *MemBase);
|
||||
void DefaultX87State(OpcodeArgs);
|
||||
void DefaultSSEState();
|
||||
void DefaultAVXState();
|
||||
|
||||
OrderedNode *GetMXCSR();
|
||||
|
||||
#undef OpcodeArgs
|
||||
|
||||
OrderedNode *AppendSegmentOffset(OrderedNode *Value, uint32_t Flags, uint32_t DefaultPrefix = 0, bool Override = false);
|
||||
@@ -996,19 +1061,123 @@ private:
|
||||
[[nodiscard]] uint32_t GetDstBitSize(X86Tables::DecodedOp Op) const;
|
||||
[[nodiscard]] uint32_t GetSrcBitSize(X86Tables::DecodedOp Op) const;
|
||||
|
||||
static inline constexpr unsigned IndexNZCV(unsigned BitOffset) {
|
||||
switch (BitOffset) {
|
||||
case FEXCore::X86State::RFLAG_OF_LOC: return 28;
|
||||
case FEXCore::X86State::RFLAG_CF_LOC: return 29;
|
||||
case FEXCore::X86State::RFLAG_ZF_LOC: return 30;
|
||||
case FEXCore::X86State::RFLAG_SF_LOC: return 31;
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode *GetNZCV() {
|
||||
if (!CachedNZCV) {
|
||||
CachedNZCV = _LoadFlag(FEXCore::X86State::RFLAG_NZCV_LOC);
|
||||
|
||||
// We don't know what's set
|
||||
PossiblySetNZCVBits = ~0;
|
||||
}
|
||||
|
||||
return CachedNZCV;
|
||||
}
|
||||
|
||||
void SetNZCV(OrderedNode *Value) {
|
||||
CachedNZCV = Value;
|
||||
}
|
||||
|
||||
void ZeroNZCV() {
|
||||
CachedNZCV = _Constant(0);
|
||||
PossiblySetNZCVBits = 0;
|
||||
}
|
||||
|
||||
void ZeroCV() {
|
||||
// Get old NZCV before we mess with PossiblySetNZCVBits
|
||||
auto OldNZCV = GetNZCV();
|
||||
|
||||
// Mask out the NZ bits, clearing CV. Even if the code sets CV after, this can end up faster
|
||||
// moves by allowing orlshl to be used instead of bfi.
|
||||
PossiblySetNZCVBits = (1u << IndexNZCV(FEXCore::X86State::RFLAG_SF_LOC)) |
|
||||
(1u << IndexNZCV(FEXCore::X86State::RFLAG_ZF_LOC));
|
||||
SetNZCV(_And(OldNZCV, _Constant(PossiblySetNZCVBits)));
|
||||
}
|
||||
|
||||
void SetN_ZeroZCV(unsigned SrcSize, OrderedNode *Res) {
|
||||
static_assert(IndexNZCV(FEXCore::X86State::RFLAG_SF_LOC) == 31);
|
||||
|
||||
unsigned NBit = 31;
|
||||
unsigned SignBit = (SrcSize * 8) - 1;
|
||||
|
||||
OrderedNode *Shifted;
|
||||
|
||||
// Shift the sign bit into the N bit
|
||||
if (SignBit > NBit)
|
||||
Shifted = _Ashr(Res, _Constant(SignBit - NBit));
|
||||
else if (SignBit < NBit)
|
||||
Shifted = _Lshl(Res, _Constant(NBit - SignBit));
|
||||
else
|
||||
Shifted = Res;
|
||||
|
||||
// Mask off just the N bit, which now equals the sign bit
|
||||
CachedNZCV = _And(Shifted, _Constant(1u << NBit));
|
||||
PossiblySetNZCVBits = (1u << NBit);
|
||||
}
|
||||
|
||||
void SetNZ_ZeroCV(unsigned SrcSize, OrderedNode *Res) {
|
||||
// The TestNZ opcode does this operation natively for 32-bit or 64-bit.
|
||||
// Otherwise we can implement the functionality ourselves with some bit math.
|
||||
if (CTX->BackendFeatures.SupportsFlags && SrcSize >= 4) {
|
||||
CachedNZCV = _TestNZ(SrcSize, Res);
|
||||
PossiblySetNZCVBits = (1u << 31) | (1u << 30);
|
||||
} else {
|
||||
// N
|
||||
SetN_ZeroZCV(SrcSize, Res);
|
||||
|
||||
// Z
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
auto SelectOp = _Select(FEXCore::IR::COND_EQ, Res, Zero, One, Zero);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(SelectOp);
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode *InsertNZCV(OrderedNode *NZCV, unsigned BitOffset, OrderedNode *Value) {
|
||||
unsigned Bit = IndexNZCV(BitOffset);
|
||||
|
||||
uint32_t SetBits = PossiblySetNZCVBits;
|
||||
PossiblySetNZCVBits |= (1u << Bit);
|
||||
|
||||
if (SetBits == 0)
|
||||
return _Lshl(Value, _Constant(Bit));
|
||||
else if (CTX->BackendFeatures.SupportsShiftedBitwise && (SetBits & (1u << Bit)) == 0)
|
||||
return _Orlshl(NZCV, Value, Bit);
|
||||
else
|
||||
return _Bfi(4, 1, Bit, NZCV, Value);
|
||||
}
|
||||
|
||||
template<unsigned BitOffset>
|
||||
void SetRFLAG(OrderedNode *Value) {
|
||||
flagsOp = SelectionFlag::Nothing;
|
||||
_StoreFlag(_Bfe(1, 0, Value), BitOffset);
|
||||
SetRFLAG(Value, BitOffset);
|
||||
}
|
||||
|
||||
void SetRFLAG(OrderedNode *Value, unsigned BitOffset) {
|
||||
flagsOp = SelectionFlag::Nothing;
|
||||
_StoreFlag(_Bfe(1, 0, Value), BitOffset);
|
||||
|
||||
if (IsNZCV(BitOffset))
|
||||
SetNZCV(InsertNZCV(GetNZCV(), BitOffset, Value));
|
||||
else
|
||||
_StoreFlag(Value, BitOffset);
|
||||
}
|
||||
|
||||
OrderedNode *GetRFLAG(unsigned BitOffset) {
|
||||
return _LoadFlag(BitOffset);
|
||||
if (IsNZCV(BitOffset)) {
|
||||
if (!CachedNZCV || (PossiblySetNZCVBits & (1u << IndexNZCV(BitOffset))))
|
||||
return _Bfe(1, 1, IndexNZCV(BitOffset), GetNZCV());
|
||||
else
|
||||
return _Constant(0);
|
||||
} else {
|
||||
return _LoadFlag(BitOffset);
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode *SelectCC(uint8_t OP, OrderedNode *TrueValue, OrderedNode *FalseValue);
|
||||
@@ -1113,34 +1282,41 @@ private:
|
||||
/**
|
||||
* @name These functions are used by the deferred flag handling while it is calculating and storing flags in to RFLAGs.
|
||||
* @{ */
|
||||
void CalculcateFlags_ADC(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF);
|
||||
void CalculcateFlags_SBB(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF);
|
||||
void CalculcateFlags_SUB(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true);
|
||||
void CalculcateFlags_ADD(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true);
|
||||
void CalculcateFlags_MUL(uint8_t SrcSize, OrderedNode *Res, OrderedNode *High);
|
||||
void CalculcateFlags_UMUL(OrderedNode *High);
|
||||
void CalculcateFlags_Logical(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_ShiftLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_ShiftLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculcateFlags_ShiftRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_ShiftRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculcateFlags_SignShiftRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_SignShiftRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculcateFlags_RotateRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_RotateLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_RotateRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculcateFlags_RotateLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculcateFlags_FCMP(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_BEXTR(OrderedNode *Src);
|
||||
void CalculcateFlags_BLSI(uint8_t SrcSize, OrderedNode *Src);
|
||||
void CalculcateFlags_BLSMSK(OrderedNode *Src);
|
||||
void CalculcateFlags_BLSR(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src);
|
||||
void CalculcateFlags_POPCOUNT(OrderedNode *Src);
|
||||
void CalculcateFlags_BZHI(uint8_t SrcSize, OrderedNode *Result, OrderedNode *Src);
|
||||
void CalculcateFlags_TZCNT(OrderedNode *Src);
|
||||
void CalculcateFlags_LZCNT(uint8_t SrcSize, OrderedNode *Src);
|
||||
void CalculcateFlags_BITSELECT(OrderedNode *Src);
|
||||
void CalculcateFlags_RDRAND(OrderedNode *Src);
|
||||
OrderedNode *LoadPF();
|
||||
void CalculatePFUncheckedABI(OrderedNode *Res, OrderedNode *condition = nullptr);
|
||||
void CalculatePF(OrderedNode *Res, OrderedNode *condition = nullptr);
|
||||
|
||||
void CalculateOF_Add(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculateFlags_ADC(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF);
|
||||
void CalculateFlags_SBB(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF);
|
||||
void CalculateFlags_SUB(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true);
|
||||
void CalculateFlags_ADD(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true);
|
||||
void CalculateFlags_MUL(uint8_t SrcSize, OrderedNode *Res, OrderedNode *High);
|
||||
void CalculateFlags_UMUL(OrderedNode *High);
|
||||
void CalculateFlags_Logical(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculateFlags_ShiftLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculateFlags_ShiftRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRightDoubleImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculateFlags_SignShiftRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculateFlags_RotateRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculateFlags_RotateLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculateFlags_RotateRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculateFlags_RotateLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculateFlags_FCMP(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculateFlags_BEXTR(OrderedNode *Src);
|
||||
void CalculateFlags_BLSI(uint8_t SrcSize, OrderedNode *Src);
|
||||
void CalculateFlags_BLSMSK(OrderedNode *Src);
|
||||
void CalculateFlags_BLSR(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src);
|
||||
void CalculateFlags_POPCOUNT(OrderedNode *Src);
|
||||
void CalculateFlags_BZHI(uint8_t SrcSize, OrderedNode *Result, OrderedNode *Src);
|
||||
void CalculateFlags_TZCNT(OrderedNode *Src);
|
||||
void CalculateFlags_LZCNT(uint8_t SrcSize, OrderedNode *Src);
|
||||
void CalculateFlags_BITSELECT(OrderedNode *Src);
|
||||
void CalculateFlags_RDRAND(OrderedNode *Src);
|
||||
/** @} */
|
||||
|
||||
/**
|
||||
@@ -1353,6 +1529,23 @@ private:
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_ShiftRightDoubleImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero.
|
||||
if (Shift == 0) return;
|
||||
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_LSHRDI,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.OneSrcImmediate = {
|
||||
.Src1 = Src1,
|
||||
.Imm = Shift,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_RotateRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
// Doesn't set all the flags, needs to calculate.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
+405
-496
File diff suppressed because it is too large.
Load diff
+296
-145
@@ -204,8 +204,8 @@ void OpDispatchBuilder::VMOVSLDUPOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::MOVSSOp(OpcodeArgs) {
|
||||
if (Op->Dest.IsGPR() && Op->Src[0].IsGPR()) {
|
||||
// MOVSS xmm1, xmm2
|
||||
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, 16, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 4, Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
auto Result = _VInsElement(16, 4, 0, 0, Dest, Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
@@ -217,7 +217,7 @@ void OpDispatchBuilder::MOVSSOp(OpcodeArgs) {
|
||||
}
|
||||
else {
|
||||
// MOVSS mem32, xmm1
|
||||
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 4, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, 4, -1);
|
||||
}
|
||||
}
|
||||
@@ -2257,40 +2257,26 @@ template
|
||||
void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<8, true>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::MASKMOVOp(OpcodeArgs) {
|
||||
// Until we get correct PHI nodes this is required to be a loop unroll
|
||||
const auto Size = uint32_t{GetSrcSize(Op)} * 8;
|
||||
const auto Size = GetSrcSize(Op);
|
||||
|
||||
OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *MaskSrc = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
// Mask only cares about the top bit of each byte
|
||||
MaskSrc = _VCMPLTZ(Size, 1, MaskSrc);
|
||||
|
||||
// Vector that will overwrite byte elements.
|
||||
OrderedNode *VectorSrc = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
|
||||
// RDI source
|
||||
auto MemDest = LoadGPRRegister(X86State::REG_RDI);
|
||||
|
||||
const size_t NumElements = Size / 64;
|
||||
for (size_t Element = 0; Element < NumElements; ++Element) {
|
||||
// Extract the current element
|
||||
auto SrcElement = _VExtractToGPR(GetSrcSize(Op), 8, Src, Element);
|
||||
auto DestElement = _VExtractToGPR(GetSrcSize(Op), 8, Dest, Element);
|
||||
// DS prefix by default.
|
||||
MemDest = AppendSegmentOffset(MemDest, Op->Flags, FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX);
|
||||
|
||||
constexpr size_t NumSelectBits = 64 / 8;
|
||||
for (size_t Select = 0; Select < NumSelectBits; ++Select) {
|
||||
auto SelectMask = _Bfe(1, 8 * Select + 7, SrcElement);
|
||||
auto CondJump = _CondJump(SelectMask, {COND_EQ});
|
||||
auto StoreBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
SetFalseJumpTarget(CondJump, StoreBlock);
|
||||
SetCurrentCodeBlock(StoreBlock);
|
||||
{
|
||||
auto DestByte = _Bfe(8, 8 * Select, DestElement);
|
||||
auto MemLocation = _Add(MemDest, _Constant(Element * 8 + Select));
|
||||
// MASKMOVDQU/MASKMOVQ is explicitly weakly-ordered on its store
|
||||
_StoreMem(GPRClass, 1, MemLocation, DestByte, 1);
|
||||
}
|
||||
auto Jump = _Jump();
|
||||
auto NextJumpTarget = CreateNewCodeBlockAfter(StoreBlock);
|
||||
SetJumpTarget(Jump, NextJumpTarget);
|
||||
SetTrueJumpTarget(CondJump, NextJumpTarget);
|
||||
SetCurrentCodeBlock(NextJumpTarget);
|
||||
}
|
||||
}
|
||||
OrderedNode *XMMReg = _LoadMem(FPRClass, Size, MemDest, 1);
|
||||
|
||||
// If the Mask element high bit is set then overwrite the element with the source, else keep the memory variant
|
||||
XMMReg = _VBSL(Size, MaskSrc, VectorSrc, XMMReg);
|
||||
_StoreMem(FPRClass, Size, MemDest, XMMReg, 1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VMASKMOVOpImpl(OpcodeArgs, size_t ElementSize, size_t DataSize, bool IsStore,
|
||||
@@ -2455,6 +2441,76 @@ void OpDispatchBuilder::FXSaveOp(OpcodeArgs) {
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false);
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
SaveX87State(Op, Mem);
|
||||
SaveSSEState(Mem);
|
||||
SaveMXCSRState(Mem);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::XSaveOp(OpcodeArgs) {
|
||||
XSaveOpImpl(Op);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::XSaveOpImpl(OpcodeArgs) {
|
||||
const auto XSaveBase = [this, Op] {
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false);
|
||||
return AppendSegmentOffset(Mem, Op->Flags);
|
||||
};
|
||||
|
||||
// NOTE: Mask should be EAX and EDX concatenated, but we only need to test
|
||||
// for features that are in the lower 32 bits, so EAX only is sufficient.
|
||||
OrderedNode *Mask = LoadGPRRegister(X86State::REG_RAX);
|
||||
OrderedNode *Base = XSaveBase();
|
||||
|
||||
const auto StoreIfFlagSet = [&](uint32_t BitIndex, auto fn, uint32_t FieldSize = 1){
|
||||
OrderedNode *BitFlag = _Bfe(FieldSize, BitIndex, Mask);
|
||||
auto CondJump = _CondJump(BitFlag, {COND_NEQ});
|
||||
|
||||
auto StoreBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
SetTrueJumpTarget(CondJump, StoreBlock);
|
||||
SetCurrentCodeBlock(StoreBlock);
|
||||
{
|
||||
fn();
|
||||
}
|
||||
auto Jump = _Jump();
|
||||
auto NextJumpTarget = CreateNewCodeBlockAfter(StoreBlock);
|
||||
SetJumpTarget(Jump, NextJumpTarget);
|
||||
SetFalseJumpTarget(CondJump, NextJumpTarget);
|
||||
SetCurrentCodeBlock(NextJumpTarget);
|
||||
};
|
||||
|
||||
// x87
|
||||
{
|
||||
StoreIfFlagSet(0, [this, Op, Base] { SaveX87State(Op, Base); });
|
||||
}
|
||||
// SSE
|
||||
{
|
||||
StoreIfFlagSet(1, [this, Base] { SaveSSEState(Base); });
|
||||
}
|
||||
// AVX
|
||||
if (CTX->HostFeatures.SupportsAVX)
|
||||
{
|
||||
StoreIfFlagSet(2, [this, Base] { SaveAVXState(Base); });
|
||||
}
|
||||
|
||||
// We need to save MXCSR and MXCSR_MASK if either SSE or AVX are requested to be saved
|
||||
{
|
||||
StoreIfFlagSet(1, [this, Base] { SaveMXCSRState(Base); }, 2);
|
||||
}
|
||||
|
||||
// Update XSTATE_BV region of the XSAVE header
|
||||
{
|
||||
OrderedNode *HeaderOffset = _Add(Base, _Constant(512));
|
||||
|
||||
// NOTE: We currently only support the first 3 bits (x87, SSE, and AVX)
|
||||
OrderedNode *RequestedFeatures = _Bfe(3, 0, Mask);
|
||||
|
||||
// XSTATE_BV section of the header is 8 bytes in size, but we only really
|
||||
// care about setting at most 3 bits in the first byte. We zero out the rest.
|
||||
_StoreMem(GPRClass, 8, HeaderOffset, RequestedFeatures);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SaveX87State(OpcodeArgs, OrderedNode *MemBase) {
|
||||
// Saves 512bytes to the memory location provided
|
||||
// Header changes depending on if REX.W is set or not
|
||||
if (Op->Flags & X86Tables::DecodeFlags::FLAG_REX_WIDENING) {
|
||||
@@ -2472,12 +2528,12 @@ void OpDispatchBuilder::FXSaveOp(OpcodeArgs) {
|
||||
|
||||
{
|
||||
auto FCW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreMem(GPRClass, 2, Mem, FCW, 2);
|
||||
_StoreMem(GPRClass, 2, MemBase, FCW, 2);
|
||||
}
|
||||
|
||||
{
|
||||
// We must construct the FSW from our various bits
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(2));
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(2));
|
||||
OrderedNode *FSW = _Constant(0);
|
||||
auto Top = GetX87Top();
|
||||
FSW = _Or(FSW, _Lshl(Top, _Constant(11)));
|
||||
@@ -2496,7 +2552,7 @@ void OpDispatchBuilder::FXSaveOp(OpcodeArgs) {
|
||||
|
||||
{
|
||||
// FTW
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(4));
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(4));
|
||||
auto FTW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FTW));
|
||||
_StoreMem(GPRClass, 2, MemLocation, FTW, 2);
|
||||
}
|
||||
@@ -2545,33 +2601,142 @@ void OpDispatchBuilder::FXSaveOp(OpcodeArgs) {
|
||||
// MXCSR_MASK: Mask for writes to the MXCSR register
|
||||
// If OSFXSR bit in CR4 is not set than FXSAVE /may/ not save the XMM registers
|
||||
// This is implementation dependent
|
||||
for (unsigned i = 0; i < 8; ++i) {
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
OrderedNode *MMReg = _LoadContext(16, FPRClass, offsetof(FEXCore::Core::CPUState, mm[i]));
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 32));
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(i * 16 + 32));
|
||||
|
||||
_StoreMem(FPRClass, 16, MemLocation, MMReg, 16);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SaveSSEState(OrderedNode *MemBase) {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (unsigned i = 0; i < NumRegs; ++i) {
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
OrderedNode *XMMReg = LoadXMMRegister(i);
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 160));
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(i * 16 + 160));
|
||||
|
||||
_StoreMem(FPRClass, 16, MemLocation, XMMReg, 16);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SaveMXCSRState(OrderedNode *MemBase) {
|
||||
OrderedNode *MXCSR = GetMXCSR();
|
||||
OrderedNode *MXCSRLocation = _Add(MemBase, _Constant(24));
|
||||
_StoreMem(GPRClass, 4, MXCSRLocation, MXCSR, 4);
|
||||
|
||||
// Store the mask for all bits.
|
||||
OrderedNode *MXCSRMaskLocation = _Add(MXCSRLocation, _Constant(4));
|
||||
_StoreMem(GPRClass, 4, MXCSRMaskLocation, _Constant(0xFFFF), 4);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SaveAVXState(OrderedNode *MemBase) {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
OrderedNode *Upper = _VDupElement(32, 16, LoadXMMRegister(i), 1);
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(i * 16 + 576));
|
||||
|
||||
_StoreMem(FPRClass, 16, MemLocation, Upper, 16);
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::GetMXCSR() {
|
||||
// Default MXCSR Value
|
||||
OrderedNode *MXCSR = _Constant(0x1F80);
|
||||
OrderedNode *RoundingMode = _GetRoundingMode();
|
||||
return _Bfi(4, 3, 13, MXCSR, RoundingMode);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FXRStoreOp(OpcodeArgs) {
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1, false);
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
RestoreX87State(Mem);
|
||||
RestoreSSEState(Mem);
|
||||
|
||||
OrderedNode *MXCSRLocation = _Add(Mem, _Constant(24));
|
||||
OrderedNode *MXCSR = _LoadMem(GPRClass, 4, MXCSRLocation, 4);
|
||||
RestoreMXCSRState(MXCSR);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::XRstorOpImpl(OpcodeArgs) {
|
||||
const auto XSaveBase = [this, Op] {
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false);
|
||||
return AppendSegmentOffset(Mem, Op->Flags);
|
||||
};
|
||||
|
||||
// Set up base address for the XSAVE region to restore from, and also read the
|
||||
// XSTATE_BV bit flags out of the XSTATE header.
|
||||
OrderedNode *Base = XSaveBase();
|
||||
OrderedNode *Mask = _LoadMem(GPRClass, 8, _Add(Base, _Constant(512)), 8);
|
||||
|
||||
// If a bit in our XSTATE_BV is set, then we restore from that region of the XSAVE area,
|
||||
// otherwise, if not set, then we need to set the relevant data the bit corresponds to
|
||||
// to it's defined initial configuration.
|
||||
const auto RestoreIfFlagSetOrDefault = [&](uint32_t BitIndex, auto restore_fn, auto default_fn, uint32_t FieldSize = 1){
|
||||
OrderedNode *BitFlag = _Bfe(FieldSize, BitIndex, Mask);
|
||||
auto CondJump = _CondJump(BitFlag, {COND_NEQ});
|
||||
|
||||
auto RestoreBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
SetTrueJumpTarget(CondJump, RestoreBlock);
|
||||
SetCurrentCodeBlock(RestoreBlock);
|
||||
{
|
||||
restore_fn();
|
||||
}
|
||||
auto RestoreExitJump = _Jump();
|
||||
auto DefaultBlock = CreateNewCodeBlockAfter(RestoreBlock);
|
||||
auto ExitBlock = CreateNewCodeBlockAfter(DefaultBlock);
|
||||
SetJumpTarget(RestoreExitJump, ExitBlock);
|
||||
SetFalseJumpTarget(CondJump, DefaultBlock);
|
||||
SetCurrentCodeBlock(DefaultBlock);
|
||||
{
|
||||
default_fn();
|
||||
}
|
||||
auto DefaultExitJump = _Jump();
|
||||
SetJumpTarget(DefaultExitJump, ExitBlock);
|
||||
SetCurrentCodeBlock(ExitBlock);
|
||||
};
|
||||
|
||||
// x87
|
||||
{
|
||||
RestoreIfFlagSetOrDefault(0,
|
||||
[this, Base] { RestoreX87State(Base); },
|
||||
[this, Op] { DefaultX87State(Op); });
|
||||
}
|
||||
// SSE
|
||||
{
|
||||
RestoreIfFlagSetOrDefault(1,
|
||||
[this, Base] { RestoreSSEState(Base); },
|
||||
[this] { DefaultSSEState(); });
|
||||
}
|
||||
// AVX
|
||||
if (CTX->HostFeatures.SupportsAVX)
|
||||
{
|
||||
RestoreIfFlagSetOrDefault(2,
|
||||
[this, Base] { RestoreAVXState(Base); },
|
||||
[this] { DefaultAVXState(); });
|
||||
}
|
||||
|
||||
{
|
||||
// We need to restore the MXCSR if either SSE or AVX are requested to be saved
|
||||
RestoreIfFlagSetOrDefault(1,
|
||||
[this, Base] {
|
||||
OrderedNode *MXCSRLocation = _Add(Base, _Constant(24));
|
||||
OrderedNode *MXCSR = _LoadMem(GPRClass, 4, MXCSRLocation, 4);
|
||||
RestoreMXCSRState(MXCSR);
|
||||
},
|
||||
[] { /* Intentionally do nothing*/ }, 2);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RestoreX87State(OrderedNode *MemBase) {
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, MemBase, 2);
|
||||
_F80LoadFCW(NewFCW);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
{
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(2));
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(2));
|
||||
auto NewFSW = _LoadMem(GPRClass, 2, MemLocation, 2);
|
||||
|
||||
// Strip out the FSW information
|
||||
@@ -2591,26 +2756,78 @@ void OpDispatchBuilder::FXRStoreOp(OpcodeArgs) {
|
||||
|
||||
{
|
||||
// FTW
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(4));
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(4));
|
||||
auto NewFTW = _LoadMem(GPRClass, 2, MemLocation, 2);
|
||||
_StoreContext(2, GPRClass, NewFTW, offsetof(FEXCore::Core::CPUState, FTW));
|
||||
}
|
||||
|
||||
for (unsigned i = 0; i < 8; ++i) {
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 32));
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(i * 16 + 32));
|
||||
auto MMReg = _LoadMem(FPRClass, 16, MemLocation, 16);
|
||||
_StoreContext(16, FPRClass, MMReg, offsetof(FEXCore::Core::CPUState, mm[i]));
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RestoreSSEState(OrderedNode *MemBase) {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (unsigned i = 0; i < NumRegs; ++i) {
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 160));
|
||||
auto XMMReg = _LoadMem(FPRClass, 16, MemLocation, 16);
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(i * 16 + 160));
|
||||
OrderedNode *XMMReg = _LoadMem(FPRClass, 16, MemLocation, 16);
|
||||
StoreXMMRegister(i, XMMReg);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RestoreMXCSRState(OrderedNode *MXCSR) {
|
||||
// We only support the rounding mode and FTZ bit being set
|
||||
OrderedNode *RoundingMode = _Bfe(4, 3, 13, MXCSR);
|
||||
_SetRoundingMode(RoundingMode);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RestoreAVXState(OrderedNode *MemBase) {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
OrderedNode *XMMReg = LoadXMMRegister(i);
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(i * 16 + 576));
|
||||
OrderedNode *YMMHReg = _LoadMem(FPRClass, 16, MemLocation, 16);
|
||||
OrderedNode *YMM = _VInsElement(32, 16, 1, 0, XMMReg, YMMHReg);
|
||||
StoreXMMRegister(i, YMM);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::DefaultX87State(OpcodeArgs) {
|
||||
// We can piggy-back on FNINIT's implementation, since
|
||||
// it performs the same behavior as required by XRSTOR for resetting flags
|
||||
FNINIT(Op);
|
||||
|
||||
// On top of resetting the flags to a default state, we also need to clear
|
||||
// all of the ST0-7/MM0-7 registers to zero.
|
||||
OrderedNode *ZeroVector = _VectorZero(Core::CPUState::MM_REG_SIZE);
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
_StoreContext(16, FPRClass, ZeroVector, offsetof(FEXCore::Core::CPUState, mm[i]));
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::DefaultSSEState() {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
OrderedNode *ZeroVector = _VectorZero(Core::CPUState::XMM_SSE_REG_SIZE);
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
StoreXMMRegister(i, ZeroVector);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::DefaultAVXState() {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; i++) {
|
||||
OrderedNode* Reg = LoadXMMRegister(i);
|
||||
OrderedNode* Dst = _VMov(16, Reg);
|
||||
StoreXMMRegister(i, Dst);
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::PALIGNROpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1,
|
||||
const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm) {
|
||||
@@ -2676,18 +2893,11 @@ void OpDispatchBuilder::UCOMISxOp<8>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::LDMXCSR(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
// We only support the rounding mode and FTZ bit being set
|
||||
OrderedNode *RoundingMode = _Bfe(4, 3, 13, Dest);
|
||||
_SetRoundingMode(RoundingMode);
|
||||
RestoreMXCSRState(Dest);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::STMXCSR(OpcodeArgs) {
|
||||
// Default MXCSR
|
||||
OrderedNode *MXCSR = _Constant(32, 0x1F80);
|
||||
OrderedNode *RoundingMode = _GetRoundingMode();
|
||||
MXCSR = _Bfi(4, 3, 13, MXCSR, RoundingMode);
|
||||
|
||||
StoreResult(GPRClass, Op, MXCSR, -1);
|
||||
StoreResult(GPRClass, Op, GetMXCSR(), -1);
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::PACKUSOpImpl(OpcodeArgs, size_t ElementSize,
|
||||
@@ -3417,36 +3627,19 @@ void OpDispatchBuilder::VPHSUBOp<2>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VPHSUBOp<4>(OpcodeArgs);
|
||||
|
||||
OrderedNode* OpDispatchBuilder::PHADDSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1,
|
||||
const X86Tables::DecodedOperand& Src2) {
|
||||
OrderedNode* OpDispatchBuilder::PHADDSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op) {
|
||||
const auto Size = GetSrcSize(Op);
|
||||
const uint8_t ElementSize = 2;
|
||||
|
||||
OrderedNode *Src1Node = LoadSource(FPRClass, Op, Src1, Op->Flags, -1);
|
||||
OrderedNode *Src2Node = LoadSource(FPRClass, Op, Src2, Op->Flags, -1);
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Src1Op, Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Src2Op, Op->Flags, -1);
|
||||
|
||||
if (Size == 8) {
|
||||
// Implementation is more efficient for 8byte registers
|
||||
OrderedNode *Src1_Larger = _VSXTL(Size * 2, 2, Src1Node);
|
||||
OrderedNode *Src2_Larger = _VSXTL(Size * 2, 2, Src2Node);
|
||||
|
||||
OrderedNode *AddRes = _VAddP(Size * 2, 4, Src1_Larger, Src2_Larger);
|
||||
|
||||
// Saturate back down to the result
|
||||
return _VSQXTN(Size * 2, 4, AddRes);
|
||||
}
|
||||
|
||||
OrderedNode *Src1_Larger = _VSXTL(Size, 2, Src1Node);
|
||||
OrderedNode *Src1_Larger_H = _VSXTL2(Size, 2, Src1Node);
|
||||
|
||||
OrderedNode *Src2_Larger = _VSXTL(Size, 2, Src2Node);
|
||||
OrderedNode *Src2_Larger_H = _VSXTL2(Size, 2, Src2Node);
|
||||
|
||||
OrderedNode *AddRes_L = _VAddP(Size, 4, Src1_Larger, Src1_Larger_H);
|
||||
OrderedNode *AddRes_H = _VAddP(Size, 4, Src2_Larger, Src2_Larger_H);
|
||||
auto Even = _VUnZip(Size, ElementSize, Src1, Src2);
|
||||
auto Odd = _VUnZip2(Size, ElementSize, Src1, Src2);
|
||||
|
||||
// Saturate back down to the result
|
||||
OrderedNode *Res = _VSQXTN(Size, 4, AddRes_L);
|
||||
return _VSQXTN2(Size, 4, Res, AddRes_H);
|
||||
return _VSQAdd(Size, ElementSize, Even, Odd);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PHADDS(OpcodeArgs) {
|
||||
@@ -3473,51 +3666,15 @@ OrderedNode* OpDispatchBuilder::PHSUBSOpImpl(OpcodeArgs, const X86Tables::Decode
|
||||
const X86Tables::DecodedOperand& Src2Op) {
|
||||
const auto Size = GetSrcSize(Op);
|
||||
const uint8_t ElementSize = 2;
|
||||
const uint8_t NumElements = Size / ElementSize;
|
||||
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Src1Op, Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Src2Op, Op->Flags, -1);
|
||||
|
||||
// This is a bit complicated since AArch64 doesn't support a pairwise subtract
|
||||
OrderedNode *Src1_Neg = _VNeg(Size, ElementSize, Src1);
|
||||
OrderedNode *Src2_Neg = _VNeg(Size, ElementSize, Src2);
|
||||
|
||||
// Now we need to swizzle the values
|
||||
OrderedNode *Swizzle_Src1 = Src1;
|
||||
OrderedNode *Swizzle_Src2 = Src2;
|
||||
|
||||
// Odd elements turn in to negated elements
|
||||
for (size_t i = 1; i < NumElements; i += 2) {
|
||||
Swizzle_Src1 = _VInsElement(Size, ElementSize, i, i, Swizzle_Src1, Src1_Neg);
|
||||
Swizzle_Src2 = _VInsElement(Size, ElementSize, i, i, Swizzle_Src2, Src2_Neg);
|
||||
}
|
||||
|
||||
Src1 = Swizzle_Src1;
|
||||
Src2 = Swizzle_Src2;
|
||||
|
||||
if (Size == 8) {
|
||||
// Implementation is more efficient for 8byte registers
|
||||
OrderedNode *Src1_Larger = _VSXTL(Size * 2, 2, Src1);
|
||||
OrderedNode *Src2_Larger = _VSXTL(Size * 2, 2, Src2);
|
||||
|
||||
OrderedNode *AddRes = _VAddP(Size * 2, 4, Src1_Larger, Src2_Larger);
|
||||
|
||||
// Saturate back down to the result
|
||||
return _VSQXTN(Size * 2, 4, AddRes);
|
||||
}
|
||||
|
||||
OrderedNode *Src1_Larger = _VSXTL(Size, 2, Src1);
|
||||
OrderedNode *Src1_Larger_H = _VSXTL2(Size, 2, Src1);
|
||||
|
||||
OrderedNode *Src2_Larger = _VSXTL(Size, 2, Src2);
|
||||
OrderedNode *Src2_Larger_H = _VSXTL2(Size, 2, Src2);
|
||||
|
||||
OrderedNode *AddRes_L = _VAddP(Size, 4, Src1_Larger, Src1_Larger_H);
|
||||
OrderedNode *AddRes_H = _VAddP(Size, 4, Src2_Larger, Src2_Larger_H);
|
||||
auto Even = _VUnZip(Size, ElementSize, Src1, Src2);
|
||||
auto Odd = _VUnZip2(Size, ElementSize, Src1, Src2);
|
||||
|
||||
// Saturate back down to the result
|
||||
OrderedNode *Res = _VSQXTN(Size, 4, AddRes_L);
|
||||
return _VSQXTN2(Size, 4, Res, AddRes_H);
|
||||
return _VSQSub(Size, ElementSize, Even, Odd);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PHSUBS(OpcodeArgs) {
|
||||
@@ -3554,33 +3711,19 @@ OrderedNode* OpDispatchBuilder::PSADBWOpImpl(OpcodeArgs,
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Src2Op, Op->Flags, -1);
|
||||
|
||||
if (Size == 8) {
|
||||
OrderedNode *Src1_Low = _VUXTL(Size*2, 1, Src1);
|
||||
OrderedNode *Src2_Low = _VUXTL(Size*2, 1, Src2);
|
||||
|
||||
OrderedNode *SubResult = _VSub(Size*2, 2, Src1_Low, Src2_Low);
|
||||
OrderedNode *AbsResult = _VAbs(Size*2, 2, SubResult);
|
||||
auto AbsResult = _VUABDL(Size * 2, 1, Src1, Src2);
|
||||
|
||||
// Now vector-wide add the results for each
|
||||
return _VAddV(Size * 2, 2, AbsResult);
|
||||
}
|
||||
|
||||
|
||||
OrderedNode *Src1_Low = _VUXTL(Size, 1, Src1);
|
||||
OrderedNode *Src1_High = _VUXTL2(Size, 1, Src1);
|
||||
|
||||
OrderedNode *Src2_Low = _VUXTL(Size, 1, Src2);
|
||||
OrderedNode *Src2_High = _VUXTL2(Size, 1, Src2);
|
||||
|
||||
OrderedNode *SubResult_Low = _VSub(Size, 2, Src1_Low, Src2_Low);
|
||||
OrderedNode *SubResult_High = _VSub(Size, 2, Src1_High, Src2_High);
|
||||
|
||||
OrderedNode *AbsResult_Low = _VAbs(Size, 2, SubResult_Low);
|
||||
OrderedNode *AbsResult_High = _VAbs(Size, 2, SubResult_High);
|
||||
auto AbsResult_Low = _VUABDL(Size, 1, Src1, Src2);
|
||||
auto AbsResult_High = _VUABDL2(Size, 1, Src1, Src2);
|
||||
|
||||
OrderedNode *Result_Low = _VAddV(16, 2, AbsResult_Low);
|
||||
OrderedNode *Result_High = _VAddV(16, 2, AbsResult_High);
|
||||
auto Low = _VZip(Size, 8, Result_Low, Result_High);
|
||||
|
||||
OrderedNode *Low = _VInsElement(Size, 8, 1, 0, Result_Low, Result_High);
|
||||
if (Is128Bit) {
|
||||
return Low;
|
||||
}
|
||||
@@ -4011,11 +4154,8 @@ OrderedNode* OpDispatchBuilder::PHMINPOSUWOpImpl(OpcodeArgs) {
|
||||
Element, MinGPR, Indexes[i - 1], Pos);
|
||||
}
|
||||
|
||||
// Insert the minimum in to bits [15:0]
|
||||
OrderedNode *Result = _VMov(2, Min);
|
||||
|
||||
// Insert position in to bits [18:16]
|
||||
return _VInsGPR(16, 2, 1, Result, Pos);
|
||||
return _VInsGPR(16, 2, 1, Min, Pos);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PHMINPOSUWOp(OpcodeArgs) {
|
||||
@@ -4593,7 +4733,7 @@ void OpDispatchBuilder::VPERMILRegOp<8>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask) {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src[1] needs to be a literal");
|
||||
const auto Control = Op->Src[1].Data.Literal.Value;
|
||||
const uint16_t Control = Op->Src[1].Data.Literal.Value;
|
||||
|
||||
// SSE4.2 string instructions modify flags, so invalidate
|
||||
// any previously deferred flags.
|
||||
@@ -4611,12 +4751,18 @@ void OpDispatchBuilder::PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask
|
||||
OrderedNode *IntermediateResult{};
|
||||
if (IsExplicit) {
|
||||
// Will be 4 in the absence of a REX.W bit and 8 in the presence of a REX.W bit.
|
||||
//
|
||||
// While the control bit immediate for the instruction itself is only ever 8 bits
|
||||
// in size, we use it as a 16-bit value so that we can use the 8th bit to signify
|
||||
// whether or not RAX and RDX should be interpreted as a 64-bit value.
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto Is64Bit = SrcSize == 8;
|
||||
const auto NewControl = uint16_t(Control | (uint16_t(Is64Bit) << 8));
|
||||
|
||||
OrderedNode *SrcRAX = LoadGPRRegister(X86State::REG_RAX);
|
||||
OrderedNode *SrcRDX = LoadGPRRegister(X86State::REG_RDX);
|
||||
|
||||
IntermediateResult = _VPCMPESTRX(SrcSize, Src1, Src2, SrcRAX, SrcRDX, Control);
|
||||
IntermediateResult = _VPCMPESTRX(Src1, Src2, SrcRAX, SrcRDX, NewControl);
|
||||
} else {
|
||||
IntermediateResult = _VPCMPISTRX(Src1, Src2, Control);
|
||||
}
|
||||
@@ -4665,7 +4811,12 @@ void OpDispatchBuilder::PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask
|
||||
OrderedNode *Result = _Select(IR::COND_EQ, ResultNoFlags, ZeroConst,
|
||||
IfZero, IfNotZero);
|
||||
|
||||
StoreGPRRegister(X86State::REG_RCX, Result, 4);
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
if (GPRSize == 8) {
|
||||
// If being stored to an 8-byte register, zero extend the 4-byte result.
|
||||
Result = _Bfe(8, 32, 0, Result);
|
||||
}
|
||||
StoreGPRRegister(X86State::REG_RCX, Result);
|
||||
}
|
||||
|
||||
// Set all of the necessary flags.
|
||||
|
||||
+14
-14
@@ -171,12 +171,14 @@ void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
auto zero = _Constant(0);
|
||||
|
||||
// Sign extend to 64bits
|
||||
if (read_width != 8)
|
||||
if (read_width != 8) {
|
||||
data = _Sext(read_width * 8, data);
|
||||
}
|
||||
|
||||
// Extract sign and make interger absolute
|
||||
auto sign = _Select(COND_SLT, data, zero, _Constant(0x8000), zero);
|
||||
auto absolute = _Select(COND_SLT, data, zero, _Sub(zero, data), data);
|
||||
|
||||
auto absolute = _Abs(data);
|
||||
|
||||
// left justify the absolute interger
|
||||
auto shift = _Sub(_Constant(63), _FindMSB(absolute));
|
||||
@@ -621,18 +623,19 @@ void OpDispatchBuilder::FXTRACT(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FNINIT(OpcodeArgs) {
|
||||
auto Zero = _Constant(0);
|
||||
// Init FCW to 0x037F
|
||||
auto NewFCW = _Constant(16, 0x037F);
|
||||
_F80LoadFCW(NewFCW);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
// Init FSW to 0
|
||||
SetX87Top(_Constant(0));
|
||||
SetX87Top(Zero);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(Zero);
|
||||
|
||||
// Tags all get set to 0b11
|
||||
_StoreContext(2, GPRClass, _Constant(0xFFFF), offsetof(FEXCore::Core::CPUState, FTW));
|
||||
@@ -1278,7 +1281,7 @@ void OpDispatchBuilder::X87FXAM(OpcodeArgs) {
|
||||
OrderedNode *Result = _VExtractToGPR(16, 8, a, 1);
|
||||
|
||||
// Extract the sign bit
|
||||
Result = _Lshr(Result, _Constant(15));
|
||||
Result = _Bfe(1, 15, Result);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Result);
|
||||
|
||||
// Claim this is a normal number
|
||||
@@ -1354,20 +1357,17 @@ void OpDispatchBuilder::X87FCMOV(OpcodeArgs) {
|
||||
break;
|
||||
}
|
||||
|
||||
auto MaskConst = _Constant(FLAGMask);
|
||||
auto RFLAG = GetPackedRFLAG(FLAGMask);
|
||||
|
||||
auto RFLAG = GetPackedRFLAG(false);
|
||||
|
||||
auto AndOp = _And(RFLAG, MaskConst);
|
||||
switch (Type) {
|
||||
case COMPARE_ZERO: {
|
||||
SrcCond = _Select(FEXCore::IR::COND_EQ,
|
||||
AndOp, ZeroConst, OneConst, ZeroConst);
|
||||
RFLAG, ZeroConst, OneConst, ZeroConst);
|
||||
break;
|
||||
}
|
||||
case COMPARE_NOTZERO: {
|
||||
SrcCond = _Select(FEXCore::IR::COND_EQ,
|
||||
AndOp, ZeroConst, ZeroConst, OneConst);
|
||||
RFLAG, ZeroConst, ZeroConst, OneConst);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -50,12 +50,12 @@ void OpDispatchBuilder::FNINITF64(OpcodeArgs) {
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
// Init FSW to 0
|
||||
SetX87Top(_Constant(0));
|
||||
SetX87Top(Zero);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(Zero);
|
||||
|
||||
// Tags all get set to 0b11
|
||||
_StoreContext(2, GPRClass, _Constant(0xFFFF), offsetof(FEXCore::Core::CPUState, FTW));
|
||||
@@ -1119,7 +1119,7 @@ void OpDispatchBuilder::X87FXAMF64(OpcodeArgs) {
|
||||
OrderedNode *Result = _VExtractToGPR(8, 8, a, 0);
|
||||
|
||||
// Extract the sign bit
|
||||
Result = _Lshr(Result, _Constant(63));
|
||||
Result = _Bfe(1, 63, Result);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Result);
|
||||
|
||||
// Claim this is a normal number
|
||||
|
||||
@@ -6,26 +6,6 @@
|
||||
#include <signal.h>
|
||||
|
||||
namespace FEXCore {
|
||||
struct ThreadState {
|
||||
FEXCore::Core::InternalThreadState *Thread{};
|
||||
};
|
||||
|
||||
thread_local ThreadState ThreadData{};
|
||||
|
||||
FEXCore::Core::InternalThreadState *SignalDelegator::GetTLSThread() {
|
||||
return ThreadData.Thread;
|
||||
}
|
||||
|
||||
void SignalDelegator::RegisterTLSState(FEXCore::Core::InternalThreadState *Thread) {
|
||||
ThreadData.Thread = Thread;
|
||||
RegisterFrontendTLSState(Thread);
|
||||
}
|
||||
|
||||
void SignalDelegator::UninstallTLSState(FEXCore::Core::InternalThreadState *Thread) {
|
||||
UninstallFrontendTLSState(Thread);
|
||||
ThreadData.Thread = nullptr;
|
||||
}
|
||||
|
||||
void SignalDelegator::RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
SetHostSignalHandler(Signal, Func, Required);
|
||||
FrontendRegisterHostSignalHandler(Signal, Func, Required);
|
||||
|
||||
@@ -146,10 +146,10 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0x9E, 1, X86InstInfo{"SAHF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x9F, 1, X86InstInfo{"LAHF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0xA4, 1, X86InstInfo{"MOVSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA5, 1, X86InstInfo{"MOVS", TYPE_INST, FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA6, 1, X86InstInfo{"CMPSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA7, 1, X86InstInfo{"CMPS", TYPE_INST, FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA4, 1, X86InstInfo{"MOVSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA5, 1, X86InstInfo{"MOVS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA6, 1, X86InstInfo{"CMPSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA7, 1, X86InstInfo{"CMPS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
|
||||
{0xA8, 1, X86InstInfo{"TEST", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1, nullptr}},
|
||||
{0xA9, 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4, nullptr}},
|
||||
|
||||
@@ -334,7 +334,7 @@ void InitializeSecondaryGroupTables() {
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 1), 1, X86InstInfo{"FXRSTOR", TYPE_INST, FLAGS_MODRM, 0, nullptr}}, // MMX/x87
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 2), 1, X86InstInfo{"LDMXCSR", TYPE_INST, GenFlagsSameSize(SIZE_32BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 3), 1, X86InstInfo{"STMXCSR", TYPE_INST, GenFlagsSameSize(SIZE_32BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 4), 1, X86InstInfo{"XSAVE", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 4), 1, X86InstInfo{"XSAVE", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 5), 1, X86InstInfo{"LFENCE/XRSTOR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 6), 1, X86InstInfo{"MFENCE/XSAVEOPT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 7), 1, X86InstInfo{"SFENCE/CLFLUSH", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
|
||||
@@ -48,7 +48,7 @@ void InitializeSecondaryModRMTables() {
|
||||
{((3 << 3) | 1), 1, X86InstInfo{"RDTSCP", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{((3 << 3) | 2), 1, X86InstInfo{"MONITORX", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{((3 << 3) | 3), 1, X86InstInfo{"MWAITX", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{((3 << 3) | 4), 1, X86InstInfo{"CLZERO", TYPE_INST, FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{((3 << 3) | 4), 1, X86InstInfo{"CLZERO", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SF_SRC_RAX | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{((3 << 3) | 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{((3 << 3) | 6), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{((3 << 3) | 7), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
@@ -342,10 +342,10 @@ void InitializeVEXTables() {
|
||||
{OPD(2, 0b01, 0x8C), 1, X86InstInfo{"VPMASKMOV", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x8E), 1, X86InstInfo{"VPMASKMOV", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SF_MOD_DST | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(2, 0b01, 0x90), 1, X86InstInfo{"VPGATHERD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x91), 1, X86InstInfo{"VPGATHERQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x92), 1, X86InstInfo{"VPGATHERD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x93), 1, X86InstInfo{"VPGATHERQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x90), 1, X86InstInfo{"VPGATHERDD/Q", TYPE_UNDEC, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_2ND_SRC | FLAGS_VEX_VSIB | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x91), 1, X86InstInfo{"VPGATHERQD/Q", TYPE_UNDEC, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_2ND_SRC | FLAGS_VEX_VSIB | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x92), 1, X86InstInfo{"VGATHERDPS/D", TYPE_UNDEC, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_2ND_SRC | FLAGS_VEX_VSIB | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x93), 1, X86InstInfo{"VGATHERQPS/D", TYPE_UNDEC, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_2ND_SRC | FLAGS_VEX_VSIB | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(2, 0b01, 0x96), 1, X86InstInfo{"VFMADDSUB132", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x97), 1, X86InstInfo{"VFMSUBADD132", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
@@ -25,7 +25,7 @@ constexpr uint32_t FLAG_ADDRESS_SIZE = (1 << 1);
|
||||
constexpr uint32_t FLAG_LOCK = (1 << 2);
|
||||
constexpr uint32_t FLAG_LEGACY_PREFIX = (1 << 3);
|
||||
constexpr uint32_t FLAG_REX_PREFIX = (1 << 4);
|
||||
// Hole where 1 << 5 is
|
||||
constexpr uint32_t FLAG_VSIB_BYTE = (1 << 5);
|
||||
// Hole where 1 << 6 is
|
||||
constexpr uint32_t FLAG_REX_WIDENING = (1 << 7);
|
||||
constexpr uint32_t FLAG_REX_XGPR_B = (1 << 8);
|
||||
@@ -138,9 +138,10 @@ struct DecodedOperand {
|
||||
}
|
||||
|
||||
union TypeUnion {
|
||||
struct {
|
||||
struct GPRType {
|
||||
bool HighBits;
|
||||
uint8_t GPR;
|
||||
auto operator<=>(const GPRType&) const = default;
|
||||
} GPR;
|
||||
|
||||
struct {
|
||||
@@ -155,9 +156,10 @@ struct DecodedOperand {
|
||||
} Value;
|
||||
} RIPLiteral;
|
||||
|
||||
struct {
|
||||
struct LiteralType {
|
||||
uint64_t Value;
|
||||
uint8_t Size;
|
||||
auto operator<=>(const LiteralType&) const = default;
|
||||
} Literal;
|
||||
|
||||
struct {
|
||||
@@ -347,6 +349,8 @@ constexpr InstFlagType FLAGS_VEX_1ST_SRC = (1ULL << 24);
|
||||
constexpr InstFlagType FLAGS_VEX_2ND_SRC = (1ULL << 25);
|
||||
// Whether or not the instruction has a VEX prefix for the destination
|
||||
constexpr InstFlagType FLAGS_VEX_DST = (1ULL << 26);
|
||||
// Whether or not the instruction has a VSIB byte
|
||||
constexpr InstFlagType FLAGS_VEX_VSIB = (1ULL << 27);
|
||||
|
||||
constexpr InstFlagType FLAGS_SIZE_DST_OFF = 58;
|
||||
constexpr InstFlagType FLAGS_SIZE_SRC_OFF = FLAGS_SIZE_DST_OFF + 3;
|
||||
|
||||
+3
-3
@@ -206,7 +206,7 @@ namespace FEXCore::IR {
|
||||
FEXCore::Allocator::YesIKnowImNotSupposedToUseTheGlibcAllocator glibc;
|
||||
|
||||
AOTIRCaptureCacheWriteoutLock.lock();
|
||||
std::function<void()> fn = std::move(AOTIRCaptureCacheWriteoutQueue.front());
|
||||
WriteOutFn fn = std::move(AOTIRCaptureCacheWriteoutQueue.front());
|
||||
bool MaybeEmpty = false;
|
||||
AOTIRCaptureCacheWriteoutQueue.pop();
|
||||
MaybeEmpty = AOTIRCaptureCacheWriteoutQueue.size() == 0;
|
||||
@@ -225,7 +225,7 @@ namespace FEXCore::IR {
|
||||
LOGMAN_MSG_A_FMT("Must never get here");
|
||||
}
|
||||
|
||||
void AOTIRCaptureCache::AOTIRCaptureCacheWriteoutQueue_Append(const std::function<void()> &fn) {
|
||||
void AOTIRCaptureCache::AOTIRCaptureCacheWriteoutQueue_Append(const WriteOutFn &fn) {
|
||||
bool Flush = false;
|
||||
|
||||
{
|
||||
@@ -242,7 +242,7 @@ namespace FEXCore::IR {
|
||||
}
|
||||
}
|
||||
|
||||
void AOTIRCaptureCache::WriteFilesWithCode(std::function<void(const fextl::string& fileid, const fextl::string& filename)> Writer) {
|
||||
void AOTIRCaptureCache::WriteFilesWithCode(const Context::AOTIRCodeFileWriterFn &Writer) {
|
||||
std::shared_lock lk(AOTIRCacheLock);
|
||||
for( const auto &Entry: AOTIRCache) {
|
||||
if (Entry.second.ContainsCode) {
|
||||
|
||||
+13
-12
@@ -89,13 +89,14 @@ namespace FEXCore::IR {
|
||||
|
||||
class AOTIRCaptureCache final {
|
||||
public:
|
||||
using WriteOutFn = std::function<void()>;
|
||||
|
||||
AOTIRCaptureCache(FEXCore::Context::ContextImpl *ctx) : CTX {ctx} {}
|
||||
|
||||
void FinalizeAOTIRCache();
|
||||
void AOTIRCaptureCacheWriteoutQueue_Flush();
|
||||
void AOTIRCaptureCacheWriteoutQueue_Append(const std::function<void()> &fn);
|
||||
void WriteFilesWithCode(std::function<void(const fextl::string& fileid, const fextl::string& filename)> Writer);
|
||||
void AOTIRCaptureCacheWriteoutQueue_Append(const WriteOutFn &fn);
|
||||
void WriteFilesWithCode(const Context::AOTIRCodeFileWriterFn &Writer);
|
||||
|
||||
struct PreGenerateIRFetchResult {
|
||||
FEXCore::IR::IRListView *IRList {};
|
||||
@@ -121,16 +122,16 @@ namespace FEXCore::IR {
|
||||
void UnloadAOTIRCacheEntry(AOTIRCacheEntry *Entry);
|
||||
|
||||
// Callbacks
|
||||
void SetAOTIRLoader(std::function<int(const fextl::string&)> CacheReader) {
|
||||
AOTIRLoader = CacheReader;
|
||||
void SetAOTIRLoader(Context::AOTIRLoaderCBFn CacheReader) {
|
||||
AOTIRLoader = std::move(CacheReader);
|
||||
}
|
||||
|
||||
void SetAOTIRWriter(std::function<fextl::unique_ptr<FEXCore::Context::AOTIRWriter>(const fextl::string&)> CacheWriter) {
|
||||
AOTIRWriter = CacheWriter;
|
||||
void SetAOTIRWriter(Context::AOTIRWriterCBFn CacheWriter) {
|
||||
AOTIRWriter = std::move(CacheWriter);
|
||||
}
|
||||
|
||||
void SetAOTIRRenamer(std::function<void(const fextl::string&)> CacheRenamer) {
|
||||
AOTIRRenamer = CacheRenamer;
|
||||
void SetAOTIRRenamer(Context::AOTIRRenamerCBFn CacheRenamer) {
|
||||
AOTIRRenamer = std::move(CacheRenamer);
|
||||
}
|
||||
|
||||
private:
|
||||
@@ -140,13 +141,13 @@ namespace FEXCore::IR {
|
||||
std::shared_mutex AOTIRCaptureCacheWriteoutLock;
|
||||
std::atomic<bool> AOTIRCaptureCacheWriteoutFlusing;
|
||||
|
||||
fextl::queue<std::function<void()>> AOTIRCaptureCacheWriteoutQueue;
|
||||
fextl::queue<WriteOutFn> AOTIRCaptureCacheWriteoutQueue;
|
||||
|
||||
FEXCore::IR::AOTCacheType AOTIRCache;
|
||||
|
||||
std::function<int(const fextl::string&)> AOTIRLoader;
|
||||
std::function<fextl::unique_ptr<FEXCore::Context::AOTIRWriter>(const fextl::string&)> AOTIRWriter;
|
||||
std::function<void(const fextl::string&)> AOTIRRenamer;
|
||||
Context::AOTIRLoaderCBFn AOTIRLoader;
|
||||
Context::AOTIRWriterCBFn AOTIRWriter;
|
||||
Context::AOTIRRenamerCBFn AOTIRRenamer;
|
||||
fextl::unordered_map<fextl::string, FEXCore::IR::AOTIRCaptureCacheEntry> AOTIRCaptureCacheMap;
|
||||
};
|
||||
}
|
||||
+47
-12
@@ -717,6 +717,13 @@
|
||||
],
|
||||
"DestSize": "std::max<uint8_t>(4, GetOpSize(_Src))"
|
||||
},
|
||||
"GPR = Abs GPR:$Src": {
|
||||
"Desc": ["Integer 2's complement absolute value",
|
||||
"Dest = std::abs(Src)",
|
||||
"Will truncate to 64 or 32bits"
|
||||
],
|
||||
"DestSize": "std::max<uint8_t>(4, GetOpSize(_Src))"
|
||||
},
|
||||
"GPR = Not GPR:$Src": {
|
||||
"Desc": ["Integer binary not",
|
||||
"op:",
|
||||
@@ -775,6 +782,14 @@
|
||||
"Desc": ["Integer binary or"
|
||||
]
|
||||
},
|
||||
"GPR = Orlshl GPR:$Src1, GPR:$Src2, u8:$BitShift": {
|
||||
"Desc": ["Integer binary or with logical shift left"
|
||||
]
|
||||
},
|
||||
"GPR = Orlshr GPR:$Src1, GPR:$Src2, u8:$BitShift": {
|
||||
"Desc": ["Integer binary or with logical shift right"
|
||||
]
|
||||
},
|
||||
"GPR = Xor GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer binary exclusive or"
|
||||
]
|
||||
@@ -787,20 +802,33 @@
|
||||
"Desc": ["Integer binary AND NOT. Performs the equivalent of Src1 & ~Src2"],
|
||||
"DestSize": "std::max<uint8_t>(4, GetOpSize(_Src1))"
|
||||
},
|
||||
"GPR = Lshl GPR:$Src1, GPR:$Src2": {
|
||||
"GPR = TestNZ u8:$Size, GPR:$Src1": {
|
||||
"Desc": ["Return NZCV for a GPR, setting N and Z accordingly and zeroing C and V"],
|
||||
"DestSize": "4"
|
||||
},
|
||||
"GPR = Lshl u8:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer logical shift left"
|
||||
],
|
||||
"DestSize": "std::max<uint8_t>(4, GetOpSize(_Src1))"
|
||||
"EmitValidation": [
|
||||
"Size >= 4"
|
||||
],
|
||||
"DestSize": "Size"
|
||||
},
|
||||
"GPR = Lshr GPR:$Src1, GPR:$Src2": {
|
||||
"GPR = Lshr u8:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer logical shift right"
|
||||
],
|
||||
"DestSize": "std::max<uint8_t>(4, GetOpSize(_Src1))"
|
||||
"EmitValidation": [
|
||||
"Size >= 4"
|
||||
],
|
||||
"DestSize": "Size"
|
||||
},
|
||||
"GPR = Ashr GPR:$Src1, GPR:$Src2": {
|
||||
"GPR = Ashr u8:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer arithmetic shift right"
|
||||
],
|
||||
"DestSize": "std::max<uint8_t>(4, GetOpSize(_Src1))"
|
||||
"EmitValidation": [
|
||||
"Size >= 4"
|
||||
],
|
||||
"DestSize": "Size"
|
||||
},
|
||||
"GPR = Ror GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer rotate right"
|
||||
@@ -869,12 +897,15 @@
|
||||
],
|
||||
"DestSize": "8"
|
||||
},
|
||||
"GPR = Select CondClass:$Cond, GPR:$Cmp1, GPR:$Cmp2, GPR:$TrueVal, GPR:$FalseVal, u8:$CompareSize": {
|
||||
"GPR = Select CondClass:$Cond, SSA:$Cmp1, SSA:$Cmp2, GPR:$TrueVal, GPR:$FalseVal, u8:$CompareSize": {
|
||||
"Desc": ["Ternary selection of GPRs",
|
||||
"op:",
|
||||
"Dest = Cmp1 <Cond> Cmp2 ? TrueVal : FalseVal"
|
||||
],
|
||||
"DestSize": "std::max<uint8_t>(4, std::max<uint8_t>(GetOpSize(_TrueVal), GetOpSize(_FalseVal)))"
|
||||
"DestSize": "std::max<uint8_t>(4, std::max<uint8_t>(GetOpSize(_TrueVal), GetOpSize(_FalseVal)))",
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Cmp1) == WalkFindRegClass($Cmp2)"
|
||||
]
|
||||
},
|
||||
"GPR = Extr GPR:$Upper, GPR:$Lower, u8:$LSB": {
|
||||
"Desc": ["Concats the two GPRs to create a value that is the size of the full two GPRs",
|
||||
@@ -1301,7 +1332,13 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
},
|
||||
|
||||
"FPR = VUABDL2 u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Desc": ["Unsigned Absolute Difference Long",
|
||||
"Using the high elements of the source vectors"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
},
|
||||
"FPR = VUShl u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, FPR:$ShiftVector": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
@@ -1405,7 +1442,7 @@
|
||||
"DestSize": "RegisterSize"
|
||||
},
|
||||
|
||||
"GPR = VPCMPESTRX u8:$GPRSize, FPR:$LHS, FPR:$RHS, GPR:$RAX, GPR:$RDX, u8:$Control": {
|
||||
"GPR = VPCMPESTRX FPR:$LHS, FPR:$RHS, GPR:$RAX, GPR:$RDX, u16:$Control": {
|
||||
"Desc": ["Performs intermediate behavior analogous to the x86 PCMPESTRI/PCMPESTRM instruction",
|
||||
"This will return the intermediate result of a PCMPESTR-type operation, but NOT the final",
|
||||
"result. This must be derived from the intermediate result",
|
||||
@@ -1414,7 +1451,6 @@
|
||||
"flags into the upper 16-bits of the 32-bit result, as these can also be derived over the",
|
||||
"course of creating the intermediate result"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "4"
|
||||
},
|
||||
"GPR = VPCMPISTRX FPR:$LHS, FPR:$RHS, u8:$Control": {
|
||||
@@ -1426,7 +1462,6 @@
|
||||
"flags into the upper 16-bits of the 32-bit result, as these can also be derived over the",
|
||||
"course of creating the intermediate result"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "4"
|
||||
}
|
||||
},
|
||||
|
||||
+8
-8
@@ -103,7 +103,7 @@ static void PrintArg(fextl::stringstream *out, IRListView const* IR, OrderedNode
|
||||
if (ArgID.IsInvalid()) {
|
||||
*out << "%Invalid";
|
||||
} else {
|
||||
*out << "%ssa" << std::dec << ArgID;
|
||||
*out << "%" << std::dec << ArgID;
|
||||
if (RAData) {
|
||||
auto PhyReg = RAData->GetNodeRegister(ArgID);
|
||||
|
||||
@@ -202,8 +202,8 @@ void Dump(fextl::stringstream *out, IRListView const* IR, IR::RegisterAllocation
|
||||
|
||||
++CurrentIndent;
|
||||
AddIndent();
|
||||
*out << "(%ssa0) " << "IRHeader ";
|
||||
*out << "%ssa" << HeaderOp->Blocks.ID() << ", ";
|
||||
*out << "(%0) " << "IRHeader ";
|
||||
*out << "%" << HeaderOp->Blocks.ID() << ", ";
|
||||
*out << "#" << std::dec << HeaderOp->BlockCount << std::endl;
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
|
||||
@@ -211,10 +211,10 @@ void Dump(fextl::stringstream *out, IRListView const* IR, IR::RegisterAllocation
|
||||
auto BlockIROp = BlockHeader->C<FEXCore::IR::IROp_CodeBlock>();
|
||||
|
||||
AddIndent();
|
||||
*out << "(%ssa" << IR->GetID(BlockNode) << ") " << "CodeBlock ";
|
||||
*out << "(%" << IR->GetID(BlockNode) << ") " << "CodeBlock ";
|
||||
|
||||
*out << "%ssa" << BlockIROp->Begin.ID() << ", ";
|
||||
*out << "%ssa" << BlockIROp->Last.ID() << std::endl;
|
||||
*out << "%" << BlockIROp->Begin.ID() << ", ";
|
||||
*out << "%" << BlockIROp->Last.ID() << std::endl;
|
||||
}
|
||||
|
||||
++CurrentIndent;
|
||||
@@ -244,7 +244,7 @@ void Dump(fextl::stringstream *out, IRListView const* IR, IR::RegisterAllocation
|
||||
NumElements /= ElementSize;
|
||||
}
|
||||
|
||||
*out << "%ssa" << std::dec << ID;
|
||||
*out << "%" << std::dec << ID;
|
||||
|
||||
if (RAData) {
|
||||
auto PhyReg = RAData->GetNodeRegister(ID);
|
||||
@@ -284,7 +284,7 @@ void Dump(fextl::stringstream *out, IRListView const* IR, IR::RegisterAllocation
|
||||
NumElements = IROp->Size / ElementSize;
|
||||
}
|
||||
|
||||
*out << "(%ssa" << std::dec << ID << ' ';
|
||||
*out << "(%" << std::dec << ID << ' ';
|
||||
*out << 'i' << std::dec << (ElementSize * 8);
|
||||
if (NumElements > 1) {
|
||||
*out << 'v' << std::dec << NumElements;
|
||||
|
||||
+1
-1
@@ -413,7 +413,7 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
}
|
||||
|
||||
// Check if we are pulling in some IR from the IR Printer
|
||||
// Prints (%ssa%d) at the start of lines without a definition
|
||||
// Prints (%%d) at the start of lines without a definition
|
||||
if (Line[0] == '(') {
|
||||
size_t DefinitionEnd = fextl::string::npos;
|
||||
if ((DefinitionEnd = Line.find_first_of(')', CurrentPos)) != fextl::string::npos) {
|
||||
|
||||
+146
-2
@@ -561,6 +561,44 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
}
|
||||
*/
|
||||
|
||||
case OP_LOADMEMTSO: {
|
||||
auto Op = IROp->CW<IR::IROp_LoadMemTSO>();
|
||||
auto AddressHeader = IREmit->GetOpHeader(Op->Addr);
|
||||
|
||||
if (Op->Class == FEXCore::IR::FPRClass && AddressHeader->Op == OP_ADD && AddressHeader->Size == 8) {
|
||||
// TODO: LRCPC3 supports a vector unscaled offset like LRCPC2.
|
||||
// Support once hardware is available to use this.
|
||||
auto [OffsetType, OffsetScale, Arg0, Arg1] = MemExtendedAddressing(IREmit, IROp->Size, AddressHeader);
|
||||
|
||||
Op->OffsetType = OffsetType;
|
||||
Op->OffsetScale = OffsetScale;
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Addr_Index, Arg0); // Addr
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Offset_Index, Arg1); // Offset
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_STOREMEMTSO: {
|
||||
auto Op = IROp->CW<IR::IROp_StoreMemTSO>();
|
||||
auto AddressHeader = IREmit->GetOpHeader(Op->Addr);
|
||||
|
||||
if (Op->Class == FEXCore::IR::FPRClass && AddressHeader->Op == OP_ADD && AddressHeader->Size == 8) {
|
||||
// TODO: LRCPC3 supports a vector unscaled offset like LRCPC2.
|
||||
// Support once hardware is available to use this.
|
||||
auto [OffsetType, OffsetScale, Arg0, Arg1] = MemExtendedAddressing(IREmit, IROp->Size, AddressHeader);
|
||||
|
||||
Op->OffsetType = OffsetType;
|
||||
Op->OffsetScale = OffsetScale;
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Addr_Index, Arg0); // Addr
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Offset_Index, Arg1); // Offset
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_LOADMEM: {
|
||||
auto Op = IROp->CW<IR::IROp_LoadMem>();
|
||||
auto AddressHeader = IREmit->GetOpHeader(Op->Addr);
|
||||
@@ -652,6 +690,20 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_TESTNZ: {
|
||||
auto Op = IROp->CW<IR::IROp_TestNZ>();
|
||||
uint64_t Constant1{};
|
||||
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1)) {
|
||||
bool N = Constant1 & (1ull << ((Op->Size * 8) - 1));
|
||||
bool Z = Constant1 == 0;
|
||||
uint32_t NZVC = (N ? (1u << 31) : 0) | (Z ? (1u << 30) : 0);
|
||||
|
||||
IREmit->ReplaceWithConstant(CodeNode, NZVC);
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_OR: {
|
||||
auto Op = IROp->CW<IR::IROp_Or>();
|
||||
uint64_t Constant1{};
|
||||
@@ -669,6 +721,32 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_ORLSHL: {
|
||||
auto Op = IROp->CW<IR::IROp_Orlshl>();
|
||||
uint64_t Constant1{};
|
||||
uint64_t Constant2{};
|
||||
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) &&
|
||||
IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
uint64_t NewConstant = Constant1 | (Constant2 << Op->BitShift);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_ORLSHR: {
|
||||
auto Op = IROp->CW<IR::IROp_Orlshr>();
|
||||
uint64_t Constant1{};
|
||||
uint64_t Constant2{};
|
||||
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) &&
|
||||
IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
uint64_t NewConstant = Constant1 | (Constant2 >> Op->BitShift);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_XOR: {
|
||||
auto Op = IROp->C<IR::IROp_Xor>();
|
||||
uint64_t Constant1{};
|
||||
@@ -694,7 +772,9 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) &&
|
||||
IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
uint64_t NewConstant = (Constant1 << Constant2) & getMask(Op);
|
||||
// Shifts mask the shift amount by 63 or 31 depending on operating size;
|
||||
uint64_t ShiftMask = IROp->Size == 8 ? 63 : 31;
|
||||
uint64_t NewConstant = (Constant1 << (Constant2 & ShiftMask)) & getMask(Op);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
Changed = true;
|
||||
}
|
||||
@@ -720,7 +800,9 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) &&
|
||||
IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
uint64_t NewConstant = (Constant1 >> Constant2) & getMask(Op);
|
||||
// Shifts mask the shift amount by 63 or 31 depending on operating size;
|
||||
uint64_t ShiftMask = IROp->Size == 8 ? 63 : 31;
|
||||
uint64_t NewConstant = (Constant1 >> (Constant2 & ShiftMask)) & getMask(Op);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
Changed = true;
|
||||
}
|
||||
@@ -775,6 +857,45 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_SBFE: {
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
uint64_t Constant;
|
||||
if (IREmit->IsValueConstant(Op->Src, &Constant)) {
|
||||
// SBFE of a constant can be converted to a constant.
|
||||
uint64_t SourceMask = (1ULL << Op->Width) - 1;
|
||||
if (Op->Width == 64)
|
||||
SourceMask = ~0ULL;
|
||||
SourceMask <<= Op->lsb;
|
||||
|
||||
int64_t NewConstant = (Constant & SourceMask) >> Op->lsb;
|
||||
NewConstant <<= 64 - Op->Width;
|
||||
NewConstant >>= 64 - Op->Width;
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_BFI: {
|
||||
auto Op = IROp->C<IR::IROp_Bfi>();
|
||||
uint64_t ConstantDest{};
|
||||
uint64_t ConstantSrc{};
|
||||
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &ConstantDest) &&
|
||||
IREmit->IsValueConstant(Op->Header.Args[1], &ConstantSrc)) {
|
||||
|
||||
uint64_t SourceMask = (1ULL << Op->Width) - 1;
|
||||
if (Op->Width == 64)
|
||||
SourceMask = ~0ULL;
|
||||
|
||||
uint64_t NewConstant = ConstantDest & ~(SourceMask << Op->lsb);
|
||||
NewConstant |= (ConstantSrc & SourceMask) << Op->lsb;
|
||||
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_MUL: {
|
||||
auto Op = IROp->C<IR::IROp_Mul>();
|
||||
uint64_t Constant1{};
|
||||
@@ -797,7 +918,25 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_SELECT: {
|
||||
auto Op = IROp->C<IR::IROp_Select>();
|
||||
uint64_t Constant1{};
|
||||
uint64_t Constant2{};
|
||||
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) &&
|
||||
IREmit->IsValueConstant(Op->Header.Args[1], &Constant2) &&
|
||||
Op->Cond == COND_EQ) {
|
||||
|
||||
Constant1 &= getMask(Op);
|
||||
Constant2 &= getMask(Op);
|
||||
|
||||
bool is_true = Constant1 == Constant2;
|
||||
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(Op->Header.Args[is_true ? 2 : 3]));
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_CONDJUMP: {
|
||||
auto Op = IROp->CW<IR::IROp_CondJump>();
|
||||
|
||||
@@ -807,6 +946,11 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
// Fold the select into the CondJump if possible. Could handle more complex cases, too.
|
||||
if (Op->Cond.Val == COND_NEQ && IREmit->IsValueConstant(Op->Cmp2, &Constant) && Constant == 0 && Select->Op == OP_SELECT) {
|
||||
|
||||
const auto SelectCmpClass = IREmit->WalkFindRegClass(Select->Args[0]);
|
||||
if (SelectCmpClass == GPRPairClass) {
|
||||
// If the comparison class is a GPRPair then don't fold the select since it isn't free.
|
||||
break;
|
||||
}
|
||||
uint64_t Constant1{};
|
||||
uint64_t Constant2{};
|
||||
|
||||
|
||||
+57
-55
@@ -69,7 +69,9 @@ namespace {
|
||||
FEXCore::IR::RegisterClassType AccessRegClass;
|
||||
uint32_t AccessOffset;
|
||||
uint8_t AccessSize;
|
||||
FEXCore::IR::OrderedNode *Node;
|
||||
///< The last value that was loaded or stored.
|
||||
FEXCore::IR::OrderedNode *ValueNode;
|
||||
///< With a store access, the store node that is doing the operation.
|
||||
FEXCore::IR::OrderedNode *StoreNode;
|
||||
};
|
||||
|
||||
@@ -464,7 +466,7 @@ ContextMemberInfo *RCLSE::FindMemberInfo(ContextInfo *ContextClassificationInfo,
|
||||
return ContextClassificationInfo->Lookup.at(Offset);
|
||||
}
|
||||
|
||||
ContextMemberInfo *RCLSE::RecordAccess(ContextMemberInfo *Info, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size, LastAccessType AccessType, FEXCore::IR::OrderedNode *Node, FEXCore::IR::OrderedNode *StoreNode) {
|
||||
ContextMemberInfo *RCLSE::RecordAccess(ContextMemberInfo *Info, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size, LastAccessType AccessType, FEXCore::IR::OrderedNode *ValueNode, FEXCore::IR::OrderedNode *StoreNode) {
|
||||
LOGMAN_THROW_AA_FMT((Offset + Size) <= (Info->Class.Offset + Info->Class.Size), "Access to context item went over member size");
|
||||
LOGMAN_THROW_AA_FMT(Info->Accessed != ACCESS_INVALID, "Tried to access invalid member");
|
||||
|
||||
@@ -480,15 +482,15 @@ ContextMemberInfo *RCLSE::RecordAccess(ContextMemberInfo *Info, FEXCore::IR::Reg
|
||||
Info->AccessRegClass = RegClass;
|
||||
Info->AccessOffset = Offset;
|
||||
Info->AccessSize = Size;
|
||||
Info->Node = Node;
|
||||
Info->ValueNode = ValueNode;
|
||||
if (StoreNode != nullptr)
|
||||
Info->StoreNode = StoreNode;
|
||||
return Info;
|
||||
}
|
||||
|
||||
ContextMemberInfo *RCLSE::RecordAccess(ContextInfo *ClassifiedInfo, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size, LastAccessType AccessType, FEXCore::IR::OrderedNode *Node, FEXCore::IR::OrderedNode *StoreNode) {
|
||||
ContextMemberInfo *RCLSE::RecordAccess(ContextInfo *ClassifiedInfo, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size, LastAccessType AccessType, FEXCore::IR::OrderedNode *ValueNode, FEXCore::IR::OrderedNode *StoreNode) {
|
||||
ContextMemberInfo *Info = FindMemberInfo(ClassifiedInfo, Offset, Size);
|
||||
return RecordAccess(Info, RegClass, Offset, Size, AccessType, Node, StoreNode);
|
||||
return RecordAccess(Info, RegClass, Offset, Size, AccessType, ValueNode, StoreNode);
|
||||
}
|
||||
|
||||
void RCLSE::CalculateControlFlowInfo(FEXCore::IR::IREmitter *IREmit) {
|
||||
@@ -546,32 +548,32 @@ void RCLSE::CalculateControlFlowInfo(FEXCore::IR::IREmitter *IREmit) {
|
||||
* @brief This pass removes redundant pairs of storecontext and loadcontext ops
|
||||
*
|
||||
* eg.
|
||||
* %ssa26 i128 = LoadMem %ssa25 i64, 0x10
|
||||
* (%%ssa27) StoreContext %ssa26 i128, 0x10, 0xb0
|
||||
* %ssa28 i128 = LoadContext 0x10, 0x90
|
||||
* %ssa29 i128 = LoadContext 0x10, 0xb0
|
||||
* %26 i128 = LoadMem %25 i64, 0x10
|
||||
* (%%27) StoreContext %26 i128, 0x10, 0xb0
|
||||
* %28 i128 = LoadContext 0x10, 0x90
|
||||
* %29 i128 = LoadContext 0x10, 0xb0
|
||||
* Converts to
|
||||
* %ssa26 i128 = LoadMem %ssa25 i64, 0x10
|
||||
* (%%ssa27) StoreContext %ssa26 i128, 0x10, 0xb0
|
||||
* %ssa28 i128 = LoadContext 0x10, 0x90
|
||||
* %26 i128 = LoadMem %25 i64, 0x10
|
||||
* (%%27) StoreContext %26 i128, 0x10, 0xb0
|
||||
* %28 i128 = LoadContext 0x10, 0x90
|
||||
*
|
||||
* eg.
|
||||
* %ssa6 i128 = LoadContext 0x10, 0x90
|
||||
* %ssa7 i128 = LoadContext 0x10, 0x90
|
||||
* %ssa8 i128 = VXor %ssa7 i128, %ssa6 i128
|
||||
* %6 i128 = LoadContext 0x10, 0x90
|
||||
* %7 i128 = LoadContext 0x10, 0x90
|
||||
* %8 i128 = VXor %7 i128, %6 i128
|
||||
* Converts to
|
||||
* %ssa6 i128 = LoadContext 0x10, 0x90
|
||||
* %ssa7 i128 = VXor %ssa6 i128, %ssa6 i128
|
||||
* %6 i128 = LoadContext 0x10, 0x90
|
||||
* %7 i128 = VXor %6 i128, %6 i128
|
||||
*
|
||||
* eg.
|
||||
* (%%ssa189) StoreContext %ssa188 i128, 0x10, 0xa0
|
||||
* %ssa190 i128 = LoadContext 0x10, 0x90
|
||||
* %ssa192 i128 = VAdd %ssa188 i128, %ssa190 i128, 0x10, 0x4
|
||||
* (%%ssa193) StoreContext %ssa192 i128, 0x10, 0xa0
|
||||
* (%%189) StoreContext %188 i128, 0x10, 0xa0
|
||||
* %190 i128 = LoadContext 0x10, 0x90
|
||||
* %192 i128 = VAdd %188 i128, %190 i128, 0x10, 0x4
|
||||
* (%%193) StoreContext %192 i128, 0x10, 0xa0
|
||||
* Converts to
|
||||
* %ssa173 i128 = LoadContext 0x10, 0x90
|
||||
* %ssa175 i128 = VAdd %ssa172 i128, %ssa173 i128, 0x10, 0x4
|
||||
* (%%ssa176) StoreContext %ssa175 i128, 0x10, 0xa0
|
||||
* %173 i128 = LoadContext 0x10, 0x90
|
||||
* %175 i128 = VAdd %172 i128, %173 i128, 0x10, 0x4
|
||||
* (%%176) StoreContext %175 i128, 0x10, 0xa0
|
||||
|
||||
*/
|
||||
bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
@@ -624,7 +626,7 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
uint32_t LastOffset = Info->AccessOffset;
|
||||
uint8_t LastSize = Info->AccessSize;
|
||||
LastAccessType LastAccess = Info->Accessed;
|
||||
OrderedNode *LastNode = Info->Node;
|
||||
OrderedNode *LastValueNode = Info->ValueNode;
|
||||
OrderedNode *LastStoreNode = Info->StoreNode;
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, CodeNode);
|
||||
|
||||
@@ -637,7 +639,7 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
if (LastClass == GPRClass) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
|
||||
uint8_t TruncateSize = IREmit->GetOpSize(LastNode);
|
||||
uint8_t TruncateSize = IREmit->GetOpSize(LastValueNode);
|
||||
|
||||
// Did store context do an implicit truncation?
|
||||
if (IREmit->GetOpSize(LastStoreNode) < TruncateSize)
|
||||
@@ -647,20 +649,20 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
if (IROp->Size < TruncateSize)
|
||||
TruncateSize = IROp->Size;
|
||||
|
||||
if (TruncateSize != IREmit->GetOpSize(LastNode)) {
|
||||
if (TruncateSize != IREmit->GetOpSize(LastValueNode)) {
|
||||
// We need to insert an explict truncation
|
||||
LastNode = IREmit->_Bfe(Info->AccessSize, TruncateSize * 8, 0, LastNode);
|
||||
LastValueNode = IREmit->_Bfe(Info->AccessSize, TruncateSize * 8, 0, LastValueNode);
|
||||
}
|
||||
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastNode);
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
} else if (LastClass == FPRClass) {
|
||||
if (LastSize == IROp->Size && LastSize == IREmit->GetOpSize(LastNode)) {
|
||||
if (LastSize == IROp->Size && LastSize == IREmit->GetOpSize(LastValueNode)) {
|
||||
if (IsFullAccess(Info->Accessed)) {
|
||||
// LoadCtx matches StoreCtx and Node Size
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastNode);
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
}
|
||||
else {
|
||||
@@ -668,37 +670,37 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
// the vector element
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
// zext to size
|
||||
LastNode = IREmit->_VMov(IROp->Size, LastNode);
|
||||
LastValueNode = IREmit->_VMov(IROp->Size, LastValueNode);
|
||||
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastNode);
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
}
|
||||
} else if (LastSize >= IROp->Size &&
|
||||
IROp->Size == IREmit->GetOpSize(LastNode)) {
|
||||
IROp->Size == IREmit->GetOpSize(LastValueNode)) {
|
||||
// LoadCtx is <= StoreCtx and Node is LoadCtx
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastNode);
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
} else if (LastSize >= IROp->Size &&
|
||||
IROp->Size < IREmit->GetOpSize(LastNode)) {
|
||||
IROp->Size < IREmit->GetOpSize(LastValueNode)) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
// trucate to size
|
||||
LastNode = IREmit->_VMov(IROp->Size, LastNode);
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastNode);
|
||||
LastValueNode = IREmit->_VMov(IROp->Size, LastValueNode);
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
} else if (LastSize >= IROp->Size &&
|
||||
IROp->Size > IREmit->GetOpSize(LastNode)) {
|
||||
IROp->Size > IREmit->GetOpSize(LastValueNode)) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
// zext to size
|
||||
LastNode = IREmit->_VMov(IROp->Size, LastNode);
|
||||
LastValueNode = IREmit->_VMov(IROp->Size, LastValueNode);
|
||||
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastNode);
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
} else {
|
||||
//fmt::print("RCLSE: Not GPR class, missed, {}, lastS: {}, S: {}, Node S: {}\n", LastClass, LastSize, IROp->Size, IREmit->GetOpSize(LastNode));
|
||||
//fmt::print("RCLSE: Not GPR class, missed, {}, lastS: {}, S: {}, Node S: {}\n", LastClass, LastSize, IROp->Size, IREmit->GetOpSize(LastValueNode));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -708,8 +710,8 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
LastOffset == Op->Offset &&
|
||||
LastSize == IROp->Size) {
|
||||
// Did we read and then read again?
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastNode);
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
@@ -753,18 +755,18 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
auto Op = IROp->CW<IR::IROp_LoadFlag>();
|
||||
auto Info = FindMemberInfo(&LocalInfo, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag, 1);
|
||||
LastAccessType LastAccess = Info->Accessed;
|
||||
OrderedNode *LastNode = Info->Node;
|
||||
OrderedNode *LastValueNode = Info->ValueNode;
|
||||
|
||||
if (IsWriteAccess(LastAccess)) { // 1 byte so always a full write
|
||||
// If the last store matches this load value then we can replace the loaded value with the previous valid one
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, LastNode);
|
||||
RecordAccess(Info, FEXCore::IR::GPRClass, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag, 1, ACCESS_READ, LastNode);
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, LastValueNode);
|
||||
RecordAccess(Info, FEXCore::IR::GPRClass, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag, 1, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
}
|
||||
else if (IsReadAccess(LastAccess)) {
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, LastNode);
|
||||
RecordAccess(Info, FEXCore::IR::GPRClass, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag, 1, ACCESS_READ, LastNode);
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, LastValueNode);
|
||||
RecordAccess(Info, FEXCore::IR::GPRClass, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag, 1, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -169,6 +169,28 @@ bool DeadStoreElimination::Run(IREmitter *IREmit) {
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
|
||||
auto ClassifyRegisterStore = [this](Info &BlockInfo, uint32_t Offset, uint8_t Size) {
|
||||
//// GPR ////
|
||||
if (IsFullGPR(Offset, Size))
|
||||
BlockInfo.gpr.writes |= GPRBit(Offset);
|
||||
else
|
||||
BlockInfo.gpr.reads |= GPRBit(Offset);
|
||||
|
||||
//// FPR ////
|
||||
if (IsTrackedWriteFPR(Offset, Size))
|
||||
BlockInfo.fpr.writes |= FPRBit(Offset, Size);
|
||||
else
|
||||
BlockInfo.fpr.reads |= FPRBit(Offset, Size);
|
||||
};
|
||||
|
||||
auto ClassifyRegisterLoad = [this](Info &BlockInfo, uint32_t Offset, uint8_t Size) {
|
||||
//// GPR ////
|
||||
BlockInfo.gpr.reads |= GPRBit(Offset);
|
||||
|
||||
//// FPR ////
|
||||
BlockInfo.fpr.reads |= FPRBit(Offset, Size);
|
||||
};
|
||||
|
||||
//// Flags ////
|
||||
if (IROp->Op == OP_STOREFLAG) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
@@ -188,43 +210,16 @@ bool DeadStoreElimination::Run(IREmitter *IREmit) {
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
BlockInfo.flag.reads |= 1UL << Op->Flag;
|
||||
} else if (IROp->Op == OP_STORECONTEXT) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
} else if (IROp->Op == OP_STOREREGISTER) {
|
||||
auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
//// GPR ////
|
||||
if (IsFullGPR(Op->Offset, IROp->Size))
|
||||
BlockInfo.gpr.writes |= GPRBit(Op->Offset);
|
||||
else
|
||||
BlockInfo.gpr.reads |= GPRBit(Op->Offset);
|
||||
|
||||
//// FPR ////
|
||||
if (IsTrackedWriteFPR(Op->Offset, IROp->Size))
|
||||
BlockInfo.fpr.writes |= FPRBit(Op->Offset, IROp->Size);
|
||||
else
|
||||
BlockInfo.fpr.reads |= FPRBit(Op->Offset, IROp->Size);
|
||||
} else if (IROp->Op == OP_STORECONTEXTINDEXED ||
|
||||
IROp->Op == OP_LOADCONTEXTINDEXED) {
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
//// GPR ////
|
||||
// We can't track through these
|
||||
BlockInfo.gpr.reads = -1;
|
||||
|
||||
//// FPR ////
|
||||
// We can't track through these
|
||||
BlockInfo.fpr.reads = -1;
|
||||
} else if (IROp->Op == OP_LOADCONTEXT) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
ClassifyRegisterStore(BlockInfo, Op->Offset, IROp->Size);
|
||||
} else if (IROp->Op == OP_LOADREGISTER) {
|
||||
auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
//// GPR ////
|
||||
BlockInfo.gpr.reads |= GPRBit(Op->Offset);
|
||||
|
||||
//// FPR ////
|
||||
BlockInfo.fpr.reads |= FPRBit(Op->Offset, IROp->Size);
|
||||
ClassifyRegisterLoad(BlockInfo, Op->Offset, IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -321,6 +316,25 @@ bool DeadStoreElimination::Run(IREmitter *IREmit) {
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
|
||||
auto RemoveDeadRegisterStore = [this](FEXCore::IR::IREmitter *IREmit, FEXCore::IR::OrderedNode *CodeNode, Info &BlockInfo, uint32_t Offset, uint8_t Size) -> bool {
|
||||
bool Changed{};
|
||||
//// GPRs ////
|
||||
// If this OP_STOREREGISTER is never read, remove it
|
||||
if (BlockInfo.gpr.kill & GPRBit(Offset)) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Changed = true;
|
||||
}
|
||||
|
||||
//// FPRs ////
|
||||
// If this OP_STOREREGISTER is never read, remove it
|
||||
if ((BlockInfo.fpr.kill & FPRBit(Offset, Size)) == FPRBit(Offset, Size) && (FPRBit(Offset, Size) != 0)) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Changed = true;
|
||||
}
|
||||
|
||||
return Changed;
|
||||
};
|
||||
|
||||
//// Flags ////
|
||||
if (IROp->Op == OP_STOREFLAG) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
@@ -332,24 +346,12 @@ bool DeadStoreElimination::Run(IREmitter *IREmit) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Changed = true;
|
||||
}
|
||||
} else if (IROp->Op == OP_STORECONTEXT) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
} else if (IROp->Op == OP_STOREREGISTER) {
|
||||
auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
//// GPRs ////
|
||||
// If this OP_STORECONTEXT is never read, remove it
|
||||
if (BlockInfo.gpr.kill & GPRBit(Op->Offset)) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Changed = true;
|
||||
}
|
||||
|
||||
//// FPRs ////
|
||||
// If this OP_STORECONTEXT is never read, remove it
|
||||
if ((BlockInfo.fpr.kill & FPRBit(Op->Offset, IROp->Size)) == FPRBit(Op->Offset, IROp->Size) && (FPRBit(Op->Offset, IROp->Size) != 0)) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Changed = true;
|
||||
}
|
||||
Changed |= RemoveDeadRegisterStore(IREmit, CodeNode, BlockInfo, Op->Offset, IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -183,7 +183,7 @@ bool IRCompaction::Run(IREmitter *IREmit) {
|
||||
|
||||
#ifndef NDEBUG
|
||||
LOGMAN_THROW_A_FMT(NewArg.Value != UINT32_MAX,
|
||||
"Tried remapping unfound node %ssa{}", OldArg);
|
||||
"Tried remapping unfound node %{}", OldArg);
|
||||
#endif
|
||||
|
||||
LocalIROp->Args[i].NodeOffset = NewArg.Value * sizeof(OrderedNode);
|
||||
|
||||
+14
-14
@@ -82,13 +82,13 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
HadError |= OpSize == 0;
|
||||
// Does the op have a destination of size 0?
|
||||
if (OpSize == 0) {
|
||||
Errors << "%ssa" << ID << ": Had destination but with no size" << std::endl;
|
||||
Errors << "%" << ID << ": Had destination but with no size" << std::endl;
|
||||
}
|
||||
|
||||
// Does the node have zero uses? Should have been DCE'd
|
||||
if (CodeNode->GetUses() == 0) {
|
||||
HadWarning |= true;
|
||||
Warnings << "%ssa" << ID << ": Destination created but had no uses" << std::endl;
|
||||
Warnings << "%" << ID << ": Destination created but had no uses" << std::endl;
|
||||
}
|
||||
|
||||
if (RAData) {
|
||||
@@ -101,20 +101,20 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
// If no register class was assigned
|
||||
if (AssignedClass == IR::InvalidClass) {
|
||||
HadError |= true;
|
||||
Errors << "%ssa" << ID << ": Had destination but with no register class assigned" << std::endl;
|
||||
Errors << "%" << ID << ": Had destination but with no register class assigned" << std::endl;
|
||||
}
|
||||
|
||||
// If no physical register was assigned
|
||||
if (PhyReg.Reg == IR::InvalidReg) {
|
||||
HadError |= true;
|
||||
Errors << "%ssa" << ID << ": Had destination but with no register assigned" << std::endl;
|
||||
Errors << "%" << ID << ": Had destination but with no register assigned" << std::endl;
|
||||
}
|
||||
|
||||
// Assigned class wasn't the expected class and it is a non-complex op
|
||||
if (AssignedClass != ExpectedClass &&
|
||||
ExpectedClass != IR::ComplexClass) {
|
||||
HadWarning |= true;
|
||||
Warnings << "%ssa" << ID << ": Destination had register class " << AssignedClass.Val << " When register class " << ExpectedClass.Val << " Was expected" << std::endl;
|
||||
Warnings << "%" << ID << ": Destination had register class " << AssignedClass.Val << " When register class " << ExpectedClass.Val << " Was expected" << std::endl;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -128,12 +128,12 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
// Was an argument defined after this node?
|
||||
if (ArgID >= ID) {
|
||||
HadError |= true;
|
||||
Errors << "%ssa" << ID << ": Arg[" << i << "] has definition after use at %ssa" << ArgID << std::endl;
|
||||
Errors << "%" << ID << ": Arg[" << i << "] has definition after use at %" << ArgID << std::endl;
|
||||
}
|
||||
|
||||
if (ArgID.IsValid() && !NodeIsLive.Get(ArgID.Value)) {
|
||||
HadError |= true;
|
||||
Errors << "%ssa" << ID << ": Arg[" << i << "] references dead %ssa" << ArgID << std::endl;
|
||||
Errors << "%" << ID << ": Arg[" << i << "] references dead %" << ArgID << std::endl;
|
||||
}
|
||||
|
||||
if (ArgID.IsValid()) {
|
||||
@@ -162,7 +162,7 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
|
||||
if (TrueTargetOp->Op != OP_CODEBLOCK) {
|
||||
HadError |= true;
|
||||
Errors << "CondJump %ssa" << ID << ": True Target Jumps to Op that isn't the begining of a block" << std::endl;
|
||||
Errors << "CondJump %" << ID << ": True Target Jumps to Op that isn't the begining of a block" << std::endl;
|
||||
}
|
||||
else {
|
||||
auto Block = OffsetToBlockMap.try_emplace(Op->TrueBlock.ID()).first;
|
||||
@@ -171,7 +171,7 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
|
||||
if (FalseTargetOp->Op != OP_CODEBLOCK) {
|
||||
HadError |= true;
|
||||
Errors << "CondJump %ssa" << ID << ": False Target Jumps to Op that isn't the begining of a block" << std::endl;
|
||||
Errors << "CondJump %" << ID << ": False Target Jumps to Op that isn't the begining of a block" << std::endl;
|
||||
}
|
||||
else {
|
||||
auto Block = OffsetToBlockMap.try_emplace(Op->FalseBlock.ID()).first;
|
||||
@@ -188,7 +188,7 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
FEXCore::IR::IROp_Header const *TargetOp = CurrentIR.GetOp<IROp_Header>(TargetNode);
|
||||
if (TargetOp->Op != OP_CODEBLOCK) {
|
||||
HadError |= true;
|
||||
Errors << "Jump %ssa" << ID << ": Jump to Op that isn't the begining of a block" << std::endl;
|
||||
Errors << "Jump %" << ID << ": Jump to Op that isn't the begining of a block" << std::endl;
|
||||
}
|
||||
else {
|
||||
auto Block = OffsetToBlockMap.try_emplace(Op->Header.Args[0].ID()).first;
|
||||
@@ -206,7 +206,7 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
size_t NumSuccessors = CurrentBlock->Successors.size();
|
||||
if (NumSuccessors > 2) {
|
||||
HadError |= true;
|
||||
Errors << "%ssa" << BlockID << " Has " << NumSuccessors << " successors which is too many" << std::endl;
|
||||
Errors << "%" << BlockID << " Has " << NumSuccessors << " successors which is too many" << std::endl;
|
||||
}
|
||||
|
||||
{
|
||||
@@ -222,7 +222,7 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
auto Op = GetOp(CodeCurrent);
|
||||
if (Op != IR::OP_ENDBLOCK) {
|
||||
HadError |= true;
|
||||
Errors << "%ssa" << BlockID << " Failed to end block with EndBlock" << std::endl;
|
||||
Errors << "%" << BlockID << " Failed to end block with EndBlock" << std::endl;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -233,7 +233,7 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
auto Op = GetOp(CodeCurrent);
|
||||
if (!IsBlockExit(Op)) {
|
||||
HadError |= true;
|
||||
Errors << "%ssa" << BlockID << " Didn't have a block exit IR op as its last instruction" << std::endl;
|
||||
Errors << "%" << BlockID << " Didn't have a block exit IR op as its last instruction" << std::endl;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -243,7 +243,7 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
auto [Node, IROp] = CurrentIR.at(IR::NodeID{i})();
|
||||
if (Node->NumUses != Uses[i] && IROp->Op != OP_CODEBLOCK && IROp->Op != OP_IRHEADER) {
|
||||
HadError |= true;
|
||||
Errors << "%ssa" << i << " Has " << Uses[i] << " Uses, but reports " << Node->NumUses << std::endl;
|
||||
Errors << "%" << i << " Has " << Uses[i] << " Uses, but reports " << Node->NumUses << std::endl;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -25,14 +25,9 @@ private:
|
||||
};
|
||||
|
||||
bool LongDivideEliminationPass::IsZeroOp(IREmitter *IREmit, OrderedNodeWrapper Arg) {
|
||||
auto IROp = IREmit->GetOpHeader(Arg);
|
||||
uint64_t Value;
|
||||
|
||||
// XOR based zero
|
||||
if (IROp->Op == OP_XOR) {
|
||||
return IROp->Args[0] == IROp->Args[1];
|
||||
}
|
||||
else if (IREmit->IsValueConstant(Arg, &Value)) {
|
||||
if (IREmit->IsValueConstant(Arg, &Value)) {
|
||||
// Zero constant based zero op
|
||||
return Value == 0;
|
||||
}
|
||||
@@ -87,8 +82,7 @@ bool LongDivideEliminationPass::Run(IREmitter *IREmit) {
|
||||
else if (IROp->Op == OP_LUDIV ||
|
||||
IROp->Op == OP_LUREM) {
|
||||
auto Op = IROp->C<IR::IROp_LUDiv>();
|
||||
// Check upper Op to see if it came from a xor zeroing op
|
||||
// XOR: Result = _Xor(Dest, Src);
|
||||
// Check upper Op to see if it came from a zeroing op
|
||||
// If it does then it we only need a 64bit UDIV
|
||||
if (IsZeroOp(IREmit, Op->Upper)) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
|
||||
@@ -47,7 +47,7 @@ bool PhiValidation::Run(IREmitter *IREmit) {
|
||||
// If we have found a non-phi IR op and then had a Phi or PhiValue value then this is a programming mistake
|
||||
// PHI values MUST be defined at the top of the block only
|
||||
HadError |= true;
|
||||
Errors << "Phi %ssa" << CurrentIR.GetID(CodeNode) << ": Was defined after non-phi operations. Which is invalid!" << std::endl;
|
||||
Errors << "Phi %" << CurrentIR.GetID(CodeNode) << ": Was defined after non-phi operations. Which is invalid!" << std::endl;
|
||||
}
|
||||
|
||||
// Check all the phi values to ensure they have the same type
|
||||
|
||||
@@ -297,28 +297,28 @@ bool RAValidation::Run(IREmitter *IREmit) {
|
||||
auto CurrentSSAAtReg = BlockRegState.Get(PhyReg);
|
||||
if (CurrentSSAAtReg == RegState::InvalidReg) {
|
||||
HadError |= true;
|
||||
Errors << fextl::fmt::format("%ssa{}: Arg[{}] unknown Reg: {}, class: {}\n", ID, i, PhyReg.Reg, PhyReg.Class);
|
||||
Errors << fextl::fmt::format("%{}: Arg[{}] unknown Reg: {}, class: {}\n", ID, i, PhyReg.Reg, PhyReg.Class);
|
||||
} else if (CurrentSSAAtReg == RegState::CorruptedPair) {
|
||||
HadError |= true;
|
||||
|
||||
auto Lower = BlockRegState.Get(PhysicalRegister(GPRClass, uint8_t(PhyReg.Reg*2) + 1));
|
||||
auto Upper = BlockRegState.Get(PhysicalRegister(GPRClass, PhyReg.Reg*2 + 1));
|
||||
|
||||
Errors << fextl::fmt::format("%ssa{}: Arg[{}] expects paired reg{} to contain %ssa{}, but it actually contains {{%ssa{}, %ssa{}}}\n",
|
||||
Errors << fextl::fmt::format("%{}: Arg[{}] expects paired reg{} to contain %{}, but it actually contains {{%{}, %{}}}\n",
|
||||
ID, i, PhyReg.Reg, ArgID, Lower, Upper);
|
||||
} else if (CurrentSSAAtReg == RegState::UninitializedValue) {
|
||||
HadError |= true;
|
||||
|
||||
Errors << fextl::fmt::format("%ssa{}: Arg[{}] expects reg{} to contain %ssa{}, but it is uninitialized\n",
|
||||
Errors << fextl::fmt::format("%{}: Arg[{}] expects reg{} to contain %{}, but it is uninitialized\n",
|
||||
ID, i, PhyReg.Reg, ArgID);
|
||||
} else if (CurrentSSAAtReg == RegState::ClobberedValue) {
|
||||
HadError |= true;
|
||||
|
||||
Errors << fextl::fmt::format("%ssa{}: Arg[{}] expects reg{} to contain %ssa{}, but contents vary depending on control flow\n",
|
||||
Errors << fextl::fmt::format("%{}: Arg[{}] expects reg{} to contain %{}, but contents vary depending on control flow\n",
|
||||
ID, i, PhyReg.Reg, ArgID);
|
||||
} else if (CurrentSSAAtReg != ArgID) {
|
||||
HadError |= true;
|
||||
Errors << fextl::fmt::format("%ssa{}: Arg[{}] expects reg{} to contain %ssa{}, but it actually contains %ssa{}\n",
|
||||
Errors << fextl::fmt::format("%{}: Arg[{}] expects reg{} to contain %{}, but it actually contains %{}\n",
|
||||
ID, i, PhyReg.Reg, ArgID, CurrentSSAAtReg);
|
||||
}
|
||||
};
|
||||
@@ -343,15 +343,15 @@ bool RAValidation::Run(IREmitter *IREmit) {
|
||||
|
||||
if (Value == RegState::UninitializedValue) {
|
||||
HadError |= true;
|
||||
Errors << fextl::fmt::format("%ssa{}: FillRegister expected %ssa{} in Slot {}, but was undefined in at least one control flow path\n",
|
||||
Errors << fextl::fmt::format("%{}: FillRegister expected %{} in Slot {}, but was undefined in at least one control flow path\n",
|
||||
ID, ExpectedValue, FillRegister->Slot);
|
||||
} else if (Value == RegState::ClobberedValue) {
|
||||
HadError |= true;
|
||||
Errors << fextl::fmt::format("%ssa{}: FillRegister expected %ssa{} in Slot {}, but contents vary depending on control flow\n",
|
||||
Errors << fextl::fmt::format("%{}: FillRegister expected %{} in Slot {}, but contents vary depending on control flow\n",
|
||||
ID, ExpectedValue, FillRegister->Slot);
|
||||
} else if (Value != ExpectedValue) {
|
||||
HadError |= true;
|
||||
Errors << fextl::fmt::format("%ssa{}: FillRegister expected %ssa{} in Slot {}, but it actually contains %ssa{}\n",
|
||||
Errors << fextl::fmt::format("%{}: FillRegister expected %{} in Slot {}, but it actually contains %{}\n",
|
||||
ID, ExpectedValue, FillRegister->Slot, Value);
|
||||
}
|
||||
break;
|
||||
|
||||
@@ -494,7 +494,7 @@ namespace {
|
||||
const auto ArgNode = Arg.ID();
|
||||
auto& ArgNodeLiveRange = LiveRanges[ArgNode.Value];
|
||||
LOGMAN_THROW_AA_FMT(ArgNodeLiveRange.Begin.Value != UINT32_MAX,
|
||||
"%ssa{} used by %ssa{} before defined?", ArgNode, Node);
|
||||
"%{} used by %{} before defined?", ArgNode, Node);
|
||||
|
||||
const auto ArgNodeBlockID = Graph->Nodes[ArgNode.Value].Head.BlockID;
|
||||
if (ArgNodeBlockID == BlockNodeID) {
|
||||
@@ -1311,7 +1311,7 @@ namespace {
|
||||
|
||||
if (!CurrentNodes.contains(InterferenceNode)) {
|
||||
InterferenceIdToSpill = InterferenceNode;
|
||||
LogMan::Msg::DFmt("Panic spilling %ssa{}, Live Range[{}, {})", InterferenceIdToSpill, InterferenceLiveRange->Begin, InterferenceLiveRange->End);
|
||||
LogMan::Msg::DFmt("Panic spilling %{}, Live Range[{}, {})", InterferenceIdToSpill, InterferenceLiveRange->Begin, InterferenceLiveRange->End);
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
@@ -1320,14 +1320,14 @@ namespace {
|
||||
|
||||
if (InterferenceIdToSpill.IsInvalid()) {
|
||||
int j = 0;
|
||||
LogMan::Msg::DFmt("node %ssa{}, was dumped in to virtual reg {}. Live Range[{}, {})",
|
||||
LogMan::Msg::DFmt("node %{}, was dumped in to virtual reg {}. Live Range[{}, {})",
|
||||
CurrentLocation, -1,
|
||||
OpLiveRange->Begin, OpLiveRange->End);
|
||||
|
||||
RegisterNode->Interferences.Iterate([&](IR::NodeID InterferenceNode) {
|
||||
auto *InterferenceLiveRange = &LiveRanges[InterferenceNode.Value];
|
||||
|
||||
LogMan::Msg::DFmt("\tInt{}: %ssa{} Remat: {} [{}, {})", j++, InterferenceNode, InterferenceLiveRange->RematCost, InterferenceLiveRange->Begin, InterferenceLiveRange->End);
|
||||
LogMan::Msg::DFmt("\tInt{}: %{} Remat: {} [{}, {})", j++, InterferenceNode, InterferenceLiveRange->RematCost, InterferenceLiveRange->Begin, InterferenceLiveRange->End);
|
||||
});
|
||||
}
|
||||
LOGMAN_THROW_A_FMT(InterferenceIdToSpill.IsValid(), "Couldn't find Node to spill");
|
||||
@@ -1395,7 +1395,7 @@ namespace {
|
||||
auto FirstUseLocation = FindFirstUse(IREmit, ConstantNode, NextIter, NodeIterator::Invalid());
|
||||
|
||||
LOGMAN_THROW_A_FMT(FirstUseLocation != IR::NodeIterator::Invalid(),
|
||||
"At %ssa{} Spilling Op %ssa{} but Failure to find op use",
|
||||
"At %{} Spilling Op %{} but Failure to find op use",
|
||||
Node, *InterferenceNode);
|
||||
|
||||
if (FirstUseLocation != IR::NodeIterator::Invalid()) {
|
||||
@@ -1454,7 +1454,7 @@ namespace {
|
||||
auto FirstUseLocation = FindFirstUse(IREmit, InterferenceOrderedNode, FirstIter, NodeIterator::Invalid());
|
||||
|
||||
LOGMAN_THROW_A_FMT(FirstUseLocation != NodeIterator::Invalid(),
|
||||
"At %ssa{} Spilling Op %ssa{} but Failure to find op use",
|
||||
"At %{} Spilling Op %{} but Failure to find op use",
|
||||
Node, *InterferenceNode);
|
||||
|
||||
if (FirstUseLocation != IR::NodeIterator::Invalid()) {
|
||||
|
||||
@@ -107,18 +107,18 @@ bool ValueDominanceValidation::Run(IREmitter *IREmit) {
|
||||
// then it must only be declared prior to this instruction
|
||||
// Eg: Valid
|
||||
// CodeBlock_1:
|
||||
// %ssa_1 = Load
|
||||
// %ssa_2 = Load
|
||||
// %ssa_3 = <Op> %ssa_1, %ssa_2
|
||||
// %_1 = Load
|
||||
// %_2 = Load
|
||||
// %_3 = <Op> %_1, %_2
|
||||
//
|
||||
// Eg: Invalid
|
||||
// CodeBlock_1:
|
||||
// %ssa_1 = Load
|
||||
// %ssa_2 = <Op> %ssa_1, %ssa_3
|
||||
// %ssa_3 = Load
|
||||
// %_1 = Load
|
||||
// %_2 = <Op> %_1, %_3
|
||||
// %_3 = Load
|
||||
if (Arg.ID() > CodeID) {
|
||||
HadError |= true;
|
||||
Errors << "Inst %ssa" << CodeID << ": Arg[" << i << "] %ssa" << Arg.ID() << " definition does not dominate this use!" << std::endl;
|
||||
Errors << "Inst %" << CodeID << ": Arg[" << i << "] %" << Arg.ID() << " definition does not dominate this use!" << std::endl;
|
||||
}
|
||||
}
|
||||
else if (Arg.ID() < BlockIROp->Begin.ID()) {
|
||||
@@ -127,21 +127,21 @@ bool ValueDominanceValidation::Run(IREmitter *IREmit) {
|
||||
|
||||
// Eg: Valid
|
||||
// CodeBlock_1:
|
||||
// %ssa_1 = Load
|
||||
// %ssa_2 = Load
|
||||
// %_1 = Load
|
||||
// %_2 = Load
|
||||
// Jump %CodeBlock_2
|
||||
//
|
||||
// CodeBlock_2:
|
||||
// %ssa_3 = <Op> %ssa_1, %ssa_2
|
||||
// %_3 = <Op> %_1, %_2
|
||||
//
|
||||
// Eg: Invalid
|
||||
// CodeBlock_1:
|
||||
// %ssa_1 = Load
|
||||
// %ssa_2 = Load
|
||||
// %_1 = Load
|
||||
// %_2 = Load
|
||||
// Jump %CodeBlock_3
|
||||
//
|
||||
// CodeBlock_2:
|
||||
// %ssa_3 = <Op> %ssa_1, %ssa_2
|
||||
// %_3 = <Op> %_1, %_2
|
||||
//
|
||||
// CodeBlock_3:
|
||||
// ...
|
||||
@@ -171,26 +171,26 @@ bool ValueDominanceValidation::Run(IREmitter *IREmit) {
|
||||
FoundPredDefine = true;
|
||||
break;
|
||||
}
|
||||
Errors << "\tChecking Pred %ssa" << CurrentIR.GetID(Pred) << std::endl;
|
||||
Errors << "\tChecking Pred %" << CurrentIR.GetID(Pred) << std::endl;
|
||||
}
|
||||
|
||||
if (!FoundPredDefine) {
|
||||
HadError |= true;
|
||||
Errors << "Inst %ssa" << CodeID << ": Arg[" << i << "] %ssa" << Arg.ID() << " definition does not dominate this use! But was defined before this block!" << std::endl;
|
||||
Errors << "Inst %" << CodeID << ": Arg[" << i << "] %" << Arg.ID() << " definition does not dominate this use! But was defined before this block!" << std::endl;
|
||||
}
|
||||
}
|
||||
else if (Arg.ID() > BlockIROp->Last.ID()) {
|
||||
// If this SSA argument is defined AFTER this block then it is just completely broken
|
||||
// Eg: Invalid
|
||||
// CodeBlock_1:
|
||||
// %ssa_1 = Load
|
||||
// %ssa_2 = <Op> %ssa_1, %ssa_3
|
||||
// %_1 = Load
|
||||
// %_2 = <Op> %_1, %_3
|
||||
// Jump %CodeBlock_2
|
||||
//
|
||||
// CodeBlock_2:
|
||||
// %ssa_3 = Load
|
||||
// %_3 = Load
|
||||
HadError |= true;
|
||||
Errors << "Inst %ssa" << CodeID << ": Arg[" << i << "] %ssa" << Arg.ID() << " definition does not dominate this use!" << std::endl;
|
||||
Errors << "Inst %" << CodeID << ": Arg[" << i << "] %" << Arg.ID() << " definition does not dominate this use!" << std::endl;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+12
@@ -340,5 +340,17 @@ namespace FEXCore::Allocator {
|
||||
::munmap(Region.Ptr, Region.Size);
|
||||
}
|
||||
}
|
||||
|
||||
void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) {
|
||||
if (Alloc64) {
|
||||
Alloc64->LockBeforeFork(Thread);
|
||||
}
|
||||
}
|
||||
|
||||
void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child) {
|
||||
if (Alloc64) {
|
||||
Alloc64->UnlockAfterFork(Thread, Child);
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
#pragma once
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::Allocator {
|
||||
void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread);
|
||||
void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child);
|
||||
}
|
||||
+17
-4
@@ -49,6 +49,19 @@ namespace Alloc::OSAllocator {
|
||||
void *Mmap(void *addr, size_t length, int prot, int flags, int fd, off_t offset) override;
|
||||
int Munmap(void *addr, size_t length) override;
|
||||
|
||||
void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) override {
|
||||
AllocationMutex.lock();
|
||||
}
|
||||
|
||||
void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child) override {
|
||||
if (Child) {
|
||||
AllocationMutex.StealAndDropActiveLocks();
|
||||
}
|
||||
else {
|
||||
AllocationMutex.unlock();
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
// Upper bound is the maximum virtual address space of the host processor
|
||||
uintptr_t UPPER_BOUND = (1ULL << 57);
|
||||
@@ -139,7 +152,7 @@ namespace Alloc::OSAllocator {
|
||||
LiveRegionListType *LiveRegions{};
|
||||
|
||||
Alloc::ForwardOnlyIntrusiveArenaAllocator *ObjectAlloc{};
|
||||
std::mutex AllocationMutex{};
|
||||
FEXCore::ForkableUniqueMutex AllocationMutex;
|
||||
void DetermineVASize();
|
||||
|
||||
LiveVMARegion *MakeRegionActive(ReservedRegionListType::iterator ReservedIterator, uint64_t UsedSize) {
|
||||
@@ -258,7 +271,7 @@ void *OSAllocator_64Bit::Mmap(void *addr, size_t length, int prot, int flags, in
|
||||
size_t NumberOfPages = length / FHU::FEX_PAGE_SIZE;
|
||||
|
||||
// This needs a mutex to be thread safe
|
||||
FEXCore::ScopedPotentialDeferredSignalWithMutex lk(AllocationMutex, TLSThread);
|
||||
FEXCore::ScopedPotentialDeferredSignalWithForkableMutex lk(AllocationMutex, TLSThread);
|
||||
|
||||
uint64_t AllocatedOffset{};
|
||||
LiveVMARegion *LiveRegion{};
|
||||
@@ -446,7 +459,7 @@ int OSAllocator_64Bit::Munmap(void *addr, size_t length) {
|
||||
}
|
||||
|
||||
// This needs a mutex to be thread safe
|
||||
FEXCore::ScopedPotentialDeferredSignalWithMutex lk(AllocationMutex, TLSThread);
|
||||
FEXCore::ScopedPotentialDeferredSignalWithForkableMutex lk(AllocationMutex, TLSThread);
|
||||
|
||||
length = FEXCore::AlignUp(length, FHU::FEX_PAGE_SIZE);
|
||||
|
||||
@@ -571,7 +584,7 @@ OSAllocator_64Bit::OSAllocator_64Bit() {
|
||||
|
||||
OSAllocator_64Bit::~OSAllocator_64Bit() {
|
||||
// This needs a mutex to be thread safe
|
||||
FEXCore::ScopedPotentialDeferredSignalWithMutex lk(AllocationMutex, TLSThread);
|
||||
FEXCore::ScopedPotentialDeferredSignalWithForkableMutex lk(AllocationMutex, TLSThread);
|
||||
|
||||
// Walk the pages and deallocate
|
||||
// First walk the live regions
|
||||
|
||||
@@ -22,6 +22,9 @@ namespace Alloc {
|
||||
|
||||
virtual void *Mmap(void *addr, size_t length, int prot, int flags, int fd, off_t offset) { return nullptr; }
|
||||
virtual int Munmap(void *addr, size_t length) { return -1; }
|
||||
|
||||
virtual void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) {}
|
||||
virtual void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child) {}
|
||||
};
|
||||
|
||||
class GlobalAllocator {
|
||||
|
||||
+11
-15
@@ -2043,8 +2043,8 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
uint32_t Size = (Instr & 0xC000'0000) >> 30;
|
||||
uint32_t AddrReg = (Instr >> 5) & 0x1F;
|
||||
uint32_t DataReg = Instr & 0x1F;
|
||||
if ((Instr & 0x3F'FF'FC'00) == 0x08'DF'FC'00 || // LDAR*
|
||||
(Instr & 0x3F'FF'FC'00) == 0x38'BF'C0'00) { // LDAPR*
|
||||
if ((Instr & LDAXR_MASK) == LDAR_INST || // LDAR*
|
||||
(Instr & LDAXR_MASK) == LDAPR_INST) { // LDAPR*
|
||||
if (ParanoidTSO) {
|
||||
if (ArchHelpers::Arm64::HandleAtomicLoad(Instr, GPRs, 0)) {
|
||||
// Skip this instruction now
|
||||
@@ -2060,15 +2060,14 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
LDR |= Size << 30;
|
||||
LDR |= AddrReg << 5;
|
||||
LDR |= DataReg;
|
||||
PC[-1] = DMB;
|
||||
PC[0] = LDR;
|
||||
PC[1] = DMB;
|
||||
PC[1] = DMB_LD; // Back-patch the half-barrier.
|
||||
ClearICache(&PC[-1], 16);
|
||||
// Back up one instruction and have another go
|
||||
return std::make_pair(true, -4);
|
||||
// With the instruction modified, now execute again.
|
||||
return std::make_pair(true, 0);
|
||||
}
|
||||
}
|
||||
else if ( (Instr & 0x3F'FF'FC'00) == 0x08'9F'FC'00) { // STLR*
|
||||
else if ( (Instr & LDAXR_MASK) == STLR_INST) { // STLR*
|
||||
if (ParanoidTSO) {
|
||||
if (ArchHelpers::Arm64::HandleAtomicStore(Instr, GPRs, 0)) {
|
||||
// Skip this instruction now
|
||||
@@ -2084,9 +2083,8 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
STR |= Size << 30;
|
||||
STR |= AddrReg << 5;
|
||||
STR |= DataReg;
|
||||
PC[-1] = DMB;
|
||||
PC[-1] = DMB; // Back-patch the half-barrier.
|
||||
PC[0] = STR;
|
||||
PC[1] = DMB;
|
||||
ClearICache(&PC[-1], 16);
|
||||
// Back up one instruction and have another go
|
||||
return std::make_pair(true, -4);
|
||||
@@ -2111,12 +2109,11 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
LDUR |= AddrReg << 5;
|
||||
LDUR |= DataReg;
|
||||
LDUR |= Instr & (0b1'1111'1111 << 9);
|
||||
PC[-1] = DMB;
|
||||
PC[0] = LDUR;
|
||||
PC[1] = DMB;
|
||||
PC[1] = DMB_LD; // Back-patch the half-barrier.
|
||||
ClearICache(&PC[-1], 16);
|
||||
// Back up one instruction and have another go
|
||||
return std::make_pair(true, -4);
|
||||
// With the instruction modified, now execute again.
|
||||
return std::make_pair(true, 0);
|
||||
}
|
||||
}
|
||||
else if ((Instr & RCPC2_MASK) == STLUR_INST) { // STLUR*
|
||||
@@ -2138,9 +2135,8 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
STUR |= AddrReg << 5;
|
||||
STUR |= DataReg;
|
||||
STUR |= Instr & (0b1'1111'1111 << 9);
|
||||
PC[-1] = DMB;
|
||||
PC[-1] = DMB; // Back-patch the half-barrier.
|
||||
PC[0] = STUR;
|
||||
PC[1] = DMB;
|
||||
ClearICache(&PC[-1], 16);
|
||||
// Back up one instruction and have another go
|
||||
return std::make_pair(true, -4);
|
||||
|
||||
+1
-1
@@ -61,7 +61,7 @@ namespace FEXCore::Telemetry {
|
||||
}
|
||||
}
|
||||
|
||||
Value &GetObject(TelemetryType Type) {
|
||||
Value &GetTelemetryValue(TelemetryType Type) {
|
||||
return TelemetryValues.at(Type);
|
||||
}
|
||||
#endif
|
||||
|
||||
Vendored
+10
-10
@@ -32,17 +32,17 @@ SSA is quite nice to work with when translating the x86-64 code to the IR, when
|
||||
* Read the python generation file to determine the extent of what it can do
|
||||
|
||||
## IR function considerations
|
||||
The first SSA node is a special case node that is considered invalid. This means %ssa0 will always be invalid for "null" node checks
|
||||
The first real SSA node also has to be a IRHeader node. This means it is safe to assume that %ssa1 will always be an IRHeader.
|
||||
The first SSA node is a special case node that is considered invalid. This means %0 will always be invalid for "null" node checks
|
||||
The first real SSA node also has to be a IRHeader node. This means it is safe to assume that %1 will always be an IRHeader.
|
||||
|
||||
|
||||
```(%%ssa1) IRHeader 0x41a9a0, %%ssa2, 5```
|
||||
```(%%1) IRHeader 0x41a9a0, %%2, 5```
|
||||
|
||||
The header provides information about that function like the entry point address.
|
||||
Additionally it also points to the first `CodeBlock` IROp
|
||||
|
||||
|
||||
```(%%ssa2) CodeBlock %%ssa7, %%ssa168, %%ssa3```
|
||||
```(%%2) CodeBlock %%7, %%168, %%3```
|
||||
|
||||
|
||||
* The `CodeBlock` Op is a jump target and must be treated as if it'll be jumped to from other blocks
|
||||
@@ -54,12 +54,12 @@ Additionally it also points to the first `CodeBlock` IROp
|
||||
### Example code block
|
||||
|
||||
```
|
||||
(%%ssa3) CodeBlock %%ssa169, %%ssa173, %%ssa4
|
||||
(%%ssa169) BeginBlock %ssa3
|
||||
%ssa170 i64 = Constant 0x41a9e1
|
||||
(%%ssa171) StoreContext %ssa170 i64, 0x8, 0x0
|
||||
(%%ssa172) ExitFunction
|
||||
(%%ssa173) EndBlock %ssa3
|
||||
(%%3) CodeBlock %%169, %%173, %%4
|
||||
(%%169) BeginBlock %3
|
||||
%170 i64 = Constant 0x41a9e1
|
||||
(%%171) StoreContext %170 i64, 0x8, 0x0
|
||||
(%%172) ExitFunction
|
||||
(%%173) EndBlock %3
|
||||
```
|
||||
|
||||
* BeginBlock points back to the CodeBlock SSA which helps with iterating across multiple blocks
|
||||
|
||||
+17
-17
@@ -17,23 +17,23 @@ Ex:
|
||||
Translates to the IR of:
|
||||
```
|
||||
BeginBlock
|
||||
%ssa8 i32 = Constant 0x1
|
||||
StoreContext 0x8, 0x8, %ssa8
|
||||
%ssa64 i32 = Constant 0x1
|
||||
StoreContext 0x8, 0x30, %ssa64
|
||||
%ssa120 i32 = Constant 0x1f
|
||||
StoreContext 0x8, 0x28, %ssa120
|
||||
%ssa176 i32 = Constant 0x1
|
||||
StoreContext 0x8, 0x20, %ssa176
|
||||
%ssa232 i64 = LoadContext 0x8, 0x8
|
||||
%ssa264 i64 = LoadContext 0x8, 0x30
|
||||
%ssa296 i64 = LoadContext 0x8, 0x28
|
||||
%ssa328 i64 = LoadContext 0x8, 0x20
|
||||
%ssa360 i64 = LoadContext 0x8, 0x58
|
||||
%ssa392 i64 = LoadContext 0x8, 0x48
|
||||
%ssa424 i64 = LoadContext 0x8, 0x50
|
||||
%ssa456 i64 = Syscall%ssa232, %ssa264, %ssa296, %ssa328, %ssa360, %ssa392, %ssa424
|
||||
StoreContext 0x8, 0x8, %ssa456
|
||||
%8 i32 = Constant 0x1
|
||||
StoreContext 0x8, 0x8, %8
|
||||
%64 i32 = Constant 0x1
|
||||
StoreContext 0x8, 0x30, %64
|
||||
%120 i32 = Constant 0x1f
|
||||
StoreContext 0x8, 0x28, %120
|
||||
%176 i32 = Constant 0x1
|
||||
StoreContext 0x8, 0x20, %176
|
||||
%232 i64 = LoadContext 0x8, 0x8
|
||||
%264 i64 = LoadContext 0x8, 0x30
|
||||
%296 i64 = LoadContext 0x8, 0x28
|
||||
%328 i64 = LoadContext 0x8, 0x20
|
||||
%360 i64 = LoadContext 0x8, 0x58
|
||||
%392 i64 = LoadContext 0x8, 0x48
|
||||
%424 i64 = LoadContext 0x8, 0x50
|
||||
%456 i64 = Syscall%232, %264, %296, %328, %360, %392, %424
|
||||
StoreContext 0x8, 0x8, %456
|
||||
BeginBlock
|
||||
EndBlock 0x1e
|
||||
ExitFunction
|
||||
|
||||
+1
-1
@@ -30,7 +30,7 @@ Large amount of x86-64 instructions load or store registers in order from the co
|
||||
We can merge these in to loadstore pair ops to improve perf
|
||||
### Function level heuristic pass
|
||||
Once we know that a function is a true full recompile we can do some additional optimizations.
|
||||
Remove any final flag stores. We know that a compiler won't pass flags past a function call boundry(It doesn't exist in the ABI)
|
||||
Remove any final flag stores. We know that a compiler won't pass flags past a function call boundary(It doesn't exist in the ABI)
|
||||
Remove any loadstores to the context mid function, only do a final store at the end of the function and do loads at the start. Which means ops just map registers directly throughout the entire function.
|
||||
### SIMD coalescing pass?
|
||||
When operating on older MMX ops(64bit SIMD) and they may end up up generating some independent ops that can be coalesced in to a 128bit op
|
||||
+43
-46
@@ -2,12 +2,16 @@
|
||||
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/EnumOperators.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/list.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
|
||||
#include <charconv>
|
||||
#include <optional>
|
||||
#include <stdint.h>
|
||||
|
||||
@@ -52,6 +56,9 @@ namespace Handler {
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
};
|
||||
|
||||
#define ENUMDEFINES
|
||||
#include <FEXCore/Config/ConfigOptions.inl>
|
||||
|
||||
enum ConfigCore {
|
||||
CONFIG_INTERPRETER,
|
||||
CONFIG_IRJIT,
|
||||
@@ -83,6 +90,42 @@ namespace Handler {
|
||||
LAYER_TOP,
|
||||
};
|
||||
|
||||
template<typename PairTypes>
|
||||
static inline fextl::string EnumParser(PairTypes const &EnumPairs, std::string_view const View) {
|
||||
uint64_t EnumMask{};
|
||||
auto Results = std::from_chars(View.data(), View.data() + View.size(), EnumMask);
|
||||
if (Results.ec == std::errc()) {
|
||||
// If the data is a valid number, just pass it through.
|
||||
return View.data();
|
||||
}
|
||||
|
||||
auto Begin = 0;
|
||||
auto End = View.find_first_of(',');
|
||||
std::string_view Option = View.substr(Begin, End);
|
||||
while (Option.size() != 0) {
|
||||
auto EnumValue = std::find_if(EnumPairs.begin(), EnumPairs.end(),
|
||||
[Option](const DisassembleConfigPair &Value) -> bool {
|
||||
return Value.first == Option;
|
||||
});
|
||||
|
||||
if (EnumValue == EnumPairs.end()) {
|
||||
LogMan::Msg::IFmt("Skipping Unknown option: {}", Option);
|
||||
}
|
||||
else {
|
||||
EnumMask |= FEXCore::ToUnderlying(EnumValue->second);
|
||||
}
|
||||
|
||||
if (End == std::string::npos) {
|
||||
break;
|
||||
}
|
||||
Begin = End + 1;
|
||||
End = View.find_first_of(',', Begin);
|
||||
Option = View.substr(Begin, End);
|
||||
}
|
||||
|
||||
return fextl::fmt::format("{}", EnumMask);
|
||||
}
|
||||
|
||||
namespace DefaultValues {
|
||||
#define P(x) x
|
||||
#define OPT_BASE(type, group, enum, json, default) extern const P(type) P(enum);
|
||||
@@ -244,50 +287,4 @@ namespace Type {
|
||||
|
||||
static void GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string> *List);
|
||||
};
|
||||
|
||||
// Application loaders
|
||||
class FEX_DEFAULT_VISIBILITY OptionMapper : public FEXCore::Config::Layer {
|
||||
public:
|
||||
explicit OptionMapper(FEXCore::Config::LayerType Layer);
|
||||
|
||||
protected:
|
||||
void MapNameToOption(const char *ConfigName, const char *ConfigString);
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Loads the global FEX config
|
||||
*
|
||||
* @return unique_ptr for that layer
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY fextl::unique_ptr<FEXCore::Config::Layer> CreateGlobalMainLayer();
|
||||
|
||||
/**
|
||||
* @brief Loads the main application config
|
||||
*
|
||||
* @param File Optional override to load a specific config file in to the main layer
|
||||
* Shouldn't be commonly used
|
||||
*
|
||||
* @return unique_ptr for that layer
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY fextl::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(fextl::string const *File = nullptr);
|
||||
|
||||
/**
|
||||
* @brief Create an application configuration loader
|
||||
*
|
||||
* @param Filename Application filename component
|
||||
* @param Global Load the global configuration or user accessible file
|
||||
*
|
||||
* @return unique_ptr for that layer
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY fextl::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const fextl::string& Filename, FEXCore::Config::LayerType Type);
|
||||
|
||||
/**
|
||||
* @brief iCreate an environment configuration loader
|
||||
*
|
||||
* @param _envp[] The environment array from main
|
||||
*
|
||||
* @return unique_ptr for that layer
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY fextl::unique_ptr<FEXCore::Config::Layer> CreateEnvironmentLayer(char *const _envp[]);
|
||||
|
||||
}
|
||||
@@ -35,6 +35,8 @@ namespace CodeSerialize {
|
||||
namespace CPU {
|
||||
struct CPUBackendFeatures {
|
||||
bool SupportsStaticRegisterAllocation = false;
|
||||
bool SupportsShiftedBitwise = false;
|
||||
bool SupportsFlags = false;
|
||||
};
|
||||
|
||||
class CPUBackend {
|
||||
@@ -89,6 +91,28 @@ namespace CPU {
|
||||
size_t Size;
|
||||
// RIP that the block's entry comes from.
|
||||
uint64_t RIP;
|
||||
|
||||
// Number of RIP entries for this JIT Code section.
|
||||
uint32_t NumberOfRIPEntries;
|
||||
|
||||
// Offset after this block to the start of the RIP entries.
|
||||
uint32_t OffsetToRIPEntries;
|
||||
};
|
||||
|
||||
// Entries that live after the JITCodeTail.
|
||||
// These entries correlate JIT code regions with guest RIP regions.
|
||||
// Using these entries FEX is able to reconstruct the guest RIP accurately when an instruction cause a signal fault.
|
||||
// Packed using 16-bit entries to ensure the size isn't too large.
|
||||
// These smaller sizes means that each entry is relative to each other instead of absolute offset from the start of the JIT block.
|
||||
// When reconstructing the RIP, each entry must be walked linearly and accumulated with the previous entries.
|
||||
// This is a trade-off between compression inside the JIT code space and execution time when reconstruction the RIP.
|
||||
// RIP reconstruction when faulting is less likely so we are requiring the accumulation.
|
||||
struct JITRIPReconstructEntries {
|
||||
// The Host PC offset from the previous entry.
|
||||
uint16_t HostPCOffset;
|
||||
|
||||
// How much to offset the RIP from the previous entry.
|
||||
uint16_t GuestRIPOffset;
|
||||
};
|
||||
|
||||
/**
|
||||
|
||||
+20
-20
@@ -86,9 +86,17 @@ namespace FEXCore::Context {
|
||||
void *VDSO_kernel_rt_sigreturn;
|
||||
};
|
||||
|
||||
using CustomCPUFactoryType = std::function<fextl::unique_ptr<FEXCore::CPU::CPUBackend> (FEXCore::Context::Context*, FEXCore::Core::InternalThreadState *Thread)>;
|
||||
using CodeRangeInvalidationFn = std::function<void(uint64_t start, uint64_t Length)>;
|
||||
|
||||
using ExitHandler = std::function<void(uint64_t ThreadId, FEXCore::Context::ExitReason)>;
|
||||
using CustomCPUFactoryType = std::function<fextl::unique_ptr<CPU::CPUBackend>(Context*, Core::InternalThreadState *Thread)>;
|
||||
using CustomIREntrypointHandler = std::function<void(uintptr_t Entrypoint, IR::IREmitter *)>;
|
||||
|
||||
using ExitHandler = std::function<void(uint64_t ThreadId, ExitReason)>;
|
||||
|
||||
using AOTIRCodeFileWriterFn = std::function<void(const fextl::string& fileid, const fextl::string& filename)>;
|
||||
using AOTIRLoaderCBFn = std::function<int(const fextl::string&)>;
|
||||
using AOTIRRenamerCBFn = std::function<void(const fextl::string&)>;
|
||||
using AOTIRWriterCBFn = std::function<fextl::unique_ptr<AOTIRWriter>(const fextl::string&)>;
|
||||
|
||||
class Context {
|
||||
public:
|
||||
@@ -225,17 +233,6 @@ namespace FEXCore::Context {
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) = 0;
|
||||
|
||||
/**
|
||||
* @brief Sets up memory regions on the guest for mirroring within the guest's VM space
|
||||
*
|
||||
* @param VirtualAddress The address we want to set to mirror a physical memory region
|
||||
* @param PhysicalAddress The physical memory region we are mapping
|
||||
* @param Size Size of the region to mirror
|
||||
*
|
||||
* @return true when successfully mapped. false if there was an error adding
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual bool AddVirtualMemoryMapping(uint64_t VirtualAddress, uint64_t PhysicalAddress, uint64_t Size) = 0;
|
||||
|
||||
/**
|
||||
* @brief Retrieves a feature struct indicating certain supported aspects from
|
||||
* the hose.
|
||||
@@ -254,7 +251,10 @@ namespace FEXCore::Context {
|
||||
FEX_DEFAULT_VISIBILITY virtual void RunThread(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void StopThread(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void DestroyThread(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void CleanupAfterFork(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
#ifndef _WIN32
|
||||
FEX_DEFAULT_VISIBILITY virtual void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) {}
|
||||
FEX_DEFAULT_VISIBILITY virtual void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child) {}
|
||||
#endif
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetSignalDelegator(FEXCore::SignalDelegator *SignalDelegation) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetSyscallHandler(FEXCore::HLE::SyscallHandler *Handler) = 0;
|
||||
|
||||
@@ -265,18 +265,18 @@ namespace FEXCore::Context {
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(const fextl::string& Name) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void UnloadAOTIRCacheEntry(FEXCore::IR::AOTIRCacheEntry *Entry) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetAOTIRLoader(std::function<int(const fextl::string&)> CacheReader) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetAOTIRWriter(std::function<fextl::unique_ptr<AOTIRWriter>(const fextl::string&)> CacheWriter) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetAOTIRRenamer(std::function<void(const fextl::string&)> CacheRenamer) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetAOTIRLoader(AOTIRLoaderCBFn CacheReader) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetAOTIRWriter(AOTIRWriterCBFn CacheWriter) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetAOTIRRenamer(AOTIRRenamerCBFn CacheRenamer) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void FinalizeAOTIRCache() = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void WriteFilesWithCode(std::function<void(const fextl::string& fileid, const fextl::string& filename)> Writer) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void WriteFilesWithCode(AOTIRCodeFileWriterFn Writer) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, std::function<void(uint64_t start, uint64_t Length)> callback) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, CodeRangeInvalidationFn callback) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void MarkMemoryShared() = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, fextl::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)> Handler, void *Creator = nullptr, void *Data = nullptr) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void *Creator = nullptr, void *Data = nullptr) = 0;
|
||||
|
||||
/**
|
||||
* @brief Allows the frontend to register its own thunk handlers independent of what is controlled in the backend.
|
||||
|
||||
@@ -29,6 +29,7 @@ class HostFeatures final {
|
||||
bool SupportsBMI2{};
|
||||
bool SupportsCLWB{};
|
||||
bool SupportsPMULL_128Bit{};
|
||||
bool SupportsCSSC{};
|
||||
|
||||
// Float exception behaviour
|
||||
bool SupportsFlushInputsToZero{};
|
||||
|
||||
+3
-11
@@ -45,8 +45,8 @@ namespace Core {
|
||||
*
|
||||
* Required to know which thread has received the signal when it occurs
|
||||
*/
|
||||
void RegisterTLSState(FEXCore::Core::InternalThreadState *Thread);
|
||||
void UninstallTLSState(FEXCore::Core::InternalThreadState *Thread);
|
||||
virtual void RegisterTLSState(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
virtual void UninstallTLSState(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
|
||||
/**
|
||||
* @brief Registers a signal handler for the host to handle a signal
|
||||
@@ -120,17 +120,9 @@ namespace Core {
|
||||
protected:
|
||||
SignalDelegatorConfig Config;
|
||||
|
||||
FEXCore::Core::InternalThreadState *GetTLSThread();
|
||||
virtual FEXCore::Core::InternalThreadState *GetTLSThread() = 0;
|
||||
virtual void HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) = 0;
|
||||
|
||||
/**
|
||||
* @brief Registers an emulated thread's object to a TLS object
|
||||
*
|
||||
* Required to know which thread has received the signal when it occurs
|
||||
*/
|
||||
virtual void RegisterFrontendTLSState(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
virtual void UninstallFrontendTLSState(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
|
||||
/**
|
||||
* @brief Registers a signal handler for the host to handle a signal
|
||||
*
|
||||
|
||||
@@ -56,6 +56,7 @@ enum X86Reg : uint32_t {
|
||||
* @{ */
|
||||
enum X86RegLocation : uint32_t {
|
||||
RFLAG_CF_LOC = 0,
|
||||
RFLAG_RESERVED_LOC = 1, // Reserved Bit, Read-as-1
|
||||
RFLAG_PF_LOC = 2,
|
||||
RFLAG_AF_LOC = 4,
|
||||
RFLAG_ZF_LOC = 6,
|
||||
@@ -73,6 +74,13 @@ enum X86RegLocation : uint32_t {
|
||||
RFLAG_VIP_LOC = 20,
|
||||
RFLAG_ID_LOC = 21,
|
||||
|
||||
// So we can implement arm64-like flag manipulaton on the interpreter/x86 jit..
|
||||
// SF/ZF/CF/OF packed into a 32-bit word, matching arm64's NZCV structure (not semantics).
|
||||
RFLAG_NZCV_LOC = 24,
|
||||
RFLAG_NZCV_1_LOC = 25,
|
||||
RFLAG_NZCV_2_LOC = 26,
|
||||
RFLAG_NZCV_3_LOC = 27,
|
||||
|
||||
// So we can share flag handling logic, we put x87 flags after RFLAGS
|
||||
X87FLAG_BASE = 32,
|
||||
X87FLAG_IE_LOC = 32,
|
||||
|
||||
+29
-5
@@ -332,8 +332,10 @@ static_assert(offsetof(OrderedNode, Header) == 0);
|
||||
static_assert(sizeof(OrderedNode) == (sizeof(OrderedNodeHeader) + sizeof(uint32_t)));
|
||||
|
||||
struct RegisterClassType final {
|
||||
uint32_t Val;
|
||||
[[nodiscard]] constexpr operator uint32_t() const {
|
||||
using value_type = uint32_t;
|
||||
|
||||
value_type Val;
|
||||
[[nodiscard]] constexpr operator value_type() const {
|
||||
return Val;
|
||||
}
|
||||
[[nodiscard]] friend constexpr bool operator==(const RegisterClassType&, const RegisterClassType&) = default;
|
||||
@@ -388,8 +390,10 @@ struct TypeDefinition final {
|
||||
static_assert(std::is_trivial_v<TypeDefinition>);
|
||||
|
||||
struct FenceType final {
|
||||
uint8_t Val;
|
||||
[[nodiscard]] constexpr operator uint8_t() const {
|
||||
using value_type = uint8_t;
|
||||
|
||||
value_type Val;
|
||||
[[nodiscard]] constexpr operator value_type() const {
|
||||
return Val;
|
||||
}
|
||||
[[nodiscard]] friend constexpr bool operator==(const FenceType&, const FenceType&) = default;
|
||||
@@ -602,7 +606,27 @@ struct fmt::formatter<FEXCore::IR::NodeID> : fmt::formatter<FEXCore::IR::NodeID:
|
||||
// Pass-through the underlying value, so IDs can
|
||||
// be formatted like any integral value.
|
||||
template <typename FormatContext>
|
||||
auto format(const FEXCore::IR::NodeID& ID, FormatContext& ctx) {
|
||||
auto format(const FEXCore::IR::NodeID& ID, FormatContext& ctx) const {
|
||||
return Base::format(ID.Value, ctx);
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct fmt::formatter<FEXCore::IR::RegisterClassType> : fmt::formatter<FEXCore::IR::RegisterClassType::value_type> {
|
||||
using Base = fmt::formatter<FEXCore::IR::RegisterClassType::value_type>;
|
||||
|
||||
template <typename FormatContext>
|
||||
auto format(const FEXCore::IR::RegisterClassType& Class, FormatContext& ctx) const {
|
||||
return Base::format(Class.Val, ctx);
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct fmt::formatter<FEXCore::IR::FenceType> : fmt::formatter<FEXCore::IR::FenceType::value_type> {
|
||||
using Base = fmt::formatter<FEXCore::IR::FenceType::value_type>;
|
||||
|
||||
template <typename FormatContext>
|
||||
auto format(const FEXCore::IR::FenceType& Fence, FormatContext& ctx) const {
|
||||
return Base::format(Fence.Val, ctx);
|
||||
}
|
||||
};
|
||||
+19
-6
@@ -93,6 +93,15 @@ friend class FEXCore::IR::PassManager;
|
||||
IRPair<IROp_StoreMemTSO> _StoreMemTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *Addr, OrderedNode *Value, uint8_t Align = 1) {
|
||||
return _StoreMemTSO(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
IRPair<IROp_Lshl> _Lshl(OrderedNode *Src1, OrderedNode *Src2) {
|
||||
return _Lshl(std::max<uint8_t>(4, GetOpSize(Src1)), Src1, Src2);
|
||||
}
|
||||
IRPair<IROp_Lshr> _Lshr(OrderedNode *Src1, OrderedNode *Src2) {
|
||||
return _Lshr(std::max<uint8_t>(4, GetOpSize(Src1)), Src1, Src2);
|
||||
}
|
||||
IRPair<IROp_Ashr> _Ashr(OrderedNode *Src1, OrderedNode *Src2) {
|
||||
return _Ashr(std::max<uint8_t>(4, GetOpSize(Src1)), Src1, Src2);
|
||||
}
|
||||
OrderedNode *Invalid() {
|
||||
return InvalidNode;
|
||||
}
|
||||
@@ -115,7 +124,7 @@ friend class FEXCore::IR::PassManager;
|
||||
|
||||
void SetJumpTarget(IR::IROp_Jump *Op, OrderedNode *Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK,
|
||||
"Tried setting Jump target to %ssa{} {}",
|
||||
"Tried setting Jump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(),
|
||||
IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
|
||||
@@ -123,7 +132,7 @@ friend class FEXCore::IR::PassManager;
|
||||
}
|
||||
void SetTrueJumpTarget(IR::IROp_CondJump *Op, OrderedNode *Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK,
|
||||
"Tried setting CondJump target to %ssa{} {}",
|
||||
"Tried setting CondJump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(),
|
||||
IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
|
||||
@@ -131,7 +140,7 @@ friend class FEXCore::IR::PassManager;
|
||||
}
|
||||
void SetFalseJumpTarget(IR::IROp_CondJump *Op, OrderedNode *Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK,
|
||||
"Tried setting CondJump target to %ssa{} {}",
|
||||
"Tried setting CondJump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(),
|
||||
IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
|
||||
@@ -140,7 +149,7 @@ friend class FEXCore::IR::PassManager;
|
||||
|
||||
void SetJumpTarget(IRPair<IROp_Jump> Op, OrderedNode *Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK,
|
||||
"Tried setting Jump target to %ssa{} {}",
|
||||
"Tried setting Jump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(),
|
||||
IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
|
||||
@@ -148,20 +157,24 @@ friend class FEXCore::IR::PassManager;
|
||||
}
|
||||
void SetTrueJumpTarget(IRPair<IROp_CondJump> Op, OrderedNode *Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK,
|
||||
"Tried setting CondJump target to %ssa{} {}",
|
||||
"Tried setting CondJump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(),
|
||||
IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
Op.first->TrueBlock.NodeOffset = Target->Wrapped(DualListData.ListBegin()).NodeOffset;
|
||||
}
|
||||
void SetFalseJumpTarget(IRPair<IROp_CondJump> Op, OrderedNode *Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK,
|
||||
"Tried setting CondJump target to %ssa{} {}",
|
||||
"Tried setting CondJump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(),
|
||||
IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
Op.first->FalseBlock.NodeOffset = Target->Wrapped(DualListData.ListBegin()).NodeOffset;
|
||||
}
|
||||
|
||||
/** @} */
|
||||
FEXCore::IR::RegisterClassType WalkFindRegClass(OrderedNodeWrapper ssa) {
|
||||
OrderedNode *RealNode = ssa.GetNode(DualListData.ListBegin());
|
||||
return WalkFindRegClass(RealNode);
|
||||
}
|
||||
|
||||
bool IsValueConstant(OrderedNodeWrapper ssa, uint64_t *Constant = nullptr) {
|
||||
OrderedNode *RealNode = ssa.GetNode(DualListData.ListBegin());
|
||||
|
||||
+2
-3
@@ -7,7 +7,6 @@
|
||||
#include <FEXCore/Utils/ThreadPoolAllocator.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <cassert>
|
||||
#include <cstddef>
|
||||
#include <cstring>
|
||||
#include <tuple>
|
||||
@@ -35,7 +34,7 @@ class DualIntrusiveAllocator {
|
||||
}
|
||||
|
||||
[[nodiscard]] void *DataAllocate(size_t Size) {
|
||||
assert(DataCheckSize(Size) &&
|
||||
LOGMAN_THROW_A_FMT(DataCheckSize(Size),
|
||||
"Ran out of space in DualIntrusiveAllocator during allocation");
|
||||
size_t NewOffset = DataCurrentOffset + Size;
|
||||
uintptr_t NewPointer = Data + DataCurrentOffset;
|
||||
@@ -44,7 +43,7 @@ class DualIntrusiveAllocator {
|
||||
}
|
||||
|
||||
[[nodiscard]] void *ListAllocate(size_t Size) {
|
||||
assert(ListCheckSize(Size) &&
|
||||
LOGMAN_THROW_A_FMT(ListCheckSize(Size),
|
||||
"Ran out of space in DualIntrusiveAllocator during allocation");
|
||||
size_t NewOffset = ListCurrentOffset + Size;
|
||||
uintptr_t NewPointer = List + ListCurrentOffset;
|
||||
|
||||
+8
-1
@@ -112,7 +112,14 @@ namespace FEXCore::Allocator {
|
||||
inline void *malloc(size_t size) { return ::malloc(size); }
|
||||
inline void *calloc(size_t n, size_t size) { return ::calloc(n, size); }
|
||||
inline void *memalign(size_t align, size_t s) { return ::memalign(align, s); }
|
||||
inline void *valloc(size_t size) { return ::valloc(size); }
|
||||
inline void *valloc(size_t size)
|
||||
{
|
||||
#ifdef __ANDROID__
|
||||
return ::aligned_alloc(4096, size);
|
||||
#else
|
||||
return ::valloc(size);
|
||||
#endif
|
||||
}
|
||||
inline int posix_memalign(void** r, size_t a, size_t s) { return ::posix_memalign(r, a, s); }
|
||||
inline void *realloc(void* ptr, size_t size) { return ::realloc(ptr, size); }
|
||||
inline void free(void* ptr) { return ::free(ptr); }
|
||||
|
||||
@@ -27,6 +27,9 @@ namespace FEXCore::ArchHelpers::Arm64 {
|
||||
|
||||
constexpr uint32_t LDAXR_MASK = 0x3F'FF'FC'00;
|
||||
constexpr uint32_t LDAXR_INST = 0x08'5F'FC'00;
|
||||
constexpr uint32_t LDAR_INST = 0x08'DF'FC'00;
|
||||
constexpr uint32_t LDAPR_INST = 0x38'BF'C0'00;
|
||||
constexpr uint32_t STLR_INST = 0x08'9F'FC'00;
|
||||
|
||||
constexpr uint32_t STLXR_MASK = 0x3F'E0'FC'00;
|
||||
constexpr uint32_t STLXR_INST = 0x08'00'FC'00;
|
||||
@@ -87,6 +90,9 @@ namespace FEXCore::ArchHelpers::Arm64 {
|
||||
constexpr uint32_t DMB = 0b1101'0101'0000'0011'0011'0000'1011'1111 |
|
||||
0b1011'0000'0000; // Inner shareable all
|
||||
|
||||
constexpr uint32_t DMB_LD = 0b1101'0101'0000'0011'0011'0000'1011'1111 |
|
||||
0b1101'0000'0000; // Inner shareable load
|
||||
|
||||
inline uint32_t GetRdReg(uint32_t Instr) {
|
||||
return (Instr >> RD_OFFSET) & REGISTER_MASK;
|
||||
}
|
||||
|
||||
+218
-15
@@ -7,10 +7,159 @@
|
||||
#include <mutex>
|
||||
#include <shared_mutex>
|
||||
#include <signal.h>
|
||||
#ifndef _WIN32
|
||||
#include <sys/syscall.h>
|
||||
#endif
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEXCore {
|
||||
#ifndef _WIN32
|
||||
// Replacement for std::mutexes to deal with unlocking issues in the face of Linux fork() semantics.
|
||||
//
|
||||
// A fork() only clones the parent's calling thread. Other threads are silently dropped, which permanently leaves any mutexes owned by them locked.
|
||||
// To address this issue, ForkableUniqueMutex and ForkableSharedMutex provide a way to forcefully remove any dangling locks and reset the mutexes to their default state.
|
||||
class ForkableUniqueMutex final {
|
||||
public:
|
||||
ForkableUniqueMutex()
|
||||
: Mutex (PTHREAD_MUTEX_INITIALIZER) {
|
||||
}
|
||||
|
||||
// Move-only type
|
||||
ForkableUniqueMutex(const ForkableUniqueMutex&) = delete;
|
||||
ForkableUniqueMutex& operator=(const ForkableUniqueMutex&) = delete;
|
||||
ForkableUniqueMutex(ForkableUniqueMutex &&rhs) = default;
|
||||
ForkableUniqueMutex& operator=(ForkableUniqueMutex &&) = default;
|
||||
|
||||
void lock() {
|
||||
[[maybe_unused]] const auto Result = pthread_mutex_lock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to lock with {}", __func__, Result);
|
||||
}
|
||||
void unlock() {
|
||||
[[maybe_unused]] const auto Result = pthread_mutex_unlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to unlock with {}", __func__, Result);
|
||||
}
|
||||
// Initialize the internal pthread object to its default initializer state.
|
||||
// Should only ever be used in the child process when a Linux fork() has occured.
|
||||
void StealAndDropActiveLocks() {
|
||||
Mutex = PTHREAD_MUTEX_INITIALIZER;
|
||||
}
|
||||
private:
|
||||
pthread_mutex_t Mutex;
|
||||
};
|
||||
|
||||
class ForkableSharedMutex final {
|
||||
public:
|
||||
ForkableSharedMutex()
|
||||
: Mutex (PTHREAD_RWLOCK_INITIALIZER) {
|
||||
}
|
||||
|
||||
// Move-only type
|
||||
ForkableSharedMutex(const ForkableSharedMutex&) = delete;
|
||||
ForkableSharedMutex& operator=(const ForkableSharedMutex&) = delete;
|
||||
ForkableSharedMutex(ForkableSharedMutex &&rhs) = default;
|
||||
ForkableSharedMutex& operator=(ForkableSharedMutex &&) = default;
|
||||
|
||||
void lock() {
|
||||
[[maybe_unused]] const auto Result = pthread_rwlock_wrlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to lock with {}", __func__, Result);
|
||||
}
|
||||
void unlock() {
|
||||
[[maybe_unused]] const auto Result = pthread_rwlock_unlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to unlock with {}", __func__, Result);
|
||||
}
|
||||
void lock_shared() {
|
||||
[[maybe_unused]] const auto Result = pthread_rwlock_rdlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to lock with {}", __func__, Result);
|
||||
}
|
||||
|
||||
void unlock_shared() {
|
||||
unlock();
|
||||
}
|
||||
|
||||
bool try_lock() {
|
||||
const auto Result = pthread_rwlock_trywrlock(&Mutex);
|
||||
return Result == 0;
|
||||
}
|
||||
|
||||
bool try_lock_shared() {
|
||||
const auto Result = pthread_rwlock_tryrdlock(&Mutex);
|
||||
return Result == 0;
|
||||
}
|
||||
// Initialize the internal pthread object to its default initializer state.
|
||||
// Should only ever be used in the child process when a Linux fork() has occured.
|
||||
void StealAndDropActiveLocks() {
|
||||
Mutex = PTHREAD_RWLOCK_INITIALIZER;
|
||||
}
|
||||
private:
|
||||
pthread_rwlock_t Mutex;
|
||||
};
|
||||
#else
|
||||
// Windows doesn't support forking, so these can be standard mutexes.
|
||||
class ForkableUniqueMutex final {
|
||||
public:
|
||||
ForkableUniqueMutex() = default;
|
||||
|
||||
// Non-moveable
|
||||
ForkableUniqueMutex(const ForkableUniqueMutex&) = delete;
|
||||
ForkableUniqueMutex& operator=(const ForkableUniqueMutex&) = delete;
|
||||
ForkableUniqueMutex(ForkableUniqueMutex &&rhs) = delete;
|
||||
ForkableUniqueMutex& operator=(ForkableUniqueMutex &&) = delete;
|
||||
|
||||
void lock() {
|
||||
Mutex.lock();
|
||||
}
|
||||
void unlock() {
|
||||
Mutex.unlock();
|
||||
}
|
||||
// Initialize the internal pthread object to its default initializer state.
|
||||
// Should only ever be used in the child process when a Linux fork() has occured.
|
||||
void StealAndDropActiveLocks() {
|
||||
LogMan::Msg::AFmt("{} is unsupported on WIN32 builds!", __func__);
|
||||
}
|
||||
private:
|
||||
std::mutex Mutex;
|
||||
};
|
||||
|
||||
class ForkableSharedMutex final {
|
||||
public:
|
||||
ForkableSharedMutex() = default;
|
||||
|
||||
// Non-moveable
|
||||
ForkableSharedMutex(const ForkableSharedMutex&) = delete;
|
||||
ForkableSharedMutex& operator=(const ForkableSharedMutex&) = delete;
|
||||
ForkableSharedMutex(ForkableSharedMutex &&rhs) = delete;
|
||||
ForkableSharedMutex& operator=(ForkableSharedMutex &&) = delete;
|
||||
|
||||
void lock() {
|
||||
Mutex.lock();
|
||||
}
|
||||
void unlock() {
|
||||
Mutex.unlock();
|
||||
}
|
||||
void lock_shared() {
|
||||
Mutex.lock_shared();
|
||||
}
|
||||
|
||||
void unlock_shared() {
|
||||
Mutex.unlock_shared();
|
||||
}
|
||||
|
||||
bool try_lock() {
|
||||
return Mutex.try_lock();
|
||||
}
|
||||
|
||||
bool try_lock_shared() {
|
||||
return Mutex.try_lock_shared();
|
||||
}
|
||||
// Initialize the internal pthread object to its default initializer state.
|
||||
// Should only ever be used in the child process when a Linux fork() has occured.
|
||||
void StealAndDropActiveLocks() {
|
||||
LogMan::Msg::AFmt("{} is unsupported on WIN32 builds!", __func__);
|
||||
}
|
||||
private:
|
||||
std::shared_mutex Mutex;
|
||||
};
|
||||
#endif
|
||||
|
||||
template<typename MutexType, void (MutexType::*lock_fn)(), void (MutexType::*unlock_fn)()>
|
||||
class ScopedDeferredSignalWithMutexBase final {
|
||||
@@ -66,10 +215,51 @@ namespace FEXCore {
|
||||
using ScopedDeferredSignalWithSharedLock = ScopedDeferredSignalWithMutexBase<std::shared_mutex, &std::shared_mutex::lock_shared, &std::shared_mutex::unlock_shared>;
|
||||
using ScopedDeferredSignalWithUniqueLock = ScopedDeferredSignalWithMutexBase<std::shared_mutex, &std::shared_mutex::lock, &std::shared_mutex::unlock>;
|
||||
|
||||
// Forkable variant
|
||||
using ScopedDeferredSignalWithForkableMutex = ScopedDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableUniqueMutex,
|
||||
&FEXCore::ForkableUniqueMutex::lock,
|
||||
&FEXCore::ForkableUniqueMutex::unlock>;
|
||||
using ScopedDeferredSignalWithForkableSharedLock = ScopedDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableSharedMutex,
|
||||
&FEXCore::ForkableSharedMutex::lock_shared,
|
||||
&FEXCore::ForkableSharedMutex::unlock_shared>;
|
||||
using ScopedDeferredSignalWithForkableUniqueLock = ScopedDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableSharedMutex,
|
||||
&FEXCore::ForkableSharedMutex::lock,
|
||||
&FEXCore::ForkableSharedMutex::unlock>;
|
||||
|
||||
class ScopedSignalMasker final {
|
||||
public:
|
||||
ScopedSignalMasker() = default;
|
||||
|
||||
void Mask(uint64_t Mask) {
|
||||
#ifndef _WIN32
|
||||
// Mask all signals, storing the original incoming mask
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &Mask, &OriginalMask, sizeof(OriginalMask));
|
||||
#endif
|
||||
}
|
||||
|
||||
// Move-only type
|
||||
ScopedSignalMasker(const ScopedSignalMasker&) = delete;
|
||||
ScopedSignalMasker& operator=(ScopedSignalMasker&) = delete;
|
||||
ScopedSignalMasker(ScopedSignalMasker &&rhs) = default;
|
||||
ScopedSignalMasker& operator=(ScopedSignalMasker &&) = default;
|
||||
|
||||
void Unmask() {
|
||||
#ifndef _WIN32
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &OriginalMask, nullptr, sizeof(OriginalMask));
|
||||
#endif
|
||||
}
|
||||
private:
|
||||
#ifndef _WIN32
|
||||
uint64_t OriginalMask{};
|
||||
#endif
|
||||
};
|
||||
|
||||
template<typename MutexType, void (MutexType::*lock_fn)(), void (MutexType::*unlock_fn)()>
|
||||
class ScopedPotentialDeferredSignalWithMutexBase final {
|
||||
public:
|
||||
|
||||
ScopedPotentialDeferredSignalWithMutexBase(MutexType &_Mutex, FEXCore::Core::InternalThreadState *Thread, uint64_t Mask = ~0ULL)
|
||||
: Mutex {&_Mutex}
|
||||
, Thread {Thread} {
|
||||
@@ -77,8 +267,7 @@ namespace FEXCore {
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Increment(1);
|
||||
}
|
||||
else {
|
||||
// Mask all signals, storing the original incoming mask
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &Mask, &OriginalMask, sizeof(OriginalMask));
|
||||
Masker.Mask(Mask);
|
||||
}
|
||||
// Lock the mutex
|
||||
(Mutex->*lock_fn)();
|
||||
@@ -102,33 +291,47 @@ namespace FEXCore {
|
||||
|
||||
if (Thread) {
|
||||
#ifdef _M_X86_64
|
||||
// Needs to be atomic so that operations can't end up getting reordered around this.
|
||||
// Without this, the refcount and the signal access could get reordered.
|
||||
auto Result = Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
|
||||
// Needs to be atomic so that operations can't end up getting reordered around this.
|
||||
// Without this, the refcount and the signal access could get reordered.
|
||||
auto Result = Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
|
||||
|
||||
// X86-64 must do an additional check around the store.
|
||||
if ((Result - 1) == 0) {
|
||||
// Must happen after the refcount store
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
|
||||
}
|
||||
// X86-64 must do an additional check around the store.
|
||||
if ((Result - 1) == 0) {
|
||||
// Must happen after the refcount store
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
|
||||
}
|
||||
#else
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
|
||||
#endif
|
||||
}
|
||||
else {
|
||||
// Unmask back to the original signal mask
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &OriginalMask, nullptr, sizeof(OriginalMask));
|
||||
Masker.Unmask();
|
||||
}
|
||||
}
|
||||
}
|
||||
private:
|
||||
MutexType *Mutex;
|
||||
uint64_t OriginalMask{};
|
||||
ScopedSignalMasker Masker;
|
||||
FEXCore::Core::InternalThreadState *Thread;
|
||||
};
|
||||
|
||||
using ScopedPotentialDeferredSignalWithMutex = ScopedPotentialDeferredSignalWithMutexBase<std::mutex, &std::mutex::lock, &std::mutex::unlock>;
|
||||
using ScopedPotentialDeferredSignalWithSharedLock = ScopedPotentialDeferredSignalWithMutexBase<std::shared_mutex, &std::shared_mutex::lock_shared, &std::shared_mutex::unlock_shared>;
|
||||
using ScopedPotentialDeferredSignalWithUniqueLock = ScopedPotentialDeferredSignalWithMutexBase<std::shared_mutex, &std::shared_mutex::lock, &std::shared_mutex::unlock>;
|
||||
|
||||
// Forkable variant
|
||||
using ScopedPotentialDeferredSignalWithForkableMutex = ScopedPotentialDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableUniqueMutex,
|
||||
&FEXCore::ForkableUniqueMutex::lock,
|
||||
&FEXCore::ForkableUniqueMutex::unlock>;
|
||||
using ScopedPotentialDeferredSignalWithForkableSharedLock = ScopedPotentialDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableSharedMutex,
|
||||
&FEXCore::ForkableSharedMutex::lock_shared,
|
||||
&FEXCore::ForkableSharedMutex::unlock_shared>;
|
||||
using ScopedPotentialDeferredSignalWithForkableUniqueLock = ScopedPotentialDeferredSignalWithMutexBase<
|
||||
FEXCore::ForkableSharedMutex,
|
||||
&FEXCore::ForkableSharedMutex::lock,
|
||||
&FEXCore::ForkableSharedMutex::unlock>;
|
||||
}
|
||||
@@ -1,6 +1,10 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <bit>
|
||||
#include <cstdint>
|
||||
#include <type_traits>
|
||||
|
||||
namespace FEXCore {
|
||||
[[nodiscard]] constexpr uint64_t AlignUp(uint64_t value, uint64_t size) {
|
||||
@@ -10,4 +14,14 @@ namespace FEXCore {
|
||||
[[nodiscard]] constexpr uint64_t AlignDown(uint64_t value, uint64_t size) {
|
||||
return value - value % size;
|
||||
}
|
||||
|
||||
// Returns the ilog2 of a power-of-2 integer.
|
||||
// Asserts in the case that the passed in integer is not a power-of-2.
|
||||
template<typename T>
|
||||
requires(std::is_unsigned_v<T>)
|
||||
[[nodiscard]] constexpr T ilog2(T Value) {
|
||||
LOGMAN_THROW_A_FMT(std::has_single_bit(Value), "ilog2 requires popcount to be one");
|
||||
return std::countr_zero(Value);
|
||||
}
|
||||
|
||||
} // namespace FEXCore
|
||||
+3
-3
@@ -41,7 +41,7 @@ namespace FEXCore::Telemetry {
|
||||
TYPE_LAST,
|
||||
};
|
||||
|
||||
Value &GetObject(TelemetryType Type);
|
||||
FEX_DEFAULT_VISIBILITY Value &GetTelemetryValue(TelemetryType Type);
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void Initialize();
|
||||
FEX_DEFAULT_VISIBILITY void Shutdown(fextl::string const &ApplicationName);
|
||||
@@ -50,8 +50,8 @@ namespace FEXCore::Telemetry {
|
||||
// This returns the internal structure to the telemetry data structures
|
||||
// One must be careful with placing these in the hot path of code execution
|
||||
// It can be fairly costly, especially in the static version where it puts barriers in the code
|
||||
#define FEXCORE_TELEMETRY_STATIC_INIT(Name, Type) static FEXCore::Telemetry::Value &Name = FEXCore::Telemetry::GetObject(FEXCore::Telemetry::Type)
|
||||
#define FEXCORE_TELEMETRY_INIT(Name, Type) FEXCore::Telemetry::Value &Name = FEXCore::Telemetry::GetObject(FEXCore::Telemetry::Type)
|
||||
#define FEXCORE_TELEMETRY_STATIC_INIT(Name, Type) static FEXCore::Telemetry::Value &Name = FEXCore::Telemetry::GetTelemetryValue(FEXCore::Telemetry::Type)
|
||||
#define FEXCORE_TELEMETRY_INIT(Name, Type) FEXCore::Telemetry::Value &Name = FEXCore::Telemetry::GetTelemetryValue(FEXCore::Telemetry::Type)
|
||||
// Telemetry ALU operations
|
||||
// These are typically 3-4 instructions depending on what you're doing
|
||||
#define FEXCORE_TELEMETRY_SET(Name, Value) Name = Value
|
||||
|
||||
+1
-1
@@ -58,7 +58,7 @@ namespace fextl::fmt {
|
||||
FMT_INLINE auto print(::fmt::format_string<T...> fmt, T&&... args)
|
||||
-> void {
|
||||
auto String = fextl::fmt::vformat(fmt, ::fmt::make_format_args(args...));
|
||||
auto f = fextl::file::File::GetStdOUT();
|
||||
auto f = FEXCore::File::File::GetStdOUT();
|
||||
f.Write(String.c_str(), String.size());
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,21 @@
|
||||
file(GLOB_RECURSE TESTS CONFIGURE_DEPENDS *.cpp)
|
||||
|
||||
set (LIBS fmt::fmt vixl Catch2::Catch2WithMain FEXCore_Base)
|
||||
foreach(TEST ${TESTS})
|
||||
get_filename_component(TEST_NAME ${TEST} NAME_WLE)
|
||||
add_executable(FEXCore_Tests_${TEST_NAME} ${TEST})
|
||||
target_link_libraries(FEXCore_Tests_${TEST_NAME} PRIVATE ${LIBS})
|
||||
target_include_directories(FEXCore_Tests_${TEST_NAME} PUBLIC "${CMAKE_CURRENT_SOURCE_DIR}/../../Source/")
|
||||
set_target_properties(FEXCore_Tests_${TEST_NAME} PROPERTIES RUNTIME_OUTPUT_DIRECTORY "${CMAKE_BINARY_DIR}/FEXCore_Tests")
|
||||
catch_discover_tests(FEXCore_Tests_${TEST_NAME} TEST_SUFFIX ".${TEST_NAME}.FEXCore_Tests")
|
||||
endforeach()
|
||||
|
||||
execute_process(COMMAND "nproc" OUTPUT_VARIABLE CORES)
|
||||
string(STRIP ${CORES} CORES)
|
||||
|
||||
add_custom_target(
|
||||
fexcore_apitests
|
||||
WORKING_DIRECTORY "${CMAKE_BINARY_DIR}/"
|
||||
USES_TERMINAL
|
||||
COMMAND "ctest" "--timeout" "302" "-j${CORES}" "-R" "\.*.FEXCore_Tests$$")
|
||||
|
||||
@@ -0,0 +1,7 @@
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <catch2/catch.hpp>
|
||||
|
||||
TEST_CASE("ILog2") {
|
||||
auto i = GENERATE(range(0, 64));
|
||||
REQUIRE(FEXCore::ilog2(1ull << i) == i);
|
||||
}
|
||||
+1
@@ -1,3 +1,4 @@
|
||||
if (NOT MINGW_BUILD)
|
||||
add_subdirectory(Emitter/)
|
||||
add_subdirectory(APITests/)
|
||||
endif()
|
||||
Loaded 100 of 177 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user