mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-07 00:00:17 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e9e5b6fb0b | ||
|
|
68cb6e61d1 | ||
|
|
5d0b2060e2 | ||
|
|
b7d05a65c7 | ||
|
|
93fe2fe06c | ||
|
|
21eb6e03c7 | ||
|
|
1b53337925 | ||
|
|
52e5b8ccd9 | ||
|
|
8c8a8c84df | ||
|
|
0d6837f1a1 | ||
|
|
7ef3cb88f9 | ||
|
|
617977357a | ||
|
|
5a53c9231b | ||
|
|
a996e5300e | ||
|
|
25b2af14fd | ||
|
|
76059949ea | ||
|
|
6e15c9c213 | ||
|
|
01fcca884b | ||
|
|
91bd3aa62a | ||
|
|
7a0119b092 | ||
|
|
fa42c1616e | ||
|
|
18783948f7 | ||
|
|
969d2e4b6a | ||
|
|
7ceaf56407 | ||
|
|
7c52375267 | ||
|
|
a4ea792d03 | ||
|
|
cec98637c9 | ||
|
|
d5026f5815 | ||
|
|
1e4456ec40 | ||
|
|
e285c7c9a0 | ||
|
|
8244c7f2a6 | ||
|
|
747c5e17f8 | ||
|
|
52ce027e9b | ||
|
|
0fb3903889 | ||
|
|
2832afd0f7 | ||
|
|
4af42477a7 | ||
|
|
4786aa479c | ||
|
|
8a049aa0c3 | ||
|
|
64bac687b5 | ||
|
|
6e05494ad0 | ||
|
|
4d13f5d97d | ||
|
|
1ad928dc2c | ||
|
|
c235ab883d | ||
|
|
4bf3a0888b | ||
|
|
6834fe32e4 | ||
|
|
d196709162 | ||
|
|
475a6a38fa | ||
|
|
0f748bd724 | ||
|
|
47bd331239 | ||
|
|
173b70d191 | ||
|
|
96aa0a844e | ||
|
|
48121cd585 | ||
|
|
52d7efda10 | ||
|
|
30ab4d3f58 | ||
|
|
f9fa0f25e2 | ||
|
|
eebcbfda96 | ||
|
|
1483ddb538 | ||
|
|
d2bca9b997 | ||
|
|
8f48021a63 | ||
|
|
1da5d7f2e6 | ||
|
|
0a5db6c404 | ||
|
|
4681061011 | ||
|
|
4e9eeb1a16 | ||
|
|
95d728b634 | ||
|
|
187e551a0c | ||
|
|
7c4e4c4409 | ||
|
|
bdd8df2100 | ||
|
|
1d1bdfb96d | ||
|
|
2fc6542d15 | ||
|
|
5e88be3c99 | ||
|
|
be52844ce0 | ||
|
|
d629884147 | ||
|
|
64c72430bf | ||
|
|
1b13e8db7d | ||
|
|
1ce0ea8f3b | ||
|
|
8557656259 | ||
|
|
c87c04db59 | ||
|
|
f5262446a0 | ||
|
|
0a1820d444 | ||
|
|
a255813e99 | ||
|
|
4c6336b560 | ||
|
|
145c7799a5 | ||
|
|
91df503e55 | ||
|
|
3c417cee84 | ||
|
|
857780b46e | ||
|
|
cb34aa3dee | ||
|
|
0d74777954 | ||
|
|
092a023900 | ||
|
|
b57dd5c3aa | ||
|
|
0bea508935 | ||
|
|
9c175da0e2 | ||
|
|
cc359f09ad | ||
|
|
dedeba0575 | ||
|
|
dc3ccbcc43 | ||
|
|
ebd9b49dbe | ||
|
|
43e21ad749 | ||
|
|
1b3beec085 | ||
|
|
c88f2e0730 | ||
|
|
1645f6e582 | ||
|
|
af35e18979 | ||
|
|
012c0ae062 | ||
|
|
dc8f063a4b | ||
|
|
759747b3c5 | ||
|
|
4c6d26f167 | ||
|
|
7562ff2308 | ||
|
|
e2baa1106b | ||
|
|
68880fd66b | ||
|
|
62eef6c548 | ||
|
|
cfdc1bda47 | ||
|
|
f9a9645ef3 | ||
|
|
8fc903cc51 | ||
|
|
c81b1c3432 | ||
|
|
db9ba16a53 | ||
|
|
0be68a54d0 | ||
|
|
6e140e9ff2 | ||
|
|
d793f7295a | ||
|
|
8f6018a1b4 | ||
|
|
0a97d4456d | ||
|
|
c2dbe8309b | ||
|
|
06d66a0188 | ||
|
|
ef254da7ca | ||
|
|
162dcf3cd0 | ||
|
|
dcc47074d1 | ||
|
|
4dfee31012 | ||
|
|
3749fa6569 | ||
|
|
94273fbf4f | ||
|
|
162bbf2937 | ||
|
|
296adf1830 | ||
|
|
a86109280a | ||
|
|
d83960d4dd | ||
|
|
bba97823a8 | ||
|
|
abf5e8c6a5 | ||
|
|
8d288300c4 | ||
|
|
0caf263c77 | ||
|
|
29dc77c44b | ||
|
|
4c1f53c1ff | ||
|
|
5821175ddb | ||
|
|
003c88e537 | ||
|
|
27a1ebc2f5 | ||
|
|
e689c6fcfa | ||
|
|
61e905a339 | ||
|
|
10fdcaa109 | ||
|
|
ba672f868c | ||
|
|
d6697fce32 | ||
|
|
9c3a843df7 | ||
|
|
8defa2b55f | ||
|
|
77c88ffe53 | ||
|
|
5b261b0d2e | ||
|
|
2ce15ddc89 | ||
|
|
3d1b55383e | ||
|
|
69deaa0976 | ||
|
|
fc72fa9e5f | ||
|
|
6baee3b7c1 | ||
|
|
362a5a019a | ||
|
|
d529a58893 | ||
|
|
c77ea3b392 | ||
|
|
20aaad15e4 | ||
|
|
6f2452e2ea | ||
|
|
b34401bb33 | ||
|
|
dff9868e9a | ||
|
|
036a196984 | ||
|
|
a7ac4fa6e4 | ||
|
|
82295b2943 | ||
|
|
eca80b6046 | ||
|
|
02fc02964b | ||
|
|
e8fcb070b3 | ||
|
|
597da88035 | ||
|
|
1e829a47fa | ||
|
|
e633ef7cf5 | ||
|
|
ff3b40400c | ||
|
|
801106faf4 | ||
|
|
536b2ed495 | ||
|
|
072f027885 | ||
|
|
754bc18813 | ||
|
|
ef257418d3 | ||
|
|
842b71cf83 | ||
|
|
8d8b64d2b7 | ||
|
|
85c6ef8097 | ||
|
|
2be16d9054 | ||
|
|
41b3c52663 | ||
|
|
f1f50b7a98 | ||
|
|
d7200e2a1e | ||
|
|
3f884fe2d0 | ||
|
|
1d453f10a0 | ||
|
|
5ebd21ca6a | ||
|
|
fb34b507a1 | ||
|
|
2c91b5cff6 | ||
|
|
79f7dcbaa5 | ||
|
|
f7b7997c77 | ||
|
|
0674dfab0a | ||
|
|
6979dc9c4e | ||
|
|
fe5f17d92e | ||
|
|
be71886990 | ||
|
|
5043e5fbc0 | ||
|
|
2acfde3cad | ||
|
|
c1eeeaf688 | ||
|
|
f2b3229a87 | ||
|
|
e20bfc0701 | ||
|
|
4caee5c9be | ||
|
|
46d3f283b8 | ||
|
|
cee5512a56 | ||
|
|
8c53a373bf | ||
|
|
f73176d5dc | ||
|
|
80ae3e632d | ||
|
|
ed75c19324 | ||
|
|
4a7fa7f2bc | ||
|
|
98f51c47fa | ||
|
|
724a8e13bf | ||
|
|
d9b52fd67d | ||
|
|
d3a2795106 | ||
|
|
3cd6c2d91a | ||
|
|
b86abfbccf | ||
|
|
ee66985ae0 | ||
|
|
daeba0625f | ||
|
|
5e6af25194 | ||
|
|
7f4528a6b0 | ||
|
|
776b7674e4 | ||
|
|
24cb2610a2 | ||
|
|
54b7a43b95 | ||
|
|
1b1e9e0fb5 | ||
|
|
ead43c6a51 | ||
|
|
e49de77225 | ||
|
|
8f4fe39b7d | ||
|
|
f250509718 | ||
|
|
6179c5a13e | ||
|
|
9e14a83442 | ||
|
|
95bfd003d2 | ||
|
|
08ca43c3c4 | ||
|
|
96b428dcaa | ||
|
|
421214e723 | ||
|
|
1a18bbb966 | ||
|
|
699c3f5762 | ||
|
|
da0a1710c0 | ||
|
|
c1205eb809 | ||
|
|
31b7cd77e9 | ||
|
|
9def04c705 | ||
|
|
68a2441e65 | ||
|
|
4ac0dec568 | ||
|
|
1d7b4bb522 | ||
|
|
22f95e627d | ||
|
|
599b64e975 | ||
|
|
5c27febbc2 | ||
|
|
58c93568f6 | ||
|
|
491e4e2c23 | ||
|
|
1ed2e24fba | ||
|
|
1fbb8bd78f | ||
|
|
b953433404 | ||
|
|
eae950be16 | ||
|
|
7e6bb04db1 | ||
|
|
68555546bc | ||
|
|
716cac35a8 | ||
|
|
9722c4c5a4 | ||
|
|
e8c0e19afc | ||
|
|
c559fec959 | ||
|
|
8d2fabe705 | ||
|
|
6455c4817a | ||
|
|
5dbd1b8dc2 | ||
|
|
7765bbc7b8 | ||
|
|
2283c73fae | ||
|
|
5fef0c29aa | ||
|
|
d387c46aab | ||
|
|
3bb7f9d6b5 | ||
|
|
70d54122b2 | ||
|
|
4e266cf9fd | ||
|
|
ddd6dbfdcc | ||
|
|
7f2557e322 | ||
|
|
810c7d926c | ||
|
|
04c325661c | ||
|
|
0121e858f2 | ||
|
|
2c3361be9e | ||
|
|
2d800b2627 | ||
|
|
55ed3e0549 | ||
|
|
55d084ebb0 | ||
|
|
457dc5dd90 | ||
|
|
98eda5e163 | ||
|
|
592935790e | ||
|
|
92d0344d6a | ||
|
|
9327435f97 | ||
|
|
573f339647 | ||
|
|
c1f18951ab | ||
|
|
8a4bfba47c | ||
|
|
69ea03f0eb | ||
|
|
462feec2a6 | ||
|
|
15f5fe658b | ||
|
|
052aa4317b | ||
|
|
debcb0e047 | ||
|
|
baf04b6a41 |
No files matched your search
@@ -172,6 +172,17 @@ jobs:
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_APITests.log || true
|
||||
|
||||
- name: FEXCore APITest tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target fexcore_apitests
|
||||
|
||||
- name: FEXCore APITest Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_FEXCoreAPITests.log || true
|
||||
|
||||
- name: ARMEmitter tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
@@ -254,6 +265,7 @@ jobs:
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
|
||||
@@ -157,6 +157,17 @@ jobs:
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_APITests.log || true
|
||||
|
||||
- name: FEXCore APITest tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target fexcore_apitests
|
||||
|
||||
- name: FEXCore APITest Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_FEXCoreAPITests.log || true
|
||||
|
||||
- name: FEXLinuxTests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
@@ -189,6 +200,7 @@ jobs:
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
|
||||
@@ -0,0 +1,96 @@
|
||||
name: Mingw build
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
pull_request:
|
||||
branches:
|
||||
- main
|
||||
|
||||
env:
|
||||
BUILD_TYPE: Debug
|
||||
FEX_ENABLEAVX: 1
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ${{ matrix.arch }}
|
||||
strategy:
|
||||
matrix:
|
||||
arch: [[self-hosted, x64, mingw], [self-hosted, ARM64, mingw]]
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
|
||||
- name: Set CC x86
|
||||
if: matrix.arch[1] == 'x64'
|
||||
run: |
|
||||
echo "CC=$HOME/llvm-mingw/build/bin/x86_64-w64-mingw32-clang" >> $GITHUB_ENV
|
||||
echo "CXX=$HOME/llvm-mingw/build/bin/x86_64-w64-mingw32-clang++" >> $GITHUB_ENV
|
||||
|
||||
- name: Set CC Arm64
|
||||
if: matrix.arch[1] == 'ARM64'
|
||||
run: |
|
||||
echo "CC=$HOME/llvm-mingw/build/bin/aarch64-w64-mingw32-clang" >> $GITHUB_ENV
|
||||
echo "CXX=$HOME/llvm-mingw/build/bin/aarch64-w64-mingw32-clang++" >> $GITHUB_ENV
|
||||
|
||||
- name: Set rootfs paths
|
||||
run: |
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
|
||||
|
||||
- name : submodule checkout
|
||||
# Need to update submodules
|
||||
run: |
|
||||
git submodule sync --recursive
|
||||
git submodule update --init --depth 1
|
||||
|
||||
- name: Clean Build Environment
|
||||
run: rm -Rf ${{runner.workspace}}/build
|
||||
|
||||
- name: Create Build Environment
|
||||
# Some projects don't allow in-source building, so create a separate build directory
|
||||
# We'll use this as our working directory for all subsequent commands
|
||||
run: cmake -E make_directory ${{runner.workspace}}/build
|
||||
|
||||
- name: Configure CMake
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DENABLE_INTERPRETER=False -DBUILD_TESTS=False -DENABLE_JEMALLOC=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the build. You can specify a specific target with "--target <NAME>"
|
||||
run: cmake --build . --config $BUILD_TYPE
|
||||
|
||||
- name: Set runner name
|
||||
if: ${{ always() }}
|
||||
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
retention-days: 3
|
||||
|
||||
@@ -112,6 +112,7 @@ jobs:
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
|
||||
@@ -0,0 +1,30 @@
|
||||
# Extracts a version from the passed in version string in the form of "<Major>.<Minor>.<Patch>".
|
||||
# If a part of the version is missing then it gets set as zero.
|
||||
# Version variables returned in:
|
||||
# ${Package}_VERSION_MAJOR
|
||||
# ${Package}_VERSION_MINOR
|
||||
# ${Package}_VERSION_PATCH
|
||||
function(version_to_variables VERSION _Package)
|
||||
string(REPLACE "." ";" VERSION_LIST "${VERSION}")
|
||||
list (LENGTH VERSION_LIST VERSION_LEN)
|
||||
if (${VERSION_LEN} GREATER 0)
|
||||
list(GET VERSION_LIST 0 VERSION_MAJOR)
|
||||
set(${_Package}_VERSION_MAJOR ${VERSION_MAJOR} PARENT_SCOPE)
|
||||
else()
|
||||
set(${_Package}_VERSION_MAJOR 0 PARENT_SCOPE)
|
||||
endif()
|
||||
|
||||
if (${VERSION_LEN} GREATER 1)
|
||||
list(GET VERSION_LIST 1 VERSION_MINOR)
|
||||
set(${_Package}_VERSION_MINOR ${VERSION_MINOR} PARENT_SCOPE)
|
||||
else()
|
||||
set(${_Package}_VERSION_MINOR 0 PARENT_SCOPE)
|
||||
endif()
|
||||
|
||||
if (${VERSION_LEN} GREATER 2)
|
||||
list(GET VERSION_LIST 2 VERSION_PATCH)
|
||||
set(${_Package}_VERSION_PATCH ${VERSION_PATCH} PARENT_SCOPE)
|
||||
else()
|
||||
set(${_Package}_VERSION_PATCH 0 PARENT_SCOPE)
|
||||
endif()
|
||||
endfunction()
|
||||
+6
-9
@@ -13,8 +13,7 @@ option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
|
||||
option(ENABLE_IWYU "Enables include what you use program" FALSE)
|
||||
option(ENABLE_LTO "Enable LTO with compilation" TRUE)
|
||||
option(ENABLE_XRAY "Enable building with LLVM X-Ray" FALSE)
|
||||
option(ENABLE_LLD "Enable linking with lld" FALSE)
|
||||
option(ENABLE_MOLD "Enable linking with mold" FALSE)
|
||||
set(USE_LINKER "" CACHE STRING "Allow overriding the linker path directly")
|
||||
option(ENABLE_ASAN "Enables Clang ASAN" FALSE)
|
||||
option(ENABLE_TSAN "Enables Clang TSAN" FALSE)
|
||||
option(ENABLE_ASSERTIONS "Enables assertions in build" FALSE)
|
||||
@@ -157,13 +156,9 @@ endif()
|
||||
|
||||
set (PTHREAD_LIB pthread)
|
||||
|
||||
if (ENABLE_LLD AND ENABLE_MOLD)
|
||||
message (FATAL_ERROR "Cannot enable both lld and mold")
|
||||
elseif (ENABLE_LLD)
|
||||
set (LD_OVERRIDE "-fuse-ld=lld")
|
||||
add_link_options(${LD_OVERRIDE})
|
||||
elseif (ENABLE_MOLD)
|
||||
add_link_options("-fuse-ld=mold")
|
||||
if (USE_LINKER)
|
||||
message(STATUS "Overriding linker to: ${USE_LINKER}")
|
||||
add_link_options("-fuse-ld=${USE_LINKER}")
|
||||
endif()
|
||||
|
||||
if (ENABLE_LIBCXX)
|
||||
@@ -267,6 +262,8 @@ if (BUILD_TESTS)
|
||||
include(Catch)
|
||||
endif()
|
||||
|
||||
# Disable fmt install
|
||||
set(FMT_INSTALL OFF)
|
||||
add_subdirectory(External/fmt/)
|
||||
|
||||
add_subdirectory(External/imgui/)
|
||||
|
||||
+66
-63
@@ -6,10 +6,10 @@
|
||||
"X11"
|
||||
],
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGL.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGL.so.1",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGL.so.1.2.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGL.so.1.7.0"
|
||||
"@PREFIX_LIB@/libGL.so",
|
||||
"@PREFIX_LIB@/libGL.so.1",
|
||||
"@PREFIX_LIB@/libGL.so.1.2.0",
|
||||
"@PREFIX_LIB@/libGL.so.1.7.0"
|
||||
]
|
||||
},
|
||||
"GLESv2": {
|
||||
@@ -18,17 +18,17 @@
|
||||
"X11"
|
||||
],
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGLESv2.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGLESv2.so.2",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGLESv2.so.2.0.0"
|
||||
"@PREFIX_LIB@/libGLESv2.so",
|
||||
"@PREFIX_LIB@/libGLESv2.so.2",
|
||||
"@PREFIX_LIB@/libGLESv2.so.2.0.0"
|
||||
]
|
||||
},
|
||||
"X11": {
|
||||
"Library": "libX11-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libX11.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libX11.so.6",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libX11.so.6.4.0"
|
||||
"@PREFIX_LIB@/libX11.so",
|
||||
"@PREFIX_LIB@/libX11.so.6",
|
||||
"@PREFIX_LIB@/libX11.so.6.4.0"
|
||||
]
|
||||
},
|
||||
"Vulkan": {
|
||||
@@ -37,8 +37,8 @@
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libvulkan.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libvulkan.so.1",
|
||||
"@PREFIX_LIB@/libvulkan.so",
|
||||
"@PREFIX_LIB@/libvulkan.so.1",
|
||||
"@HOME@/.local/share/Steam/ubuntu12_32/steam-runtime/pinned_libs_64/libvulkan.so.1"
|
||||
],
|
||||
"Comment": [
|
||||
@@ -46,139 +46,142 @@
|
||||
]
|
||||
},
|
||||
"xcb": {
|
||||
"Depends": [
|
||||
"X11"
|
||||
],
|
||||
"Library": "libxcb-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb.so.1",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb.so.1.1.0"
|
||||
"@PREFIX_LIB@/libxcb.so",
|
||||
"@PREFIX_LIB@/libxcb.so.1",
|
||||
"@PREFIX_LIB@/libxcb.so.1.1.0"
|
||||
]
|
||||
},
|
||||
"xcb-dri2": {
|
||||
"Library": "libxcb-dri2-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri2.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri2.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri2.so.0.0.0"
|
||||
"@PREFIX_LIB@/libxcb-dri2.so",
|
||||
"@PREFIX_LIB@/libxcb-dri2.so.0",
|
||||
"@PREFIX_LIB@/libxcb-dri2.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-dri3": {
|
||||
"Library": "libxcb-dri3-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri3.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri3.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri3.so.0.0.0"
|
||||
"@PREFIX_LIB@/libxcb-dri3.so",
|
||||
"@PREFIX_LIB@/libxcb-dri3.so.0",
|
||||
"@PREFIX_LIB@/libxcb-dri3.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-xfixes": {
|
||||
"Library": "libxcb-xfixes-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-xfixes.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-xfixes.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-xfixes.so.0.0.0"
|
||||
"@PREFIX_LIB@/libxcb-xfixes.so",
|
||||
"@PREFIX_LIB@/libxcb-xfixes.so.0",
|
||||
"@PREFIX_LIB@/libxcb-xfixes.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-shm": {
|
||||
"Library": "libxcb-shm-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-shm.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-shm.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-shm.so.0.0.0"
|
||||
"@PREFIX_LIB@/libxcb-shm.so",
|
||||
"@PREFIX_LIB@/libxcb-shm.so.0",
|
||||
"@PREFIX_LIB@/libxcb-shm.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-sync": {
|
||||
"Library": "libxcb-sync-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-sync.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-sync.so.1",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-sync.so.1.0.0"
|
||||
"@PREFIX_LIB@/libxcb-sync.so",
|
||||
"@PREFIX_LIB@/libxcb-sync.so.1",
|
||||
"@PREFIX_LIB@/libxcb-sync.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-randr": {
|
||||
"Library": "libxcb-randr-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-randr.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-randr.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-randr.so.0.1.0"
|
||||
"@PREFIX_LIB@/libxcb-randr.so",
|
||||
"@PREFIX_LIB@/libxcb-randr.so.0",
|
||||
"@PREFIX_LIB@/libxcb-randr.so.0.1.0"
|
||||
]
|
||||
},
|
||||
"xcb-present": {
|
||||
"Library": "libxcb-present-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-present.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-present.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-present.so.0.0.0"
|
||||
"@PREFIX_LIB@/libxcb-present.so",
|
||||
"@PREFIX_LIB@/libxcb-present.so.0",
|
||||
"@PREFIX_LIB@/libxcb-present.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-glx": {
|
||||
"Library": "libxcb-glx-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-glx.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-glx.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-glx.so.0.0.0"
|
||||
"@PREFIX_LIB@/libxcb-glx.so",
|
||||
"@PREFIX_LIB@/libxcb-glx.so.0",
|
||||
"@PREFIX_LIB@/libxcb-glx.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xshmfence": {
|
||||
"Library": "libxshmfence-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxshmfence.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxshmfence.so.1",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxshmfence.so.1.0.0"
|
||||
"@PREFIX_LIB@/libxshmfence.so",
|
||||
"@PREFIX_LIB@/libxshmfence.so.1",
|
||||
"@PREFIX_LIB@/libxshmfence.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"drm": {
|
||||
"Library": "libdrm-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libdrm.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libdrm.so.2",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libdrm.so.2.4.0"
|
||||
"@PREFIX_LIB@/libdrm.so",
|
||||
"@PREFIX_LIB@/libdrm.so.2",
|
||||
"@PREFIX_LIB@/libdrm.so.2.4.0"
|
||||
]
|
||||
},
|
||||
"asound": {
|
||||
"Library": "libasound-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libasound.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libasound.so.2",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libasound.so.2.0.0"
|
||||
"@PREFIX_LIB@/libasound.so",
|
||||
"@PREFIX_LIB@/libasound.so.2",
|
||||
"@PREFIX_LIB@/libasound.so.2.0.0"
|
||||
]
|
||||
},
|
||||
"Xrender": {
|
||||
"Library": "libXrender-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXrender.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXrender.so.1",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXrender.so.1.3.0"
|
||||
"@PREFIX_LIB@/libXrender.so",
|
||||
"@PREFIX_LIB@/libXrender.so.1",
|
||||
"@PREFIX_LIB@/libXrender.so.1.3.0"
|
||||
]
|
||||
},
|
||||
"Xext": {
|
||||
"Library": "libXext-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXext.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXext.so.6",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXext.so.6.4.0"
|
||||
"@PREFIX_LIB@/libXext.so",
|
||||
"@PREFIX_LIB@/libXext.so.6",
|
||||
"@PREFIX_LIB@/libXext.so.6.4.0"
|
||||
]
|
||||
},
|
||||
"Xfixes": {
|
||||
"Library": "libXfixes-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXfixes.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXfixes.so.3",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXfixes.so.3.1.0"
|
||||
"@PREFIX_LIB@/libXfixes.so",
|
||||
"@PREFIX_LIB@/libXfixes.so.3",
|
||||
"@PREFIX_LIB@/libXfixes.so.3.1.0"
|
||||
]
|
||||
},
|
||||
"OpenCL": {
|
||||
"Library" : "libOpenCL-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libOpenCL.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libOpenCL.so.1",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libOpenCL.so.1.0.0"
|
||||
"@PREFIX_LIB@/libOpenCL.so",
|
||||
"@PREFIX_LIB@/libOpenCL.so.1",
|
||||
"@PREFIX_LIB@/libOpenCL.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"WaylandClient": {
|
||||
"Library" : "libwayland-client-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libwayland-client.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libwayland-client.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libwayland-client.so.0.20.0"
|
||||
"@PREFIX_LIB@/libwayland-client.so",
|
||||
"@PREFIX_LIB@/libwayland-client.so.0",
|
||||
"@PREFIX_LIB@/libwayland-client.so.0.20.0"
|
||||
]
|
||||
},
|
||||
"":{}
|
||||
|
||||
+70
-5
@@ -27,7 +27,9 @@ def print_header():
|
||||
#ifndef OPT_STRARRAY
|
||||
#define OPT_STRARRAY(group, enum, json, default) OPT_BASE(fextl::string, group, enum, json, default)
|
||||
#endif
|
||||
|
||||
#ifndef OPT_STRENUM
|
||||
#define OPT_STRENUM(group, enum, json, default) OPT_BASE(uint64_t, group, enum, json, default)
|
||||
#endif
|
||||
'''
|
||||
output_file.write(header)
|
||||
|
||||
@@ -40,6 +42,7 @@ def print_tail():
|
||||
#undef OPT_UINT64
|
||||
#undef OPT_STR
|
||||
#undef OPT_STRARRAY
|
||||
#undef OPT_STRENUM
|
||||
'''
|
||||
output_file.write(tail)
|
||||
|
||||
@@ -127,12 +130,13 @@ def print_man_options(options):
|
||||
short = op_vals["ShortArg"]
|
||||
|
||||
default = op_vals["Default"]
|
||||
value_type = op_vals["Type"]
|
||||
|
||||
# Textual default rather than enum based
|
||||
if ("TextDefault" in op_vals):
|
||||
default = op_vals["TextDefault"]
|
||||
|
||||
if (op_vals["Type"] == "str" or op_vals["Type"] == "strarray"):
|
||||
if (value_type == "str" or value_type == "strarray" or value_type == "strenum"):
|
||||
# Wrap the string argument in quotes
|
||||
default = "'" + default + "'"
|
||||
print_man_option(
|
||||
@@ -141,6 +145,12 @@ def print_man_options(options):
|
||||
op_vals["Desc"],
|
||||
default
|
||||
)
|
||||
if (value_type == "strenum"):
|
||||
Enums = op_vals["Enums"]
|
||||
output_man.write("\\fBAvailable Options:\\fR\n")
|
||||
for enum_op_key, enum_op_vals in Enums.items():
|
||||
output_man.write("{}, ".format(enum_op_vals))
|
||||
output_man.write("\n")
|
||||
|
||||
output_man.write(".El\n")
|
||||
|
||||
@@ -150,12 +160,13 @@ def print_man_environment(options):
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
default = op_vals["Default"]
|
||||
value_type = op_vals["Type"]
|
||||
|
||||
# Textual default rather than enum based
|
||||
if ("TextDefault" in op_vals):
|
||||
default = op_vals["TextDefault"]
|
||||
|
||||
if (op_vals["Type"] == "str" or op_vals["Type"] == "strarray"):
|
||||
if (value_type == "str" or value_type == "strarray" or value_type == "strenum"):
|
||||
# Wrap the string argument in quotes
|
||||
default = "'" + default + "'"
|
||||
print_man_env_option(
|
||||
@@ -165,6 +176,13 @@ def print_man_environment(options):
|
||||
False
|
||||
)
|
||||
|
||||
if (value_type == "strenum"):
|
||||
Enums = op_vals["Enums"]
|
||||
output_man.write("\\fBAvailable Options:\\fR\n")
|
||||
for enum_op_key, enum_op_vals in Enums.items():
|
||||
output_man.write("{}, ".format(enum_op_vals))
|
||||
output_man.write("\n")
|
||||
|
||||
print_man_environment_tail()
|
||||
output_man.write(".El\n")
|
||||
|
||||
@@ -334,7 +352,7 @@ def print_argloader_options(options):
|
||||
for op_key, op_vals in group_vals.items():
|
||||
default = op_vals["Default"]
|
||||
|
||||
if (op_vals["Type"] == "str" or op_vals["Type"] == "strarray"):
|
||||
if (op_vals["Type"] == "str" or op_vals["Type"] == "strarray" or op_vals["Type"] == "strenum"):
|
||||
# Wrap the string argument in quotes
|
||||
default = "\"" + default + "\""
|
||||
|
||||
@@ -382,7 +400,10 @@ def print_parse_argloader_options(options):
|
||||
# boolean values need a decimal specifier. Otherwise fmt prints strings.
|
||||
conversion_func = "fextl::fmt::format(\"{:d}\", "
|
||||
|
||||
if (value_type == "strarray"):
|
||||
if (value_type == "strenum"):
|
||||
output_argloader.write("\tfextl::string UserValue = Options[\"{0}\"];\n".format(op_key))
|
||||
output_argloader.write("\tSet(FEXCore::Config::ConfigOption::CONFIG_{}, FEXCore::Config::EnumParser(FEXCore::Config::{}_EnumPairs, UserValue));\n".format(op_key.upper(), op_key, op_key))
|
||||
elif (value_type == "strarray"):
|
||||
# these need a bit more help
|
||||
output_argloader.write("\tauto Array = Options.all(\"{0}\");\n".format(op_key))
|
||||
output_argloader.write("\tfor (auto iter = Array.begin(); iter != Array.end(); ++iter) {\n")
|
||||
@@ -407,6 +428,12 @@ def print_parse_envloader_options(options):
|
||||
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
value_type = op_vals["Type"]
|
||||
if (value_type == "strenum"):
|
||||
output_argloader.write("else if (Key == \"FEX_{0}\") {{\n".format(op_key.upper()))
|
||||
output_argloader.write("Value = FEXCore::Config::EnumParser(FEXCore::Config::{}_EnumPairs, Value);\n".format(op_key, op_key))
|
||||
output_argloader.write("}\n")
|
||||
|
||||
if ("ArgumentHandler" in op_vals):
|
||||
conversion_func = "FEXCore::Config::Handler::{0}".format(op_vals["ArgumentHandler"])
|
||||
output_argloader.write("else if (Key == \"FEX_{0}\") {{\n".format(op_key.upper()))
|
||||
@@ -414,6 +441,41 @@ def print_parse_envloader_options(options):
|
||||
output_argloader.write("}\n")
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
def print_parse_enum_options(options):
|
||||
output_argloader.write("#ifdef ENUMDEFINES\n")
|
||||
output_argloader.write("#undef ENUMDEFINES\n")
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
if (op_vals["Type"] == "strenum"):
|
||||
output_argloader.write("enum {} : uint64_t {{\n".format(op_key))
|
||||
Enums = op_vals["Enums"]
|
||||
i = 0
|
||||
# Always have an OFF.
|
||||
output_argloader.write("\tOFF = 0,\n")
|
||||
for enum_op_key, enum_op_vals in Enums.items():
|
||||
output_argloader.write("\t{} = 1ULL << {},\n".format(enum_op_key.upper(), i))
|
||||
i += 1
|
||||
|
||||
output_argloader.write("};\n")
|
||||
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
if (op_vals["Type"] == "strenum"):
|
||||
Enums = op_vals["Enums"]
|
||||
|
||||
output_argloader.write("using {}ConfigPair = std::pair<std::string_view, FEXCore::Config::{}>;\n".format(op_key, op_key))
|
||||
output_argloader.write("constexpr static std::array<{}ConfigPair, {}> {}_EnumPairs = {{{{\n".format(op_key, len(Enums) + 1, op_key))
|
||||
i = 0
|
||||
# Always have an OFF.
|
||||
output_argloader.write("\t{{ \"off\", FEXCore::Config::{}::OFF }},\n".format(op_key))
|
||||
for enum_op_key, enum_op_vals in Enums.items():
|
||||
output_argloader.write("\t{{ \"{}\", FEXCore::Config::{}::{} }},\n".format(enum_op_vals, op_key, enum_op_key.upper()))
|
||||
i += 1
|
||||
|
||||
output_argloader.write("}};\n")
|
||||
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
def check_for_duplicate_options(options):
|
||||
short_map = []
|
||||
long_map = []
|
||||
@@ -492,4 +554,7 @@ print_parse_argloader_options(options);
|
||||
# Generate environment loader code
|
||||
print_parse_envloader_options(options);
|
||||
|
||||
# Generate enum variable options
|
||||
print_parse_enum_options(options);
|
||||
|
||||
output_argloader.close()
|
||||
+2
-6
@@ -251,11 +251,8 @@ def parse_ops(ops):
|
||||
|
||||
# Print out enum values
|
||||
def print_enums():
|
||||
if len(IROps) > 255:
|
||||
ExitError("We have more than uint8_t ops. We have {}. Time to upgrade to uint16_t".format(len(IROps)))
|
||||
|
||||
output_file.write("#ifdef IROP_ENUM\n")
|
||||
output_file.write("enum IROps : uint8_t {\n")
|
||||
output_file.write("enum IROps : uint16_t {\n")
|
||||
|
||||
for op in IROps:
|
||||
output_file.write("\tOP_{},\n" .format(op.Name.upper()))
|
||||
@@ -282,7 +279,6 @@ def print_ir_structs(defines):
|
||||
output_file.write("\tIROps Op;\n\n")
|
||||
output_file.write("\tuint8_t Size;\n")
|
||||
output_file.write("\tuint8_t ElementSize;\n")
|
||||
output_file.write("\tuint8_t _pad;\n")
|
||||
|
||||
output_file.write("\ttemplate<typename T>\n")
|
||||
output_file.write("\tT const* C() const { return reinterpret_cast<T const*>(Data); }\n")
|
||||
@@ -679,7 +675,7 @@ def print_ir_allocator_helpers():
|
||||
output_file.write("\t\t#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED\n")
|
||||
|
||||
for Validation in op.EmitValidation:
|
||||
output_file.write("\t\tassert({});\n".format(Validation))
|
||||
output_file.write("\tLOGMAN_THROW_A_FMT({}, \"\");\n".format(Validation))
|
||||
output_file.write("\t\t#endif\n")
|
||||
|
||||
output_file.write("\t\treturn Op;\n")
|
||||
|
||||
+1
-1
@@ -1,4 +1,4 @@
|
||||
set (MAN_DIR ${CMAKE_INSTALL_PREFIX}/share/man CACHE PATH "MAN_DIR")
|
||||
set (MAN_DIR share/man CACHE PATH "MAN_DIR")
|
||||
|
||||
set (FEXCORE_BASE_SRCS
|
||||
Interface/Config/Config.cpp
|
||||
|
||||
+3
-2
@@ -1,5 +1,6 @@
|
||||
#include "Common/StringConv.h"
|
||||
#include "Common/StringUtils.h"
|
||||
#include "FEXCore/Utils/EnumUtils.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
@@ -16,7 +17,6 @@
|
||||
#include <FEXHeaderUtils/Filesystem.h>
|
||||
|
||||
#include <array>
|
||||
#include <assert.h>
|
||||
#include <cstdlib>
|
||||
#include <functional>
|
||||
#include <optional>
|
||||
@@ -37,6 +37,7 @@ namespace DefaultValues {
|
||||
#define OPT_BASE(type, group, enum, json, default) const P(type) P(enum) = P(default);
|
||||
#define OPT_STR(group, enum, json, default) const std::string_view P(enum) = P(default);
|
||||
#define OPT_STRARRAY(group, enum, json, default) OPT_STR(group, enum, json, default)
|
||||
#define OPT_STRENUM(group, enum, json, default) const uint64_t P(enum) = FEXCore::ToUnderlying(P(default));
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}
|
||||
|
||||
@@ -451,7 +452,7 @@ namespace DefaultValues {
|
||||
auto Value = FEXCore::Config::Get(Option);
|
||||
|
||||
if (!FEXCore::StrConv::Conv(**Value, &Result)) {
|
||||
assert(0 && "Attempted to convert invalid value");
|
||||
LOGMAN_MSG_A_FMT("Attempted to convert invalid value");
|
||||
}
|
||||
return Result;
|
||||
}
|
||||
|
||||
@@ -251,6 +251,20 @@
|
||||
"Set this in an application configuration for injecting in to only specific applications.",
|
||||
"\tNote: If x86/x86_64 libSegFault.so isn't installed then this option won't work."
|
||||
]
|
||||
},
|
||||
"Disassemble": {
|
||||
"Type": "strenum",
|
||||
"Default": "FEXCore::Config::Disassemble::OFF",
|
||||
"Enums": {
|
||||
"DISPATCHER": "dispatcher",
|
||||
"BLOCKS": "blocks"
|
||||
},
|
||||
"Desc": [
|
||||
"Allows controlling of the vixl disassembler.",
|
||||
"\toff: No disassembly will be output",
|
||||
"\tdispatcher: Will enable disassembly of the JIT dispatcher loop",
|
||||
"\tblocks: Will enable disassembly of the translated instruction code blocks"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Logging": {
|
||||
|
||||
+3
-2
@@ -1,5 +1,4 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Core.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
@@ -27,7 +26,9 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
bool FEXCore::Context::ContextImpl::InitializeContext() {
|
||||
return FEXCore::CPU::CreateCPUCore(this);
|
||||
// This should be used for generating things that are shared between threads
|
||||
CPUID.Init(this);
|
||||
return true;
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetExitHandler(ExitHandler handler) {
|
||||
|
||||
+14
-11
@@ -153,8 +153,10 @@ namespace FEXCore::Context {
|
||||
*/
|
||||
void DestroyThread(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
|
||||
#ifndef _WIN32
|
||||
void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child) override;
|
||||
#endif
|
||||
void SetSignalDelegator(FEXCore::SignalDelegator *SignalDelegation) override;
|
||||
void SetSyscallHandler(FEXCore::HLE::SyscallHandler *Handler) override;
|
||||
|
||||
@@ -165,29 +167,29 @@ namespace FEXCore::Context {
|
||||
FEXCore::IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(const fextl::string& Name) override;
|
||||
void UnloadAOTIRCacheEntry(FEXCore::IR::AOTIRCacheEntry *Entry) override;
|
||||
|
||||
void SetAOTIRLoader(std::function<int(const fextl::string&)> CacheReader) override {
|
||||
IRCaptureCache.SetAOTIRLoader(CacheReader);
|
||||
void SetAOTIRLoader(AOTIRLoaderCBFn CacheReader) override {
|
||||
IRCaptureCache.SetAOTIRLoader(std::move(CacheReader));
|
||||
}
|
||||
void SetAOTIRWriter(std::function<fextl::unique_ptr<AOTIRWriter>(const fextl::string&)> CacheWriter) override {
|
||||
IRCaptureCache.SetAOTIRWriter(CacheWriter);
|
||||
void SetAOTIRWriter(AOTIRWriterCBFn CacheWriter) override {
|
||||
IRCaptureCache.SetAOTIRWriter(std::move(CacheWriter));
|
||||
}
|
||||
void SetAOTIRRenamer(std::function<void(const fextl::string&)> CacheRenamer) override {
|
||||
IRCaptureCache.SetAOTIRRenamer(CacheRenamer);
|
||||
void SetAOTIRRenamer(AOTIRRenamerCBFn CacheRenamer) override {
|
||||
IRCaptureCache.SetAOTIRRenamer(std::move(CacheRenamer));
|
||||
}
|
||||
|
||||
void FinalizeAOTIRCache() override {
|
||||
IRCaptureCache.FinalizeAOTIRCache();
|
||||
}
|
||||
void WriteFilesWithCode(std::function<void(const fextl::string& fileid, const fextl::string& filename)> Writer) override {
|
||||
void WriteFilesWithCode(AOTIRCodeFileWriterFn Writer) override {
|
||||
IRCaptureCache.WriteFilesWithCode(Writer);
|
||||
}
|
||||
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) override;
|
||||
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, std::function<void(uint64_t start, uint64_t Length)> callback) override;
|
||||
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, CodeRangeInvalidationFn callback) override;
|
||||
void MarkMemoryShared() override;
|
||||
|
||||
void ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, fextl::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) override;
|
||||
// returns false if a handler was already registered
|
||||
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)> Handler, void *Creator = nullptr, void *Data = nullptr) override;
|
||||
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void *Creator = nullptr, void *Data = nullptr) override;
|
||||
|
||||
void AppendThunkDefinitions(fextl::vector<FEXCore::IR::ThunkDefinition> const& Definitions) override;
|
||||
|
||||
@@ -379,6 +381,8 @@ namespace FEXCore::Context {
|
||||
|
||||
bool ExitOnHLTEnabled() const { return ExitOnHLT; }
|
||||
|
||||
FEXCore::CPU::CPUBackendFeatures BackendFeatures;
|
||||
|
||||
protected:
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
@@ -433,8 +437,7 @@ namespace FEXCore::Context {
|
||||
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
|
||||
|
||||
std::shared_mutex CustomIRMutex;
|
||||
fextl::unordered_map<uint64_t, std::tuple<std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)>, void *, void *>> CustomIRHandlers;
|
||||
FEXCore::CPU::CPUBackendFeatures BackendFeatures;
|
||||
fextl::unordered_map<uint64_t, std::tuple<CustomIREntrypointHandler, void *, void *>> CustomIRHandlers;
|
||||
FEXCore::CPU::DispatcherConfig DispatcherConfig;
|
||||
};
|
||||
|
||||
|
||||
@@ -192,6 +192,23 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
|
||||
return;
|
||||
}
|
||||
|
||||
if ((Constant >> 32) == 0) {
|
||||
// If the upper 32-bits is all zero, we can now switch to a 32-bit move.
|
||||
s = ARMEmitter::Size::i32Bit;
|
||||
Is64Bit = false;
|
||||
Segments = 2;
|
||||
}
|
||||
|
||||
// If this can be loaded with a mov bitmask.
|
||||
const auto IsImm = vixl::aarch64::Assembler::IsImmLogical(Constant, RegSizeInBits(s));
|
||||
if (IsImm) {
|
||||
orr(s, Reg, ARMEmitter::Reg::zr, Constant);
|
||||
if (NOPPad) {
|
||||
nop(); nop(); nop();
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
int NumMoves = 1;
|
||||
int RequiredMoveSegments{};
|
||||
|
||||
|
||||
@@ -175,6 +175,7 @@ protected:
|
||||
#endif
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
vixl::aarch64::PrintDisassembler Disasm {stderr};
|
||||
FEX_CONFIG_OPT(Disassemble, DISASSEMBLE);
|
||||
#endif
|
||||
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
|
||||
};
|
||||
|
||||
@@ -129,7 +129,9 @@ public:
|
||||
constexpr uint32_t Op = 0b0011'0001'0 << 23;
|
||||
DataProcessing_AddSub_Imm(Op, s, rd, rn, Imm, LSL12);
|
||||
}
|
||||
|
||||
void cmn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, uint32_t Imm, bool LSL12 = false) {
|
||||
adds(s, FEXCore::ARMEmitter::Reg::zr, rn, Imm, LSL12);
|
||||
}
|
||||
void sub(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t Imm, bool LSL12 = false) {
|
||||
constexpr uint32_t Op = 0b0101'0001'0 << 23;
|
||||
DataProcessing_AddSub_Imm(Op, s, rd, rn, Imm, LSL12);
|
||||
@@ -219,6 +221,10 @@ public:
|
||||
eor(s, rd, rn, n, immr, imms);
|
||||
}
|
||||
|
||||
void tst(ARMEmitter::Size s, Register rn, uint64_t imm) {
|
||||
ands(s, Reg::zr, rn, imm);
|
||||
}
|
||||
|
||||
// Move wide immediate
|
||||
void movn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, uint32_t Imm, uint32_t Offset = 0) {
|
||||
LOGMAN_THROW_A_FMT((Imm & 0xFFFF0000U) == 0, "Upper bits of move wide not valid");
|
||||
@@ -286,6 +292,9 @@ public:
|
||||
LOGMAN_THROW_A_FMT((lsb + width) <= RegSizeInBits(s), "Tried to sbfx a region larger than the register");
|
||||
sbfm(s, rd, rn, lsb, lsb + width - 1);
|
||||
}
|
||||
void sbfiz(ARMEmitter::Size s, Register rd, Register rn, uint32_t lsb, uint32_t width) {
|
||||
xbfiz_helper(true, s, rd, rn, lsb, width);
|
||||
}
|
||||
void asr(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t shift) {
|
||||
LOGMAN_THROW_A_FMT(shift <= RegSizeInBits(s), "Tried to asr a region larger than the register");
|
||||
sbfm(s, rd, rn, shift, RegSizeInBits(s) - 1);
|
||||
@@ -306,6 +315,10 @@ public:
|
||||
DataProcessing_Logical_Imm(Op, s, rd, rn, s == ARMEmitter::Size::i64Bit, immr, imms);
|
||||
}
|
||||
|
||||
void ubfiz(ARMEmitter::Size s, Register rd, Register rn, uint32_t lsb, uint32_t width) {
|
||||
xbfiz_helper(false, s, rd, rn, lsb, width);
|
||||
}
|
||||
|
||||
void lsl(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t shift) {
|
||||
const auto RegSize = RegSizeInBits(s);
|
||||
LOGMAN_THROW_A_FMT(shift < RegSize, "Tried to lsl a region larger than the register");
|
||||
@@ -324,10 +337,23 @@ public:
|
||||
|
||||
void bfi(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t lsb, uint32_t width) {
|
||||
const auto RegSize = RegSizeInBits(s);
|
||||
LOGMAN_THROW_A_FMT(width > 0, "bfi needs width > 0");
|
||||
LOGMAN_THROW_A_FMT((lsb + width) <= RegSize, "Tried to bfi a region larger than the register");
|
||||
LOGMAN_THROW_A_FMT(width > 0, "bfc/bfi needs width > 0");
|
||||
LOGMAN_THROW_A_FMT((lsb + width) <= RegSize, "Tried to bfc/bfi a region larger than the register");
|
||||
bfm(s, rd, rn, (RegSize - lsb) & (RegSize - 1), width - 1);
|
||||
}
|
||||
void bfc(ARMEmitter::Size s, Register rd, uint32_t lsb, uint32_t width) {
|
||||
bfi(s, rd, Reg::zr, lsb, width);
|
||||
}
|
||||
void bfxil(ARMEmitter::Size s, Register rd, Register rn, uint32_t lsb, uint32_t width) {
|
||||
[[maybe_unused]] const auto reg_size_bits = RegSizeInBits(s);
|
||||
const auto lsb_p_width = lsb + width;
|
||||
|
||||
LOGMAN_THROW_A_FMT(width >= 1, "bfxil needs width >= 1");
|
||||
LOGMAN_THROW_A_FMT(lsb_p_width <= reg_size_bits, "bfxil lsb + width ({}) must be <= {}. lsb={}, width={}",
|
||||
lsb_p_width, reg_size_bits, lsb, width);
|
||||
|
||||
bfm(s, rd, rn, lsb, lsb_p_width - 1);
|
||||
}
|
||||
|
||||
// Extract
|
||||
void extr(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, uint32_t Imm) {
|
||||
@@ -564,6 +590,9 @@ public:
|
||||
constexpr uint32_t Op = 0b010'1010'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void tst(ARMEmitter::Size s, Register rn, Register rm, ShiftType shift = ShiftType::LSL, uint32_t amt = 0) {
|
||||
ands(s, Reg::zr, rn, rm, shift, amt);
|
||||
}
|
||||
|
||||
void orn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
constexpr uint32_t Op = 0b010'1010'001U << 21;
|
||||
@@ -585,6 +614,9 @@ public:
|
||||
void adds(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
adds(ARMEmitter::Size::i64Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void cmn(FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
adds(ARMEmitter::Size::i64Bit, FEXCore::ARMEmitter::XReg::zr, rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void sub(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
sub(ARMEmitter::Size::i64Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
@@ -607,6 +639,9 @@ public:
|
||||
void adds(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
adds(ARMEmitter::Size::i32Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void cmn(FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
adds(ARMEmitter::Size::i32Bit, FEXCore::ARMEmitter::WReg::zr, rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void sub(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
sub(ARMEmitter::Size::i32Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
@@ -633,6 +668,9 @@ public:
|
||||
constexpr uint32_t Op = 0b010'1011'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void cmn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
adds(s, FEXCore::ARMEmitter::Reg::zr, rn, rm, Shift, amt);
|
||||
}
|
||||
void sub(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
LOGMAN_THROW_AA_FMT(Shift != FEXCore::ARMEmitter::ShiftType::ROR, "Doesn't support ROR");
|
||||
constexpr uint32_t Op = 0b100'1011'000U << 21;
|
||||
@@ -664,6 +702,9 @@ public:
|
||||
constexpr uint32_t Op = 0b010'1011'001U << 21;
|
||||
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, Option, Shift);
|
||||
}
|
||||
void cmn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ExtendedType Option, uint32_t Shift = 0) {
|
||||
adds(s, FEXCore::ARMEmitter::Reg::zr, rn, rm, Option, Shift);
|
||||
}
|
||||
void sub(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ExtendedType Option, uint32_t Shift = 0) {
|
||||
constexpr uint32_t Op = 0b100'1011'001U << 21;
|
||||
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, Option, Shift);
|
||||
@@ -694,6 +735,12 @@ public:
|
||||
constexpr uint32_t Op = 0b0111'1010'000U << 21;
|
||||
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, FEXCore::ARMEmitter::ExtendedType::UXTB, 0);
|
||||
}
|
||||
void ngc(ARMEmitter::Size s, Register rd, Register rm) {
|
||||
sbc(s, rd, Reg::zr, rm);
|
||||
}
|
||||
void ngcs(ARMEmitter::Size s, Register rd, Register rm) {
|
||||
sbcs(s, rd, Reg::zr, rm);
|
||||
}
|
||||
|
||||
// Rotate right into flags
|
||||
void rmif(XRegister rn, uint32_t shift, uint32_t mask) {
|
||||
@@ -761,6 +808,18 @@ public:
|
||||
constexpr uint32_t Op = 0b0001'1010'100 << 21;
|
||||
ConditionalCompare(Op, 1, 0b01, s, rd, rn, rm, Cond);
|
||||
}
|
||||
void cneg(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Condition Cond) {
|
||||
csneg(s, rd, rn, rn, InvertCondition(Cond));
|
||||
}
|
||||
void cinc(ARMEmitter::Size s, Register rd, Register rn, Condition cond) {
|
||||
csinc(s, rd, rn, rn, InvertCondition(cond));
|
||||
}
|
||||
void cinv(ARMEmitter::Size s, Register rd, Register rn, Condition cond) {
|
||||
csinv(s, rd, rn, rn, InvertCondition(cond));
|
||||
}
|
||||
void csetm(ARMEmitter::Size s, Register rd, Condition cond) {
|
||||
csinv(s, rd, Reg::zr, Reg::zr, InvertCondition(cond));
|
||||
}
|
||||
|
||||
// Data processing - 3 source
|
||||
void madd(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::Register ra) {
|
||||
@@ -815,6 +874,13 @@ public:
|
||||
}
|
||||
|
||||
private:
|
||||
static constexpr Condition InvertCondition(Condition cond) {
|
||||
// These behave as always, so it makes no sense to allow inverting these.
|
||||
LOGMAN_THROW_AA_FMT(cond != Condition::CC_AL && cond != Condition::CC_NV,
|
||||
"Cannot invert CC_AL or CC_NV");
|
||||
return static_cast<Condition>(FEXCore::ToUnderlying(cond) ^ 1);
|
||||
}
|
||||
|
||||
void and_(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t n, uint32_t immr, uint32_t imms) {
|
||||
constexpr uint32_t Op = 0b001'0010'00 << 22;
|
||||
DataProcessing_Logical_Imm(Op, s, rd, rn, n, immr, imms);
|
||||
@@ -922,6 +988,24 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
void xbfiz_helper(bool is_signed, ARMEmitter::Size s, Register rd, Register rn, uint32_t lsb, uint32_t width) {
|
||||
[[maybe_unused]] const auto lsb_p_width = lsb + width;
|
||||
const auto reg_size_bits = RegSizeInBits(s);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(lsb_p_width <= reg_size_bits, "lsb + width ({}) must be <= {}. lsb={}, width={}",
|
||||
lsb_p_width, reg_size_bits, lsb, width);
|
||||
LOGMAN_THROW_AA_FMT(width >= 1, "xbfiz width must be >= 1");
|
||||
|
||||
const auto immr = (reg_size_bits - lsb) & (reg_size_bits - 1);
|
||||
const auto imms = width - 1;
|
||||
|
||||
if (is_signed) {
|
||||
sbfm(s, rd, rn, immr, imms);
|
||||
} else {
|
||||
ubfm(s, rd, rn, immr, imms);
|
||||
}
|
||||
}
|
||||
|
||||
void DataProcessing_Extract(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, uint32_t Imm) {
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
|
||||
@@ -6,6 +6,7 @@
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
@@ -207,58 +208,94 @@ namespace FEXCore::ARMEmitter {
|
||||
*/
|
||||
class SVEMemOperand final {
|
||||
public:
|
||||
// Used for scalar + vector variants to determine
|
||||
// extension behavior on the index values.
|
||||
enum class ModType : uint8_t {
|
||||
MOD_UXTW,
|
||||
MOD_SXTW,
|
||||
MOD_LSL,
|
||||
MOD_NONE,
|
||||
};
|
||||
|
||||
enum class Type {
|
||||
ScalarPlusScalar,
|
||||
ScalarPlusImm,
|
||||
ScalarPlusVector,
|
||||
VectorPlusImm,
|
||||
};
|
||||
|
||||
SVEMemOperand(XRegister rn, XRegister rm = XReg::zr)
|
||||
: rn {rn}
|
||||
, MemType{Type::ScalarPlusScalar}
|
||||
, MetaType {
|
||||
.ScalarScalarType {
|
||||
.Header = { .MemType = TYPE_SCALAR_SCALAR },
|
||||
.rm = rm,
|
||||
}
|
||||
} {}
|
||||
SVEMemOperand(XRegister rn, int32_t imm = 0)
|
||||
: rn {rn}
|
||||
, MemType{Type::ScalarPlusImm}
|
||||
, MetaType {
|
||||
.ScalarImmType {
|
||||
.Header = { .MemType = TYPE_SCALAR_IMM },
|
||||
.Imm = imm,
|
||||
}
|
||||
} {}
|
||||
SVEMemOperand(XRegister rn, ZRegister zm, ModType mod = ModType::MOD_NONE, uint8_t scale = 0)
|
||||
: rn{rn}
|
||||
, MemType{Type::ScalarPlusVector}
|
||||
, MetaType {
|
||||
.ScalarVectorType {
|
||||
.zm = zm,
|
||||
.mod = mod,
|
||||
.scale = scale,
|
||||
}
|
||||
} {}
|
||||
SVEMemOperand(ZRegister zn, uint32_t imm)
|
||||
: rn{Register{zn.Idx()}}
|
||||
, MemType{Type::VectorPlusImm}
|
||||
, MetaType {
|
||||
.VectorImmType{
|
||||
.Imm = imm,
|
||||
}
|
||||
} {}
|
||||
|
||||
Register rn;
|
||||
enum Type {
|
||||
TYPE_SCALAR_SCALAR,
|
||||
TYPE_SCALAR_IMM,
|
||||
TYPE_SCALAR_VECTOR,
|
||||
TYPE_VECTOR_IMM,
|
||||
};
|
||||
struct HeaderStruct {
|
||||
Type MemType;
|
||||
};
|
||||
[[nodiscard]] bool IsScalarPlusScalar() const {
|
||||
return MemType == Type::ScalarPlusScalar;
|
||||
}
|
||||
[[nodiscard]] bool IsScalarPlusImm() const {
|
||||
return MemType == Type::ScalarPlusImm;
|
||||
}
|
||||
[[nodiscard]] bool IsScalarPlusVector() const {
|
||||
return MemType == Type::ScalarPlusVector;
|
||||
}
|
||||
[[nodiscard]] bool IsVectorPlusImm() const {
|
||||
return MemType == Type::VectorPlusImm;
|
||||
}
|
||||
|
||||
union {
|
||||
HeaderStruct Header;
|
||||
union Data {
|
||||
struct {
|
||||
HeaderStruct Header;
|
||||
Register rm;
|
||||
} ScalarScalarType;
|
||||
|
||||
struct {
|
||||
HeaderStruct Header;
|
||||
int32_t Imm;
|
||||
} ScalarImmType;
|
||||
|
||||
struct {
|
||||
HeaderStruct Header;
|
||||
ZRegister zm;
|
||||
// TODO: Implement support for modifier
|
||||
ModType mod;
|
||||
uint8_t scale;
|
||||
} ScalarVectorType;
|
||||
|
||||
struct {
|
||||
HeaderStruct Header;
|
||||
// rn will be a ZRegister
|
||||
int32_t Imm;
|
||||
uint32_t Imm;
|
||||
} VectorImmType;
|
||||
} MetaType;
|
||||
};
|
||||
|
||||
Register rn;
|
||||
Type MemType;
|
||||
Data MetaType;
|
||||
};
|
||||
|
||||
/* This `ExtendedMemOperand` class is used for the helper load-store instructions.
|
||||
@@ -475,6 +512,20 @@ namespace FEXCore::ARMEmitter {
|
||||
SVE_ALL = 0b11111,
|
||||
};
|
||||
|
||||
// Used with SVE FP immediate arithmetic instructions
|
||||
enum class SVEFAddSubImm : uint32_t {
|
||||
_0_5,
|
||||
_1_0,
|
||||
};
|
||||
enum class SVEFMulImm : uint32_t {
|
||||
_0_5,
|
||||
_2_0,
|
||||
};
|
||||
enum class SVEFMaxMinImm : uint32_t {
|
||||
_0_0,
|
||||
_1_0,
|
||||
};
|
||||
|
||||
/* This `BackwardLabel` struct used for retaining a location for PC-Relative instructions.
|
||||
* This is specifically a label for a target that is logically `below` an instruction that uses it.
|
||||
* Which means that a branch would jump backwards.
|
||||
|
||||
+436
-546
File diff suppressed because it is too large.
Load diff
+1356
-1238
File diff suppressed because it is too large.
Load diff
+11
-18
@@ -11,7 +11,6 @@ $end_info$
|
||||
#include "FEXCore/Utils/DeferredSignalMutex.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/Core.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/GdbServer.h"
|
||||
@@ -77,22 +76,11 @@ $end_info$
|
||||
#include <utility>
|
||||
#include <xxhash.h>
|
||||
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
bool CreateCPUCore(Context::ContextImpl *CTX) {
|
||||
// This should be used for generating things that are shared between threads
|
||||
CTX->CPUID.Init(CTX);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct ThreadLocalData {
|
||||
FEXCore::Core::InternalThreadState* Thread;
|
||||
};
|
||||
|
||||
thread_local ThreadLocalData ThreadData{};
|
||||
|
||||
constexpr std::array<std::string_view const, 22> FlagNames = {
|
||||
"CF",
|
||||
"",
|
||||
@@ -665,6 +653,7 @@ namespace FEXCore::Context {
|
||||
delete Thread;
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
void ContextImpl::UnlockAfterFork(FEXCore::Core::InternalThreadState *LiveThread, bool Child) {
|
||||
Allocator::UnlockAfterFork(LiveThread, Child);
|
||||
|
||||
@@ -718,6 +707,7 @@ namespace FEXCore::Context {
|
||||
CodeInvalidationMutex.lock();
|
||||
Allocator::LockBeforeFork(Thread);
|
||||
}
|
||||
#endif
|
||||
|
||||
void ContextImpl::AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr) {
|
||||
Thread->LookupCache->AddBlockMapping(Address, Ptr);
|
||||
@@ -873,14 +863,14 @@ namespace FEXCore::Context {
|
||||
|
||||
if (TableInfo && TableInfo->OpcodeDispatcher) {
|
||||
auto Fn = TableInfo->OpcodeDispatcher;
|
||||
Thread->OpDispatcher->HandledLock = false;
|
||||
Thread->OpDispatcher->ResetHandledLock();
|
||||
Thread->OpDispatcher->ResetDecodeFailure();
|
||||
std::invoke(Fn, Thread->OpDispatcher, DecodedInfo);
|
||||
if (Thread->OpDispatcher->HadDecodeFailure()) {
|
||||
HadDispatchError = true;
|
||||
}
|
||||
else {
|
||||
if (Thread->OpDispatcher->HandledLock != IsLocked) {
|
||||
if (Thread->OpDispatcher->HasHandledLock() != IsLocked) {
|
||||
HadDispatchError = true;
|
||||
LogMan::Msg::EFmt("Missing LOCK HANDLER at 0x{:x}{{'{}'}}", Block.Entry + BlockInstructionsLength, TableInfo->Name ?: "UND");
|
||||
}
|
||||
@@ -930,7 +920,7 @@ namespace FEXCore::Context {
|
||||
|
||||
IR::IREmitter *IREmitter = Thread->OpDispatcher.get();
|
||||
|
||||
auto ShouldDump = static_cast<ContextImpl*>(Thread->CTX)->Config.DumpIR() != "no" || Thread->OpDispatcher->ShouldDump;
|
||||
auto ShouldDump = static_cast<ContextImpl*>(Thread->CTX)->Config.DumpIR() != "no" || Thread->OpDispatcher->ShouldDumpIR();
|
||||
// Debug
|
||||
{
|
||||
if (ShouldDump) {
|
||||
@@ -1161,11 +1151,12 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
void ContextImpl::ExecutionThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
Core::ThreadData.Thread = Thread;
|
||||
Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_WAITING;
|
||||
|
||||
InitializeThreadTLSData(Thread);
|
||||
#ifndef _WIN32
|
||||
Alloc::OSAllocator::RegisterTLSData(Thread);
|
||||
#endif
|
||||
|
||||
++IdleWaitRefCount;
|
||||
|
||||
@@ -1210,7 +1201,9 @@ namespace FEXCore::Context {
|
||||
--IdleWaitRefCount;
|
||||
IdleWaitCV.notify_all();
|
||||
|
||||
#ifndef _WIN32
|
||||
Alloc::OSAllocator::UninstallTLSData(Thread);
|
||||
#endif
|
||||
SignalDelegation->UninstallTLSState(Thread);
|
||||
|
||||
// If the parent thread is waiting to join, then we can't destroy our thread object
|
||||
@@ -1250,7 +1243,7 @@ namespace FEXCore::Context {
|
||||
InvalidateGuestCodeRangeInternal(this, Start, Length);
|
||||
}
|
||||
|
||||
void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, std::function<void(uint64_t start, uint64_t Length)> CallAfter) {
|
||||
void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, CodeRangeInvalidationFn CallAfter) {
|
||||
// Potential deferred since Thread might not be valid.
|
||||
// Thread object isn't valid very early in frontend's initialization.
|
||||
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
|
||||
@@ -1296,7 +1289,7 @@ namespace FEXCore::Context {
|
||||
Thread->LookupCache->Erase(GuestRIP);
|
||||
}
|
||||
|
||||
CustomIRResult ContextImpl::AddCustomIREntrypoint(uintptr_t Entrypoint, std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)> Handler, void *Creator, void *Data) {
|
||||
CustomIRResult ContextImpl::AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void *Creator, void *Data) {
|
||||
LOGMAN_THROW_A_FMT(Config.Is64BitMode || !(Entrypoint >> 32), "64-bit Entrypoint in 32-bit mode {:x}", Entrypoint);
|
||||
|
||||
std::unique_lock lk(CustomIRMutex);
|
||||
|
||||
-23
@@ -1,23 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
namespace FEXCore {
|
||||
class CodeLoader;
|
||||
}
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
/**
|
||||
* @brief Create the CPU core backend for the context passed in
|
||||
*
|
||||
* @param CTX
|
||||
*
|
||||
* @return true if core was able to be create
|
||||
*/
|
||||
bool CreateCPUCore(FEXCore::Context::ContextImpl *CTX);
|
||||
|
||||
bool LoadCode(FEXCore::Context::ContextImpl *CTX, FEXCore::CodeLoader *Loader);
|
||||
}
|
||||
@@ -503,8 +503,10 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
}
|
||||
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
const auto DisasmEnd = GetCursorAddress<const vixl::aarch64::Instruction*>();
|
||||
Disasm.DisassembleBuffer(DisasmBegin, DisasmEnd);
|
||||
if (Disassemble() & FEXCore::Config::Disassemble::DISPATCHER) {
|
||||
const auto DisasmEnd = GetCursorAddress<const vixl::aarch64::Instruction*>();
|
||||
Disasm.DisassembleBuffer(DisasmBegin, DisasmEnd);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
|
||||
+18
-13
@@ -10,7 +10,6 @@ $end_info$
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <array>
|
||||
#include <assert.h>
|
||||
#include <algorithm>
|
||||
#include <cstring>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
@@ -239,8 +238,19 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModR
|
||||
Operand->Data.SIB.Scale = 1 << SIB.scale;
|
||||
|
||||
// The invalid encoding types are described at Table 1-12. "promoted nsigned is always non-zero"
|
||||
Operand->Data.SIB.Index = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_X ? 1 : 0, SIB.index, false, false, false, false, 0b100);
|
||||
Operand->Data.SIB.Base = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, SIB.base, false, false, false, false, ModRM.mod == 0 ? 0b101 : 16);
|
||||
{
|
||||
// If we have a VSIB byte (as opposed to SIB), then the index register is a vector.
|
||||
const bool IsIndexVector = (DecodeInst->TableInfo->Flags & InstFlags::FLAGS_VEX_VSIB) != 0;
|
||||
if (IsIndexVector) {
|
||||
DecodeInst->Flags |= X86Tables::DecodeFlags::FLAG_VSIB_BYTE;
|
||||
}
|
||||
|
||||
const uint8_t IndexREX = (DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_X) != 0 ? 1 : 0;
|
||||
const uint8_t BaseREX = (DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B) != 0 ? 1 : 0;
|
||||
|
||||
Operand->Data.SIB.Index = MapModRMToReg(IndexREX, SIB.index, false, false, IsIndexVector, false, 0b100);
|
||||
Operand->Data.SIB.Base = MapModRMToReg(BaseREX, SIB.base, false, false, false, false, ModRM.mod == 0 ? 0b101 : 16);
|
||||
}
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Displacement <= 4, "Number of bytes should be <= 4 for literal src");
|
||||
|
||||
@@ -671,9 +681,6 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
DecodeInst->ModRM = ModRMByte;
|
||||
DecodeInst->DecodedModRM = true;
|
||||
|
||||
FEXCore::X86Tables::ModRMDecoded ModRM;
|
||||
ModRM.Hex = DecodeInst->ModRM;
|
||||
|
||||
uint16_t X87Op = ((Op - 0xD8) << 8) | ModRMByte;
|
||||
return NormalOp(&X87Ops[X87Op], X87Op);
|
||||
}
|
||||
@@ -957,7 +964,7 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
|
||||
}
|
||||
|
||||
if (DecodeInst->Dest.IsGPR()) {
|
||||
assert(DecodeInst->Dest.Data.GPR.GPR != FEXCore::X86State::REG_INVALID);
|
||||
LOGMAN_THROW_A_FMT(DecodeInst->Dest.Data.GPR.GPR != FEXCore::X86State::REG_INVALID, "Destination GPR was invalid");
|
||||
}
|
||||
|
||||
return true;
|
||||
@@ -1015,15 +1022,13 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
|
||||
// If we are conditional then a target can be the instruction past the conditional instruction
|
||||
uint64_t FallthroughRIP = DecodeInst->PC + DecodeInst->InstSize;
|
||||
if (HasBlocks.find(FallthroughRIP) == HasBlocks.end() &&
|
||||
BlocksToDecode.find(FallthroughRIP) == BlocksToDecode.end()) {
|
||||
BlocksToDecode.emplace(FallthroughRIP);
|
||||
if (!HasBlocks.contains(FallthroughRIP)) {
|
||||
BlocksToDecode.insert(FallthroughRIP);
|
||||
}
|
||||
}
|
||||
|
||||
if (HasBlocks.find(TargetRIP) == HasBlocks.end() &&
|
||||
BlocksToDecode.find(TargetRIP) == BlocksToDecode.end()) {
|
||||
BlocksToDecode.emplace(TargetRIP);
|
||||
if (!HasBlocks.contains(TargetRIP)) {
|
||||
BlocksToDecode.insert(TargetRIP);
|
||||
}
|
||||
} else {
|
||||
if (ExternalBranches) {
|
||||
|
||||
@@ -71,6 +71,7 @@ HostFeatures::HostFeatures() {
|
||||
SupportsRCPC = Features.Has(vixl::CPUFeatures::Feature::kRCpc);
|
||||
SupportsTSOImm9 = Features.Has(vixl::CPUFeatures::Feature::kRCpcImm);
|
||||
SupportsPMULL_128Bit = Features.Has(vixl::CPUFeatures::Feature::kPmull1Q);
|
||||
SupportsCSSC = Features.Has(vixl::CPUFeatures::Feature::kCSSC);
|
||||
|
||||
Supports3DNow = true;
|
||||
SupportsSSE4A = true;
|
||||
|
||||
@@ -113,6 +113,22 @@ DEF_OP(Neg) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Abs) {
|
||||
auto Op = IROp->C<IR::IROp_Abs>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
const int64_t Src = *GetSrc<int64_t*>(Data->SSAData, Op->Src);
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
GD = std::abs(static_cast<int32_t>(Src));
|
||||
break;
|
||||
case 8:
|
||||
GD = std::abs(static_cast<int64_t>(Src));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Abs Size: {}\n", OpSize); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Mul) {
|
||||
auto Op = IROp->C<IR::IROp_Mul>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
@@ -869,12 +885,8 @@ DEF_OP(Select) {
|
||||
auto Op = IROp->C<IR::IROp_Select>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp2);
|
||||
|
||||
uint64_t ArgTrue;
|
||||
uint64_t ArgFalse;
|
||||
|
||||
if (OpSize == 4) {
|
||||
ArgTrue = *GetSrc<uint32_t*>(Data->SSAData, Op->TrueVal);
|
||||
ArgFalse = *GetSrc<uint32_t*>(Data->SSAData, Op->FalseVal);
|
||||
@@ -885,10 +897,25 @@ DEF_OP(Select) {
|
||||
|
||||
bool CompResult;
|
||||
|
||||
if (Op->CompareSize == 4)
|
||||
if (Op->CompareSize == 4) {
|
||||
const auto Src1 = *GetSrc<uint32_t*>(Data->SSAData, Op->Cmp1);
|
||||
const auto Src2 = *GetSrc<uint32_t*>(Data->SSAData, Op->Cmp2);
|
||||
CompResult = IsConditionTrue<uint32_t, int32_t, float>(Op->Cond.Val, Src1, Src2);
|
||||
else
|
||||
}
|
||||
else if (Op->CompareSize == 8) {
|
||||
const auto Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp1);
|
||||
const auto Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp2);
|
||||
CompResult = IsConditionTrue<uint64_t, int64_t, double>(Op->Cond.Val, Src1, Src2);
|
||||
}
|
||||
else if (Op->CompareSize == 16) {
|
||||
const auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Cmp1);
|
||||
const auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Cmp2);
|
||||
CompResult = IsConditionTrue<__uint128_t, __int128_t, double>(Op->Cond.Val, Src1, Src2);
|
||||
}
|
||||
else {
|
||||
LOGMAN_MSG_A_FMT("Unknown select size: {}", Op->CompareSize);
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
GD = CompResult ? ArgTrue : ArgFalse;
|
||||
}
|
||||
|
||||
@@ -211,7 +211,7 @@ DEF_OP(Vector_FToF) {
|
||||
// Little bit tricky here
|
||||
// Sometimes is used to convert from a 128bit vector register
|
||||
// in to a 64bit vector register with different sized elements
|
||||
// eg: %ssa5 i32v2 = Vector_FToF %ssa4 i128, #0x8
|
||||
// eg: %5 i32v2 = Vector_FToF %4 i128, #0x8
|
||||
uint8_t Elements = OpSize == 8 ? 2 : OpSize / Op->SrcElementSize;
|
||||
DO_VECTOR_1SRC_2TYPE_OP_NOSIZE(float, double, Func, 0, 0)
|
||||
break;
|
||||
|
||||
+1
-1
@@ -3,7 +3,7 @@
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
enum IROps : uint8_t;
|
||||
enum IROps : uint16_t;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
@@ -52,6 +52,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(ADD, Add);
|
||||
REGISTER_OP(SUB, Sub);
|
||||
REGISTER_OP(NEG, Neg);
|
||||
REGISTER_OP(ABS, Abs);
|
||||
REGISTER_OP(MUL, Mul);
|
||||
REGISTER_OP(UMUL, UMul);
|
||||
REGISTER_OP(DIV, Div);
|
||||
@@ -266,6 +267,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(VUMULL2, VUMull2);
|
||||
REGISTER_OP(VSMULL2, VSMull2);
|
||||
REGISTER_OP(VUABDL, VUABDL);
|
||||
REGISTER_OP(VUABDL2, VUABDL2);
|
||||
REGISTER_OP(VTBL1, VTBL1);
|
||||
REGISTER_OP(VREV64, VRev64);
|
||||
REGISTER_OP(VPCMPESTRX, VPCMPESTRX);
|
||||
|
||||
@@ -85,6 +85,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(Add);
|
||||
DEF_OP(Sub);
|
||||
DEF_OP(Neg);
|
||||
DEF_OP(Abs);
|
||||
DEF_OP(Mul);
|
||||
DEF_OP(UMul);
|
||||
DEF_OP(Div);
|
||||
@@ -292,6 +293,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(VUMull2);
|
||||
DEF_OP(VSMull2);
|
||||
DEF_OP(VUABDL);
|
||||
DEF_OP(VUABDL2);
|
||||
DEF_OP(VTBL1);
|
||||
DEF_OP(VRev64);
|
||||
DEF_OP(VPCMPESTRX);
|
||||
@@ -348,6 +350,15 @@ namespace FEXCore::CPU {
|
||||
template<typename unsigned_type, typename signed_type, typename float_type>
|
||||
[[nodiscard]] static bool IsConditionTrue(uint8_t Cond, uint64_t Src1, uint64_t Src2) {
|
||||
bool CompResult = false;
|
||||
if constexpr (sizeof(unsigned_type) == 16) {
|
||||
LOGMAN_THROW_A_FMT(Cond != FEXCore::IR::COND_FLU &&
|
||||
Cond != FEXCore::IR::COND_FGE &&
|
||||
Cond != FEXCore::IR::COND_FLEU &&
|
||||
Cond != FEXCore::IR::COND_FGT &&
|
||||
Cond != FEXCore::IR::COND_FU &&
|
||||
Cond != FEXCore::IR::COND_FNU, "Unsupported comparison for 128-bit floats");
|
||||
}
|
||||
|
||||
switch (Cond) {
|
||||
case FEXCore::IR::COND_EQ:
|
||||
CompResult = static_cast<unsigned_type>(Src1) == static_cast<unsigned_type>(Src2);
|
||||
|
||||
@@ -190,19 +190,31 @@ DEF_OP(LoadFlag) {
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += offsetof(FEXCore::Core::CPUState, flags[0]);
|
||||
ContextPtr += Op->Flag;
|
||||
uint8_t const *MemData = reinterpret_cast<uint8_t const*>(ContextPtr);
|
||||
GD = *MemData;
|
||||
|
||||
if (Op->Flag == 24 /* NZCV */) {
|
||||
uint32_t const *MemData = reinterpret_cast<uint32_t const*>(ContextPtr);
|
||||
GD = *MemData;
|
||||
} else {
|
||||
uint8_t const *MemData = reinterpret_cast<uint8_t const*>(ContextPtr);
|
||||
GD = *MemData;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreFlag) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
uint8_t Arg = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
uint32_t Arg = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += offsetof(FEXCore::Core::CPUState, flags[0]);
|
||||
ContextPtr += Op->Flag;
|
||||
uint8_t *MemData = reinterpret_cast<uint8_t*>(ContextPtr);
|
||||
*MemData = Arg;
|
||||
|
||||
if (Op->Flag == 24 /* NZCV */) {
|
||||
uint32_t *MemData = reinterpret_cast<uint32_t*>(ContextPtr);
|
||||
*MemData = Arg;
|
||||
} else {
|
||||
uint8_t *MemData = reinterpret_cast<uint8_t*>(ContextPtr);
|
||||
*MemData = Arg;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadMem) {
|
||||
|
||||
@@ -2227,6 +2227,33 @@ DEF_OP(VUABDL) {
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VUABDL2) {
|
||||
const auto Op = IROp->C<IR::IROp_VUABDL2>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Vector1);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Vector2);
|
||||
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE];
|
||||
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
|
||||
const auto Func8 = [](auto a, auto b) { return std::abs((int16_t)a - (int16_t)b); };
|
||||
const auto Func16 = [](auto a, auto b) { return std::abs((int32_t)a - (int32_t)b); };
|
||||
const auto Func32 = [](auto a, auto b) { return std::abs((int64_t)a - (int64_t)b); };
|
||||
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_2SRC_2TYPE_OP_TOP_SRC(2, uint16_t, uint8_t, Func8)
|
||||
DO_VECTOR_2SRC_2TYPE_OP_TOP_SRC(4, uint32_t, uint16_t, Func16)
|
||||
DO_VECTOR_2SRC_2TYPE_OP_TOP_SRC(8, uint64_t, uint32_t, Func32)
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VTBL1) {
|
||||
const auto Op = IROp->C<IR::IROp_VTBL1>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
+108
-9
@@ -4,6 +4,7 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
@@ -83,6 +84,21 @@ DEF_OP(Add) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(TestNZ) {
|
||||
auto Op = IROp->C<IR::IROp_TestNZ>();
|
||||
const uint8_t OpSize = Op->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src1.ID());
|
||||
tst(EmitSize, Src, Src);
|
||||
|
||||
// TODO: Optimize this out
|
||||
mrs(Dst, ARMEmitter::SystemRegister::NZCV);
|
||||
}
|
||||
|
||||
DEF_OP(Sub) {
|
||||
auto Op = IROp->C<IR::IROp_Sub>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
@@ -108,6 +124,26 @@ DEF_OP(Neg) {
|
||||
neg(EmitSize, GetReg(Node), GetReg(Op->Src.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(Abs) {
|
||||
auto Op = IROp->C<IR::IROp_Abs>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsCSSC) {
|
||||
// On CSSC supporting processors, this turns in to one instruction and doesn't modify flags.
|
||||
abs(EmitSize, Dst, Src);
|
||||
}
|
||||
else {
|
||||
cmp(EmitSize, Src, 0);
|
||||
cneg(EmitSize, Dst, Src, ARMEmitter::Condition::CC_MI);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Mul) {
|
||||
auto Op = IROp->C<IR::IROp_Mul>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
@@ -310,6 +346,40 @@ DEF_OP(Or) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Orlshl) {
|
||||
auto Op = IROp->C<IR::IROp_Orlshl>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
orr(EmitSize, Dst, Src1, Const << Op->BitShift);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
orr(EmitSize, Dst, Src1, Src2, ARMEmitter::ShiftType::LSL, Op->BitShift);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Orlshr) {
|
||||
auto Op = IROp->C<IR::IROp_Orlshr>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
orr(EmitSize, Dst, Src1, Const >> Op->BitShift);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
orr(EmitSize, Dst, Src1, Src2, ARMEmitter::ShiftType::LSR, Op->BitShift);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(And) {
|
||||
auto Op = IROp->C<IR::IROp_And>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
@@ -633,7 +703,10 @@ DEF_OP(LDiv) {
|
||||
mov(EmitSize, ARMEmitter::Reg::r2, Divisor);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LDIVHandler));
|
||||
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
|
||||
// Move result to its destination register
|
||||
mov(EmitSize, Dst, ARMEmitter::Reg::r0);
|
||||
@@ -697,7 +770,11 @@ DEF_OP(LUDiv) {
|
||||
mov(EmitSize, ARMEmitter::Reg::r2, Divisor);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LUDIVHandler));
|
||||
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
|
||||
// Move result to its destination register
|
||||
mov(EmitSize, Dst, ARMEmitter::Reg::r0);
|
||||
|
||||
@@ -768,7 +845,11 @@ DEF_OP(LRem) {
|
||||
mov(EmitSize, ARMEmitter::Reg::r2, Divisor);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LREMHandler));
|
||||
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
|
||||
// Move result to its destination register
|
||||
mov(EmitSize, Dst, ARMEmitter::Reg::r0);
|
||||
|
||||
@@ -833,7 +914,11 @@ DEF_OP(LURem) {
|
||||
mov(EmitSize, ARMEmitter::Reg::r2, Divisor);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LUREMHandler));
|
||||
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
|
||||
// Move result to its destination register
|
||||
mov(EmitSize, Dst, ARMEmitter::Reg::r0);
|
||||
|
||||
@@ -1020,14 +1105,22 @@ DEF_OP(Bfi) {
|
||||
const auto SrcDst = GetReg(Op->Dest.ID());
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
mov(EmitSize, TMP1, SrcDst);
|
||||
bfi(EmitSize, TMP1, Src, Op->lsb, Op->Width);
|
||||
|
||||
if (OpSize == 8) {
|
||||
mov(EmitSize, Dst, TMP1.R());
|
||||
if (Dst == SrcDst) {
|
||||
// If Dst and SrcDst match then this turns in to a simple BFI instruction.
|
||||
bfi(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
}
|
||||
else {
|
||||
ubfx(EmitSize, Dst, TMP1, 0, OpSize * 8);
|
||||
// Destination didn't match the dst source register.
|
||||
// TODO: Inefficient until FEX can have RA constraints here.
|
||||
mov(EmitSize, TMP1, SrcDst);
|
||||
bfi(EmitSize, TMP1, Src, Op->lsb, Op->Width);
|
||||
|
||||
if (OpSize == 8) {
|
||||
mov(EmitSize, Dst, TMP1.R());
|
||||
}
|
||||
else {
|
||||
ubfx(EmitSize, Dst, TMP1, 0, OpSize * 8);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1085,6 +1178,7 @@ DEF_OP(Select) {
|
||||
const auto CompareEmitSize = Op->CompareSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
uint64_t Const;
|
||||
auto cc = MapSelectCC(Op->Cond);
|
||||
|
||||
if (IsGPR(Op->Cmp1.ID())) {
|
||||
const auto Src1 = GetReg(Op->Cmp1.ID());
|
||||
@@ -1095,7 +1189,14 @@ DEF_OP(Select) {
|
||||
const auto Src2 = GetReg(Op->Cmp2.ID());
|
||||
cmp(CompareEmitSize, Src1, Src2);
|
||||
}
|
||||
} else if (IsFPR(Op->Cmp1.ID())) {
|
||||
}
|
||||
else if (IsGPRPair(Op->Cmp1.ID())) {
|
||||
const auto Src1 = GetRegPair(Op->Cmp1.ID());
|
||||
const auto Src2 = GetRegPair(Op->Cmp2.ID());
|
||||
cmp(EmitSize, Src1.first, Src2.first);
|
||||
ccmp(EmitSize, Src1.second, Src2.second, ARMEmitter::StatusFlags::None, cc);
|
||||
}
|
||||
else if (IsFPR(Op->Cmp1.ID())) {
|
||||
const auto Src1 = GetVReg(Op->Cmp1.ID());
|
||||
const auto Src2 = GetVReg(Op->Cmp2.ID());
|
||||
fcmp(Op->CompareSize == 8 ? ARMEmitter::ScalarRegSize::i64Bit : ARMEmitter::ScalarRegSize::i32Bit, Src1, Src2);
|
||||
@@ -1103,8 +1204,6 @@ DEF_OP(Select) {
|
||||
LOGMAN_MSG_A_FMT("Select: Expected GPR or FPR");
|
||||
}
|
||||
|
||||
auto cc = MapSelectCC(Op->Cond);
|
||||
|
||||
uint64_t const_true, const_false;
|
||||
bool is_const_true = IsInlineConstant(Op->TrueVal, &const_true);
|
||||
bool is_const_false = IsInlineConstant(Op->FalseVal, &const_false);
|
||||
|
||||
+24
-6
@@ -549,7 +549,7 @@ static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Fram
|
||||
FEXCore::ARMEmitter::Emitter::ClearICache((void*)branch, 24);
|
||||
|
||||
// Add de-linking handler
|
||||
Context::ContextImpl::ThreadAddBlockLink(Thread, GuestRip, (uintptr_t)record, [branch, LinkerAddress]{
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [branch, LinkerAddress]{
|
||||
FEXCore::ARMEmitter::Emitter emit((uint8_t*)(branch), 24);
|
||||
FEXCore::ARMEmitter::ForwardLabel l_BranchHost;
|
||||
emit.ldr(FEXCore::ARMEmitter::XReg::x0, &l_BranchHost);
|
||||
@@ -563,7 +563,7 @@ static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Fram
|
||||
record[0] = HostCode;
|
||||
|
||||
// Add de-linking handler
|
||||
Context::ContextImpl::ThreadAddBlockLink(Thread, GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
record[0] = LinkerAddress;
|
||||
});
|
||||
}
|
||||
@@ -724,6 +724,12 @@ bool Arm64JITCore::IsGPR(IR::NodeID Node) const {
|
||||
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
|
||||
}
|
||||
|
||||
bool Arm64JITCore::IsGPRPair(IR::NodeID Node) const {
|
||||
auto Class = GetRegClass(Node);
|
||||
|
||||
return Class == IR::GPRPairClass;
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
@@ -845,8 +851,10 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
REGISTER_OP(INLINEENTRYPOINTOFFSET, InlineEntrypointOffset);
|
||||
REGISTER_OP(CYCLECOUNTER, CycleCounter);
|
||||
REGISTER_OP(ADD, Add);
|
||||
REGISTER_OP(TESTNZ, TestNZ);
|
||||
REGISTER_OP(SUB, Sub);
|
||||
REGISTER_OP(NEG, Neg);
|
||||
REGISTER_OP(ABS, Abs);
|
||||
REGISTER_OP(MUL, Mul);
|
||||
REGISTER_OP(UMUL, UMul);
|
||||
REGISTER_OP(DIV, Div);
|
||||
@@ -856,6 +864,8 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
REGISTER_OP(MULH, MulH);
|
||||
REGISTER_OP(UMULH, UMulH);
|
||||
REGISTER_OP(OR, Or);
|
||||
REGISTER_OP(ORLSHL, Orlshl);
|
||||
REGISTER_OP(ORLSHR, Orlshr);
|
||||
REGISTER_OP(AND, And);
|
||||
REGISTER_OP(ANDN, Andn);
|
||||
REGISTER_OP(XOR, Xor);
|
||||
@@ -1072,6 +1082,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
REGISTER_OP(VUMULL2, VUMull2);
|
||||
REGISTER_OP(VSMULL2, VSMull2);
|
||||
REGISTER_OP(VUABDL, VUABDL);
|
||||
REGISTER_OP(VUABDL2, VUABDL2);
|
||||
REGISTER_OP(VTBL1, VTBL1);
|
||||
REGISTER_OP(VREV64, VRev64);
|
||||
#undef REGISTER_OP
|
||||
@@ -1097,6 +1108,9 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
}
|
||||
PendingTargetLabel = nullptr;
|
||||
|
||||
// CodeSize not including the tail data.
|
||||
const uint64_t CodeOnlySize = GetCursorAddress<uint8_t *>() - CodeData.BlockBegin;
|
||||
|
||||
// Add the JitCodeTail
|
||||
auto JITBlockTailLocation = GetCursorAddress<uint8_t *>();
|
||||
auto JITBlockTail = GetCursorAddress<JITCodeTail*>();
|
||||
@@ -1135,11 +1149,13 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
|
||||
JITBlockTail->Size = CodeData.Size;
|
||||
|
||||
ClearICache(CodeData.BlockBegin, CodeData.Size);
|
||||
ClearICache(CodeData.BlockBegin, CodeOnlySize);
|
||||
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
const auto DisasmEnd = GetCursorAddress<const vixl::aarch64::Instruction*>();
|
||||
Disasm.DisassembleBuffer(DisasmBegin, DisasmEnd);
|
||||
if (Disassemble() & FEXCore::Config::Disassemble::BLOCKS) {
|
||||
const auto DisasmEnd = reinterpret_cast<const vixl::aarch64::Instruction*>(JITBlockTailLocation);
|
||||
Disasm.DisassembleBuffer(DisasmBegin, DisasmEnd);
|
||||
}
|
||||
#endif
|
||||
|
||||
if (DebugData) {
|
||||
@@ -1174,7 +1190,9 @@ fextl::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl *
|
||||
|
||||
CPUBackendFeatures GetArm64JITBackendFeatures() {
|
||||
return CPUBackendFeatures {
|
||||
.SupportsStaticRegisterAllocation = true
|
||||
.SupportsStaticRegisterAllocation = true,
|
||||
.SupportsShiftedBitwise = true,
|
||||
.SupportsFlags = true,
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
@@ -112,6 +112,7 @@ private:
|
||||
|
||||
[[nodiscard]] bool IsFPR(IR::NodeID Node) const;
|
||||
[[nodiscard]] bool IsGPR(IR::NodeID Node) const;
|
||||
[[nodiscard]] bool IsGPRPair(IR::NodeID Node) const;
|
||||
|
||||
[[nodiscard]] FEXCore::ARMEmitter::ExtendedMemOperand GenerateMemOperand(uint8_t AccessSize,
|
||||
FEXCore::ARMEmitter::Register Base,
|
||||
@@ -236,8 +237,10 @@ private:
|
||||
DEF_OP(InlineEntrypointOffset);
|
||||
DEF_OP(CycleCounter);
|
||||
DEF_OP(Add);
|
||||
DEF_OP(TestNZ);
|
||||
DEF_OP(Sub);
|
||||
DEF_OP(Neg);
|
||||
DEF_OP(Abs);
|
||||
DEF_OP(Mul);
|
||||
DEF_OP(UMul);
|
||||
DEF_OP(Div);
|
||||
@@ -247,6 +250,8 @@ private:
|
||||
DEF_OP(MulH);
|
||||
DEF_OP(UMulH);
|
||||
DEF_OP(Or);
|
||||
DEF_OP(Orlshl);
|
||||
DEF_OP(Orlshr);
|
||||
DEF_OP(And);
|
||||
DEF_OP(Andn);
|
||||
DEF_OP(Xor);
|
||||
@@ -452,6 +457,7 @@ private:
|
||||
DEF_OP(VUMull2);
|
||||
DEF_OP(VSMull2);
|
||||
DEF_OP(VUABDL);
|
||||
DEF_OP(VUABDL2);
|
||||
DEF_OP(VTBL1);
|
||||
DEF_OP(VRev64);
|
||||
|
||||
|
||||
+26
-30
@@ -10,6 +10,7 @@ $end_info$
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
@@ -676,10 +677,7 @@ DEF_OP(LoadContextIndexed) {
|
||||
case 2:
|
||||
case 4:
|
||||
case 8: {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Op->Stride);
|
||||
mul(ARMEmitter::Size::i64Bit, TMP1, Index, TMP1);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, TMP1.R());
|
||||
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, Index, FEXCore::ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Op->Stride));
|
||||
const auto Dst = GetReg(Node);
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
@@ -716,10 +714,7 @@ DEF_OP(LoadContextIndexed) {
|
||||
case 8:
|
||||
case 16:
|
||||
case 32: {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Op->Stride);
|
||||
mul(ARMEmitter::Size::i64Bit, TMP1, Index, TMP1);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, TMP1.R());
|
||||
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, Index, FEXCore::ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Op->Stride));
|
||||
const auto Dst = GetVReg(Node);
|
||||
|
||||
switch (OpSize) {
|
||||
@@ -774,9 +769,7 @@ DEF_OP(StoreContextIndexed) {
|
||||
case 2:
|
||||
case 4:
|
||||
case 8: {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Op->Stride);
|
||||
mul(ARMEmitter::Size::i64Bit, TMP1, Index, TMP1);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, TMP1.R());
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, Index, FEXCore::ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Op->Stride));
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
@@ -815,9 +808,7 @@ DEF_OP(StoreContextIndexed) {
|
||||
case 8:
|
||||
case 16:
|
||||
case 32: {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Op->Stride);
|
||||
mul(ARMEmitter::Size::i64Bit, TMP1, Index, TMP1);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, TMP1.R());
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, Index, FEXCore::ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Op->Stride));
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
@@ -1060,12 +1051,20 @@ DEF_OP(FillRegister) {
|
||||
DEF_OP(LoadFlag) {
|
||||
auto Op = IROp->C<IR::IROp_LoadFlag>();
|
||||
auto Dst = GetReg(Node);
|
||||
ldrb(Dst, STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
|
||||
if (Op->Flag == 24 /* NZCV */)
|
||||
ldr(Dst.W(), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
else
|
||||
ldrb(Dst, STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
}
|
||||
|
||||
DEF_OP(StoreFlag) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
strb(GetReg(Op->Value.ID()), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
|
||||
if (Op->Flag == 24 /* NZCV */)
|
||||
str(GetReg(Op->Value.ID()).W(), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
else
|
||||
strb(GetReg(Op->Value.ID()), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
}
|
||||
|
||||
FEXCore::ARMEmitter::ExtendedMemOperand Arm64JITCore::GenerateMemOperand(uint8_t AccessSize,
|
||||
@@ -1085,9 +1084,9 @@ FEXCore::ARMEmitter::ExtendedMemOperand Arm64JITCore::GenerateMemOperand(uint8_t
|
||||
} else {
|
||||
auto RegOffset = GetReg(Offset.ID());
|
||||
switch(OffsetType.Val) {
|
||||
case IR::MEM_OFFSET_SXTX.Val: return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::SXTX, (int)std::log2(OffsetScale) );
|
||||
case IR::MEM_OFFSET_UXTW.Val: return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::UXTW, (int)std::log2(OffsetScale) );
|
||||
case IR::MEM_OFFSET_SXTW.Val: return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::SXTW, (int)std::log2(OffsetScale) );
|
||||
case IR::MEM_OFFSET_SXTX.Val: return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::SXTX, FEXCore::ilog2(OffsetScale) );
|
||||
case IR::MEM_OFFSET_UXTW.Val: return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::UXTW, FEXCore::ilog2(OffsetScale) );
|
||||
case IR::MEM_OFFSET_SXTW.Val: return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale) );
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled GenerateMemOperand OffsetType: {}", OffsetType.Val); break;
|
||||
}
|
||||
}
|
||||
@@ -1240,8 +1239,6 @@ DEF_OP(LoadMemTSO) {
|
||||
ldapurb(Dst, MemReg, Offset);
|
||||
}
|
||||
else {
|
||||
// Aligned
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
ldapurh(Dst, MemReg, Offset);
|
||||
@@ -1256,6 +1253,7 @@ DEF_OP(LoadMemTSO) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
}
|
||||
@@ -1266,8 +1264,6 @@ DEF_OP(LoadMemTSO) {
|
||||
ldaprb(Dst.W(), MemReg);
|
||||
}
|
||||
else {
|
||||
// Aligned
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
ldaprh(Dst.W(), MemReg);
|
||||
@@ -1282,6 +1278,7 @@ DEF_OP(LoadMemTSO) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
}
|
||||
@@ -1292,8 +1289,6 @@ DEF_OP(LoadMemTSO) {
|
||||
ldarb(Dst, MemReg);
|
||||
}
|
||||
else {
|
||||
// Aligned
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
ldarh(Dst, MemReg);
|
||||
@@ -1308,11 +1303,11 @@ DEF_OP(LoadMemTSO) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
}
|
||||
else {
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
switch (OpSize) {
|
||||
@@ -1340,7 +1335,8 @@ DEF_OP(LoadMemTSO) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
// Half-barrier.
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISHLD);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1525,6 +1521,7 @@ DEF_OP(StoreMemTSO) {
|
||||
stlurb(Src, MemReg, Offset);
|
||||
}
|
||||
else {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
@@ -1540,7 +1537,6 @@ DEF_OP(StoreMemTSO) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
nop();
|
||||
}
|
||||
}
|
||||
else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
@@ -1551,6 +1547,7 @@ DEF_OP(StoreMemTSO) {
|
||||
stlrb(Src, MemReg);
|
||||
}
|
||||
else {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
@@ -1566,10 +1563,10 @@ DEF_OP(StoreMemTSO) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
nop();
|
||||
}
|
||||
}
|
||||
else {
|
||||
// Half-Barrier.
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
const auto Src = GetVReg(Op->Value.ID());
|
||||
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
@@ -1598,7 +1595,6 @@ DEF_OP(StoreMemTSO) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+59
-26
@@ -8,6 +8,8 @@ $end_info$
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
DEF_OP(VectorZero) {
|
||||
@@ -2293,40 +2295,47 @@ DEF_OP(VInsElement) {
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const uint32_t ElementSize = Op->Header.ElementSize;
|
||||
|
||||
const auto DestIdx = Op->DestIdx;
|
||||
const auto SrcIdx = Op->SrcIdx;
|
||||
const uint32_t DestIdx = Op->DestIdx;
|
||||
const uint32_t SrcIdx = Op->SrcIdx;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto SrcVector = GetVReg(Op->SrcVector.ID());
|
||||
auto Reg = GetVReg(Op->DestVector.ID());
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 1 || ElementSize == 2 || ElementSize == 4 || ElementSize == 8 || ElementSize == 16, "Invalid size");
|
||||
const auto SubRegSize =
|
||||
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i128Bit;
|
||||
|
||||
// We're going to use this to create our predicate register literal.
|
||||
// On an SVE 256-bit capable system, the predicate register will be
|
||||
// 32-bit in size. We want to set up only the element corresponding
|
||||
// to the destination index, since we're going to copy over the equivalent
|
||||
// indexed element from the source vector.
|
||||
auto Data = [ElementSize, DestIdx]() -> uint32_t {
|
||||
const auto Data = [ElementSize, DestIdx]() -> uint32_t {
|
||||
const auto Log2ElementSize = FEXCore::ilog2(ElementSize);
|
||||
|
||||
[[maybe_unused]] const auto MaxIndex = (32U >> Log2ElementSize) - 1;
|
||||
LOGMAN_THROW_AA_FMT(DestIdx <= MaxIndex, "DestIdx ({}) out of range. Must be within [0, {}]",
|
||||
DestIdx, MaxIndex);
|
||||
|
||||
const auto ShiftAmount = DestIdx << Log2ElementSize;
|
||||
|
||||
switch (ElementSize) {
|
||||
case 1:
|
||||
LOGMAN_THROW_AA_FMT(DestIdx <= 31, "DestIdx out of range: {}", DestIdx);
|
||||
return 1U << DestIdx;
|
||||
case 2:
|
||||
LOGMAN_THROW_AA_FMT(DestIdx <= 15, "DestIdx out of range: {}", DestIdx);
|
||||
return 1U << (DestIdx * 2);
|
||||
case 4:
|
||||
LOGMAN_THROW_AA_FMT(DestIdx <= 7, "DestIdx out of range: {}", DestIdx);
|
||||
return 1U << (DestIdx * 4);
|
||||
case 8:
|
||||
LOGMAN_THROW_AA_FMT(DestIdx <= 3, "DestIdx out of range: {}", DestIdx);
|
||||
return 1U << (DestIdx * 8);
|
||||
return 1U << ShiftAmount;
|
||||
case 16:
|
||||
LOGMAN_THROW_AA_FMT(DestIdx <= 1, "DestIdx out of range: {}", DestIdx);
|
||||
// Predicates can't be subdivided into the Q format, so we can just set up
|
||||
// the predicate to select the two adjacent doublewords.
|
||||
return 0x101U << (DestIdx * 16);
|
||||
return 0x101U << ShiftAmount;
|
||||
default:
|
||||
FEX_UNREACHABLE;
|
||||
return UINT32_MAX;
|
||||
@@ -2339,13 +2348,6 @@ DEF_OP(VInsElement) {
|
||||
adr(TMP1, &DataLocation);
|
||||
ldr(Predicate, TMP1);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 1 || ElementSize == 2 || ElementSize == 4 || ElementSize == 8 || ElementSize == 16, "Invalid size");
|
||||
const auto SubRegSize =
|
||||
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i128Bit;
|
||||
|
||||
// Broadcast our source value across a temporary,
|
||||
// then combine with the destination.
|
||||
dup(SubRegSize, VTMP2.Z(), SrcVector.Z(), SrcIdx);
|
||||
@@ -2365,11 +2367,6 @@ DEF_OP(VInsElement) {
|
||||
Bind(&PastConstant);
|
||||
}
|
||||
else {
|
||||
if (Dst.Idx() != Reg.Idx()) {
|
||||
mov(VTMP1.Q(), Reg.Q());
|
||||
Reg = VTMP1;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 1 || ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize =
|
||||
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
@@ -2377,6 +2374,11 @@ DEF_OP(VInsElement) {
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (Dst.Idx() != Reg.Idx()) {
|
||||
mov(VTMP1.Q(), Reg.Q());
|
||||
Reg = VTMP1;
|
||||
}
|
||||
|
||||
ins(SubRegSize, Reg.Q(), DestIdx, SrcVector.Q(), SrcIdx);
|
||||
|
||||
if (Dst.Idx() != Reg.Idx()) {
|
||||
@@ -3025,6 +3027,37 @@ DEF_OP(VUABDL) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VUABDL2) {
|
||||
const auto Op = IROp->C<IR::IROp_VUABDL2>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize =
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
// To mimic the behavior of AdvSIMD UABDL, we need to get the
|
||||
// absolute difference of the even elements (UADBLB), get the
|
||||
// absolute difference of the odd elemenets (UABDLT), then
|
||||
// interleave the results in both vectors together.
|
||||
|
||||
uabdlb(SubRegSize, VTMP1.Z(), Vector1.Z(), Vector2.Z());
|
||||
uabdlt(SubRegSize, VTMP2.Z(), Vector1.Z(), Vector2.Z());
|
||||
zip2(SubRegSize, Dst.Z(), VTMP1.Z(), VTMP2.Z());
|
||||
} else {
|
||||
uabdl2(SubRegSize, Dst.Q(), Vector1.Q(), Vector2.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VTBL1) {
|
||||
const auto Op = IROp->C<IR::IROp_VTBL1>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -161,6 +161,30 @@ DEF_OP(Neg) {
|
||||
neg(Dst);
|
||||
}
|
||||
|
||||
DEF_OP(Abs) {
|
||||
auto Op = IROp->C<IR::IROp_Abs>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
Xbyak::Reg Src;
|
||||
Xbyak::Reg Dst;
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
Src = GetSrc<RA_32>(Op->Src.ID());
|
||||
Dst = GetDst<RA_32>(Node);
|
||||
break;
|
||||
case 8:
|
||||
Src = GetSrc<RA_64>(Op->Src.ID());
|
||||
Dst = GetDst<RA_64>(Node);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Abs size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
mov(Dst, Src);
|
||||
mov(TMP1, Src);
|
||||
neg(Dst);
|
||||
cmovs(Dst, TMP1);
|
||||
}
|
||||
|
||||
DEF_OP(Mul) {
|
||||
auto Op = IROp->C<IR::IROp_Mul>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
@@ -1116,7 +1140,28 @@ DEF_OP(Select) {
|
||||
} else {
|
||||
cmp(GRCMP(Op->Cmp1.ID()), GRCMP(Op->Cmp2.ID()));
|
||||
}
|
||||
} else if (IsFPR(Op->Cmp1.ID())) {
|
||||
}
|
||||
else if (IsGPRPair(Op->Cmp1.ID())) {
|
||||
if (Op->CompareSize == 4) {
|
||||
const auto Src1 = GetSrcPair<RA_32>(Op->Cmp1.ID());
|
||||
const auto Src2 = GetSrcPair<RA_32>(Op->Cmp2.ID());
|
||||
mov (TMP1.cvt32(), Src1.first);
|
||||
mov (TMP2.cvt32(), Src1.second);
|
||||
xor_(TMP1.cvt32(), Src2.first);
|
||||
xor_(TMP2.cvt32(), Src2.second);
|
||||
or_(TMP1.cvt32(), TMP2.cvt32());
|
||||
}
|
||||
else {
|
||||
const auto Src1 = GetSrcPair<RA_64>(Op->Cmp1.ID());
|
||||
const auto Src2 = GetSrcPair<RA_64>(Op->Cmp2.ID());
|
||||
mov (TMP1, Src1.first);
|
||||
mov (TMP2, Src1.second);
|
||||
xor_(TMP1, Src2.first);
|
||||
xor_(TMP2, Src2.second);
|
||||
or_(TMP1, TMP2);
|
||||
}
|
||||
}
|
||||
else if (IsFPR(Op->Cmp1.ID())) {
|
||||
if (Op->CompareSize == 4)
|
||||
ucomiss(GetSrc(Op->Cmp1.ID()), GetSrc(Op->Cmp2.ID()));
|
||||
else
|
||||
@@ -1298,6 +1343,7 @@ void X86JITCore::RegisterALUHandlers() {
|
||||
REGISTER_OP(ADD, Add);
|
||||
REGISTER_OP(SUB, Sub);
|
||||
REGISTER_OP(NEG, Neg);
|
||||
REGISTER_OP(ABS, Abs);
|
||||
REGISTER_OP(MUL, Mul);
|
||||
REGISTER_OP(UMUL, UMul);
|
||||
REGISTER_OP(DIV, Div);
|
||||
|
||||
@@ -384,7 +384,7 @@ static uint64_t X86JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame,
|
||||
}
|
||||
|
||||
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
|
||||
Context::ContextImpl::ThreadAddBlockLink(Thread, GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
// undo the link
|
||||
record[0] = LinkerAddress;
|
||||
});
|
||||
@@ -484,11 +484,15 @@ IR::PhysicalRegister X86JITCore::GetPhys(IR::NodeID Node) const {
|
||||
}
|
||||
|
||||
bool X86JITCore::IsFPR(IR::NodeID Node) const {
|
||||
return RAData->GetNodeRegister(Node).Class == IR::FPRClass.Val;
|
||||
return RAData->GetNodeRegister(Node).Class == IR::FPRClass;
|
||||
}
|
||||
|
||||
bool X86JITCore::IsGPR(IR::NodeID Node) const {
|
||||
return RAData->GetNodeRegister(Node).Class == IR::GPRClass.Val;
|
||||
return RAData->GetNodeRegister(Node).Class == IR::GPRClass;
|
||||
}
|
||||
|
||||
bool X86JITCore::IsGPRPair(IR::NodeID Node) const {
|
||||
return RAData->GetNodeRegister(Node).Class == IR::GPRPairClass;
|
||||
}
|
||||
|
||||
template<uint8_t RAType>
|
||||
|
||||
@@ -169,6 +169,7 @@ private:
|
||||
|
||||
[[nodiscard]] bool IsFPR(IR::NodeID Node) const;
|
||||
[[nodiscard]] bool IsGPR(IR::NodeID Node) const;
|
||||
[[nodiscard]] bool IsGPRPair(IR::NodeID Node) const;
|
||||
|
||||
template<uint8_t RAType>
|
||||
[[nodiscard]] Xbyak::Reg GetSrc(IR::NodeID Node) const;
|
||||
@@ -245,6 +246,7 @@ private:
|
||||
DEF_OP(Add);
|
||||
DEF_OP(Sub);
|
||||
DEF_OP(Neg);
|
||||
DEF_OP(Abs);
|
||||
DEF_OP(Mul);
|
||||
DEF_OP(UMul);
|
||||
DEF_OP(Div);
|
||||
@@ -453,6 +455,7 @@ private:
|
||||
DEF_OP(VUMull2);
|
||||
DEF_OP(VSMull2);
|
||||
DEF_OP(VUABDL);
|
||||
DEF_OP(VUABDL2);
|
||||
DEF_OP(VTBL1);
|
||||
DEF_OP(VRev64);
|
||||
|
||||
|
||||
@@ -601,14 +601,22 @@ DEF_OP(LoadFlag) {
|
||||
auto Op = IROp->C<IR::IROp_LoadFlag>();
|
||||
|
||||
auto Dst = GetDst<RA_64>(Node);
|
||||
movzx(Dst, byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)]);
|
||||
|
||||
if (Op->Flag == 24 /* NZCV */)
|
||||
mov(Dst.cvt32(), dword [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)]);
|
||||
else
|
||||
movzx(Dst, byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)]);
|
||||
}
|
||||
|
||||
DEF_OP(StoreFlag) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
|
||||
mov (rax, GetSrc<RA_64>(Op->Value.ID()));
|
||||
mov(byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)], al);
|
||||
|
||||
if (Op->Flag == 24 /* NZCV */)
|
||||
mov(dword [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)], eax);
|
||||
else
|
||||
mov(byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)], al);
|
||||
}
|
||||
|
||||
Xbyak::RegExp X86JITCore::GenerateModRM(Xbyak::Reg Base, IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale) const {
|
||||
|
||||
@@ -4367,6 +4367,93 @@ DEF_OP(VUABDL) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VUABDL2) {
|
||||
const auto Op = IROp->C<IR::IROp_VUABDL2>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector1 = GetSrc(Op->Vector1.ID());
|
||||
const auto Vector2 = GetSrc(Op->Vector2.ID());
|
||||
|
||||
switch (ElementSize) {
|
||||
case 2: {
|
||||
vpxor(xmm15, xmm15, xmm15);
|
||||
vpxor(xmm14, xmm14, xmm14);
|
||||
|
||||
if (Is256Bit) {
|
||||
vextracti128(xmm15, ToYMM(Vector1), 1);
|
||||
vextracti128(xmm14, ToYMM(Vector2), 1);
|
||||
|
||||
vpxor(xmm12, xmm12, xmm12);
|
||||
vpxor(xmm13, xmm13, xmm13);
|
||||
vpxor(Dst, Dst, Dst);
|
||||
|
||||
// Bottom half
|
||||
vpunpcklbw(xmm13, xmm15, xmm13);
|
||||
vpunpcklbw(xmm12, xmm14, xmm12);
|
||||
|
||||
// Top half
|
||||
vpunpckhbw(xmm15, xmm15, Dst);
|
||||
vpunpckhbw(xmm14, xmm14, Dst);
|
||||
|
||||
// Reinsert
|
||||
vinserti128(ymm13, ymm13, xmm15, 1);
|
||||
vinserti128(ymm12, ymm12, xmm14, 1);
|
||||
|
||||
vpsubw(ToYMM(Dst), ymm12, ymm13);
|
||||
vpabsw(ToYMM(Dst), ToYMM(Dst));
|
||||
} else {
|
||||
vpunpckhbw(xmm15, Vector1, xmm15);
|
||||
vpunpckhbw(xmm14, Vector2, xmm14);
|
||||
vpsubw(Dst, xmm14, xmm15);
|
||||
vpabsw(Dst, Dst);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
vpxor(xmm15, xmm15, xmm15);
|
||||
vpxor(xmm14, xmm14, xmm14);
|
||||
|
||||
if (Is256Bit) {
|
||||
vextracti128(xmm15, ToYMM(Vector1), 1);
|
||||
vextracti128(xmm14, ToYMM(Vector2), 1);
|
||||
|
||||
vpxor(xmm12, xmm12, xmm12);
|
||||
vpxor(xmm13, xmm13, xmm13);
|
||||
vpxor(Dst, Dst, Dst);
|
||||
|
||||
// Bottom half
|
||||
vpunpcklwd(xmm13, xmm15, xmm13);
|
||||
vpunpcklwd(xmm12, xmm14, xmm12);
|
||||
|
||||
// Top half
|
||||
vpunpckhwd(xmm15, xmm15, Dst);
|
||||
vpunpckhwd(xmm14, xmm14, Dst);
|
||||
|
||||
// Reinsert
|
||||
vinserti128(ymm13, ymm13, xmm15, 1);
|
||||
vinserti128(ymm12, ymm12, xmm14, 1);
|
||||
|
||||
vpsubd(ToYMM(Dst), ymm12, ymm13);
|
||||
vpabsd(ToYMM(Dst), ToYMM(Dst));
|
||||
} else {
|
||||
vpunpckhwd(xmm15, Vector1, xmm15);
|
||||
vpunpckhwd(xmm14, Vector2, xmm14);
|
||||
vpsubd(Dst, xmm14, xmm15);
|
||||
vpabsd(Dst, Dst);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
DEF_OP(VTBL1) {
|
||||
const auto Op = IROp->C<IR::IROp_VTBL1>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -4609,6 +4696,7 @@ void X86JITCore::RegisterVectorHandlers() {
|
||||
REGISTER_OP(VUMULL2, VUMull2);
|
||||
REGISTER_OP(VSMULL2, VSMull2);
|
||||
REGISTER_OP(VUABDL, VUABDL);
|
||||
REGISTER_OP(VUABDL2, VUABDL2);
|
||||
REGISTER_OP(VTBL1, VTBL1);
|
||||
REGISTER_OP(VREV64, VRev64);
|
||||
#undef REGISTER_OP
|
||||
|
||||
+191
-138
@@ -145,6 +145,7 @@ void OpDispatchBuilder::SyscallOp(OpcodeArgs) {
|
||||
if (Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_BLOCK_END) {
|
||||
// RIP could have been updated after coming back from the Syscall.
|
||||
NewRIP = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, rip));
|
||||
CalculateDeferredFlags();
|
||||
_ExitFunction(NewRIP);
|
||||
}
|
||||
}
|
||||
@@ -178,6 +179,7 @@ void OpDispatchBuilder::ThunkOp(OpcodeArgs) {
|
||||
|
||||
// Store the new stack pointer
|
||||
StoreGPRRegister(X86State::REG_RSP, NewSP);
|
||||
CalculateDeferredFlags();
|
||||
|
||||
// Store the new RIP
|
||||
_ExitFunction(NewRIP);
|
||||
@@ -240,6 +242,7 @@ void OpDispatchBuilder::RETOp(OpcodeArgs) {
|
||||
|
||||
// Store the new stack pointer
|
||||
StoreGPRRegister(X86State::REG_RSP, NewSP);
|
||||
CalculateDeferredFlags();
|
||||
|
||||
// Store the new RIP
|
||||
_ExitFunction(NewRIP);
|
||||
@@ -304,6 +307,7 @@ void OpDispatchBuilder::IRETOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RSP, SP);
|
||||
}
|
||||
|
||||
CalculateDeferredFlags();
|
||||
_ExitFunction(NewRIP);
|
||||
BlockSetRIP = true;
|
||||
}
|
||||
@@ -313,6 +317,7 @@ void OpDispatchBuilder::CallbackReturnOp(OpcodeArgs) {
|
||||
// Store the new RIP
|
||||
_CallbackReturn();
|
||||
auto NewRIP = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, rip));
|
||||
CalculateDeferredFlags();
|
||||
// This ExitFunction won't actually get hit but needs to exist
|
||||
_ExitFunction(NewRIP);
|
||||
BlockSetRIP = true;
|
||||
@@ -669,7 +674,6 @@ void OpDispatchBuilder::POPOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::POPAOp(OpcodeArgs) {
|
||||
// 32bit only
|
||||
const uint8_t Size = GetSrcSize(Op);
|
||||
const uint8_t GPRSize = 4;
|
||||
|
||||
auto Constant = _Constant(Size);
|
||||
auto OldSP = LoadGPRRegister(X86State::REG_RSP);
|
||||
@@ -687,32 +691,32 @@ void OpDispatchBuilder::POPAOp(OpcodeArgs) {
|
||||
OrderedNode *Src{};
|
||||
OrderedNode *NewSP = OldSP;
|
||||
Src = _LoadMem(GPRClass, Size, NewSP, Size);
|
||||
StoreGPRRegister(X86State::REG_RDI, Src, GPRSize);
|
||||
StoreGPRRegister(X86State::REG_RDI, Src, Size);
|
||||
NewSP = _Add(NewSP, Constant);
|
||||
|
||||
Src = _LoadMem(GPRClass, Size, NewSP, Size);
|
||||
StoreGPRRegister(X86State::REG_RSI, Src, GPRSize);
|
||||
StoreGPRRegister(X86State::REG_RSI, Src, Size);
|
||||
NewSP = _Add(NewSP, Constant);
|
||||
|
||||
Src = _LoadMem(GPRClass, Size, NewSP, Size);
|
||||
StoreGPRRegister(X86State::REG_RBP, Src, GPRSize);
|
||||
StoreGPRRegister(X86State::REG_RBP, Src, Size);
|
||||
NewSP = _Add(NewSP, _Constant(Size * 2));
|
||||
|
||||
// Skip SP loading
|
||||
Src = _LoadMem(GPRClass, Size, NewSP, Size);
|
||||
StoreGPRRegister(X86State::REG_RBX, Src, GPRSize);
|
||||
StoreGPRRegister(X86State::REG_RBX, Src, Size);
|
||||
NewSP = _Add(NewSP, Constant);
|
||||
|
||||
Src = _LoadMem(GPRClass, Size, NewSP, Size);
|
||||
StoreGPRRegister(X86State::REG_RDX, Src, GPRSize);
|
||||
StoreGPRRegister(X86State::REG_RDX, Src, Size);
|
||||
NewSP = _Add(NewSP, Constant);
|
||||
|
||||
Src = _LoadMem(GPRClass, Size, NewSP, Size);
|
||||
StoreGPRRegister(X86State::REG_RCX, Src, GPRSize);
|
||||
StoreGPRRegister(X86State::REG_RCX, Src, Size);
|
||||
NewSP = _Add(NewSP, Constant);
|
||||
|
||||
Src = _LoadMem(GPRClass, Size, NewSP, Size);
|
||||
StoreGPRRegister(X86State::REG_RAX, Src, GPRSize);
|
||||
StoreGPRRegister(X86State::REG_RAX, Src, Size);
|
||||
NewSP = _Add(NewSP, Constant);
|
||||
|
||||
// Store the new stack pointer
|
||||
@@ -810,6 +814,7 @@ void OpDispatchBuilder::CALLOp(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[0].IsLiteral(), "Had wrong operand type");
|
||||
const uint64_t TargetRIP = Op->PC + Op->InstSize + Op->Src[0].Data.Literal.Value;
|
||||
|
||||
CalculateDeferredFlags();
|
||||
if (NextRIP != TargetRIP) {
|
||||
// Store the RIP
|
||||
_ExitFunction(NewRIP); // If we get here then leave the function now
|
||||
@@ -840,6 +845,7 @@ void OpDispatchBuilder::CALLAbsoluteOp(OpcodeArgs) {
|
||||
_StoreMem(GPRClass, Size, NewSP, ConstantPCReturn, Size);
|
||||
|
||||
// Store the RIP
|
||||
CalculateDeferredFlags();
|
||||
_ExitFunction(JMPPCOffset); // If we get here then leave the function now
|
||||
}
|
||||
|
||||
@@ -915,15 +921,11 @@ OrderedNode *OpDispatchBuilder::SelectCC(uint8_t OP, OrderedNode *TrueValue, Ord
|
||||
break;
|
||||
}
|
||||
case 0xA: { // JP - Jump if PF == 1
|
||||
auto Flag = GetRFLAG(FEXCore::X86State::RFLAG_PF_LOC);
|
||||
SrcCond = _Select(FEXCore::IR::COND_NEQ,
|
||||
Flag, ZeroConst, TrueValue, FalseValue);
|
||||
SrcCond = _Select(FEXCore::IR::COND_NEQ, LoadPF(), ZeroConst, TrueValue, FalseValue);
|
||||
break;
|
||||
}
|
||||
case 0xB: { // JNP - Jump if PF == 0
|
||||
auto Flag = GetRFLAG(FEXCore::X86State::RFLAG_PF_LOC);
|
||||
SrcCond = _Select(FEXCore::IR::COND_EQ,
|
||||
Flag, ZeroConst, TrueValue, FalseValue);
|
||||
SrcCond = _Select(FEXCore::IR::COND_EQ, LoadPF(), ZeroConst, TrueValue, FalseValue);
|
||||
break;
|
||||
}
|
||||
case 0xC: { // SF <> OF
|
||||
@@ -1128,6 +1130,7 @@ void OpDispatchBuilder::CondJUMPOp(OpcodeArgs) {
|
||||
Target &= 0xFFFFFFFFU;
|
||||
}
|
||||
|
||||
CalculateDeferredFlags();
|
||||
auto TrueBlock = JumpTargets.find(Target);
|
||||
auto FalseBlock = JumpTargets.find(Op->PC + Op->InstSize);
|
||||
|
||||
@@ -1273,6 +1276,7 @@ void OpDispatchBuilder::LoopOp(OpcodeArgs) {
|
||||
SrcCond = _And(SrcCond, ZF);
|
||||
}
|
||||
|
||||
CalculateDeferredFlags();
|
||||
auto TrueBlock = JumpTargets.find(Target);
|
||||
auto FalseBlock = JumpTargets.find(Op->PC + Op->InstSize);
|
||||
|
||||
@@ -1341,6 +1345,7 @@ void OpDispatchBuilder::JUMPOp(OpcodeArgs) {
|
||||
TargetRIP &= 0xFFFFFFFFU;
|
||||
}
|
||||
|
||||
CalculateDeferredFlags();
|
||||
// This is just an unconditional relative literal jump
|
||||
if (Multiblock) {
|
||||
auto JumpBlock = JumpTargets.find(TargetRIP);
|
||||
@@ -1379,6 +1384,7 @@ void OpDispatchBuilder::JUMPAbsoluteOp(OpcodeArgs) {
|
||||
// This uses ModRM to determine its location
|
||||
// No way to use this effectively in multiblock
|
||||
auto RIPOffset = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
CalculateDeferredFlags();
|
||||
|
||||
// Store the new RIP
|
||||
_ExitFunction(RIPOffset);
|
||||
@@ -1554,7 +1560,7 @@ void OpDispatchBuilder::SAHFOp(OpcodeArgs) {
|
||||
}
|
||||
void OpDispatchBuilder::LAHFOp(OpcodeArgs) {
|
||||
// Load the lower 8 bits of the Rflags register
|
||||
auto RFLAG = GetPackedRFLAG(true);
|
||||
auto RFLAG = GetPackedRFLAG(0xFF);
|
||||
|
||||
// Store the lower 8 bits of the rflags register in to AH
|
||||
StoreGPRRegister(X86State::REG_RAX, RFLAG, 1, 8);
|
||||
@@ -1804,15 +1810,7 @@ void OpDispatchBuilder::SHLOp(OpcodeArgs) {
|
||||
}
|
||||
const auto Size = GetSrcBitSize(Op);
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op
|
||||
if (Size == 64) {
|
||||
Src = _And(Src, _Constant(0x3F));
|
||||
}
|
||||
else {
|
||||
Src = _And(Src, _Constant(0x1F));
|
||||
}
|
||||
|
||||
OrderedNode *Result = _Lshl(Dest, Src);
|
||||
OrderedNode *Result = _Lshl(std::max<uint8_t>(4, GetSrcSize(Op)), Dest, Src);
|
||||
StoreResult(GPRClass, Op, Result, -1);
|
||||
|
||||
if (Size < 32) {
|
||||
@@ -1866,17 +1864,7 @@ void OpDispatchBuilder::SHROp(OpcodeArgs) {
|
||||
Src = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
}
|
||||
|
||||
const auto Size = GetSrcBitSize(Op);
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op
|
||||
if (Size == 64) {
|
||||
Src = _And(Src, _Constant(0x3F));
|
||||
}
|
||||
else {
|
||||
Src = _And(Src, _Constant(0x1F));
|
||||
}
|
||||
|
||||
auto ALUOp = _Lshr(Dest, Src);
|
||||
auto ALUOp = _Lshr(std::max<uint8_t>(4, GetSrcSize(Op)), Dest, Src);
|
||||
StoreResult(GPRClass, Op, ALUOp, -1);
|
||||
|
||||
if constexpr (SHR1Bit) {
|
||||
@@ -1985,20 +1973,23 @@ void OpDispatchBuilder::SHLDImmediateOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
if (Shift != 0) {
|
||||
OrderedNode *ShiftLeft = _Constant(Shift);
|
||||
auto ShiftRight = _Constant(Size - Shift);
|
||||
OrderedNode *Res{};
|
||||
if (Size < 32) {
|
||||
OrderedNode *ShiftLeft = _Constant(Shift);
|
||||
auto ShiftRight = _Constant(Size - Shift);
|
||||
|
||||
auto Tmp1 = _Lshl(Dest, ShiftLeft);
|
||||
Tmp1.first->Header.Size = 8;
|
||||
auto Tmp2 = _Lshr(Src, ShiftRight);
|
||||
auto Tmp1 = _Lshl(Dest, ShiftLeft);
|
||||
Tmp1.first->Header.Size = 8;
|
||||
auto Tmp2 = _Lshr(Src, ShiftRight);
|
||||
|
||||
OrderedNode *Res = _Or(Tmp1, Tmp2);
|
||||
Res = _Or(Tmp1, Tmp2);
|
||||
}
|
||||
else {
|
||||
// 32-bit and 64-bit SHLD behaves like an EXTR where the lower bits are filled from the source.
|
||||
Res = _Extr(Dest, Src, Size - Shift);
|
||||
}
|
||||
|
||||
StoreResult(GPRClass, Op, Res, -1);
|
||||
|
||||
if (Size != 64) {
|
||||
Res = _Bfe(Size, 0, Res);
|
||||
}
|
||||
GenerateFlags_ShiftLeftImmediate(Op, Res, Dest, Shift);
|
||||
}
|
||||
else if (Shift == 0 && Size == 32) {
|
||||
@@ -2082,21 +2073,25 @@ void OpDispatchBuilder::SHRDImmediateOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
if (Shift != 0) {
|
||||
OrderedNode *ShiftRight = _Constant(Shift);
|
||||
auto ShiftLeft = _Constant(Size - Shift);
|
||||
|
||||
auto Tmp1 = _Lshr(Dest, ShiftRight);
|
||||
auto Tmp2 = _Lshl(Src, ShiftLeft);
|
||||
Tmp2.first->Header.Size = 8;
|
||||
OrderedNode *Res{};
|
||||
if (Size < 32) {
|
||||
OrderedNode *ShiftRight = _Constant(Shift);
|
||||
auto ShiftLeft = _Constant(Size - Shift);
|
||||
|
||||
OrderedNode *Res = _Or(Tmp1, Tmp2);
|
||||
auto Tmp1 = _Lshr(Dest, ShiftRight);
|
||||
auto Tmp2 = _Lshl(Src, ShiftLeft);
|
||||
Tmp2.first->Header.Size = 8;
|
||||
|
||||
Res = _Or(Tmp1, Tmp2);
|
||||
}
|
||||
else {
|
||||
// 32-bit and 64-bit SHRD behaves like an EXTR where the upper bits are filled from the source.
|
||||
Res = _Extr(Src, Dest, Shift);
|
||||
}
|
||||
|
||||
StoreResult(GPRClass, Op, Res, -1);
|
||||
|
||||
if (Size != 64) {
|
||||
Res = _Bfe(Size, 0, Res);
|
||||
}
|
||||
GenerateFlags_ShiftRightImmediate(Op, Res, Dest, Shift);
|
||||
GenerateFlags_ShiftRightDoubleImmediate(Op, Res, Dest, Shift);
|
||||
}
|
||||
else if (Shift == 0 && Size == 32) {
|
||||
// Ensure Zext still occurs
|
||||
@@ -2117,18 +2112,11 @@ void OpDispatchBuilder::ASHROp(OpcodeArgs) {
|
||||
Src = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
}
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op
|
||||
if (Size == 64) {
|
||||
Src = _And(Src, _Constant(Size, 0x3F));
|
||||
} else {
|
||||
Src = _And(Src, _Constant(Size, 0x1F));
|
||||
}
|
||||
|
||||
if (Size < 32) {
|
||||
Dest = _Sbfe(Size, 0, Dest);
|
||||
}
|
||||
|
||||
OrderedNode *Result = _Ashr(Dest, Src);
|
||||
OrderedNode *Result = _Ashr(std::max<uint8_t>(4, GetSrcSize(Op)), Dest, Src);
|
||||
StoreResult(GPRClass, Op, Result, -1);
|
||||
|
||||
if constexpr (SHR1Bit) {
|
||||
@@ -2412,29 +2400,20 @@ void OpDispatchBuilder::BMI2Shift(OpcodeArgs) {
|
||||
|
||||
auto* Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
auto* Shift = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, -1);
|
||||
const auto OperandSize = GetSrcBitSize(Op);
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op
|
||||
auto SanitizedShift = [&] {
|
||||
if (OperandSize == 64) {
|
||||
return _And(Shift, _Constant(0x3F));
|
||||
} else {
|
||||
return _And(Shift, _Constant(0x1F));
|
||||
}
|
||||
}();
|
||||
const auto Size = GetSrcSize(Op);
|
||||
|
||||
auto* Result = [&]() -> OrderedNode* {
|
||||
// SARX
|
||||
if (Op->OP == 0x6F7) {
|
||||
return _Ashr(Src, SanitizedShift);
|
||||
return _Ashr(Size, Src, Shift);
|
||||
}
|
||||
// SHLX
|
||||
if (Op->OP == 0x5F7) {
|
||||
return _Lshl(Src, SanitizedShift);
|
||||
return _Lshl(Size, Src, Shift);
|
||||
}
|
||||
|
||||
// SHRX
|
||||
return _Lshr(Src, SanitizedShift);
|
||||
return _Lshr(Size, Src, Shift);
|
||||
}();
|
||||
|
||||
StoreResult(GPRClass, Op, Result, -1);
|
||||
@@ -2631,19 +2610,12 @@ void OpDispatchBuilder::RCROp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC);
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op
|
||||
if (Size == 64) {
|
||||
Src = _And(Src, _Constant(Size, 0x3F));
|
||||
} else {
|
||||
Src = _And(Src, _Constant(Size, 0x1F));
|
||||
}
|
||||
|
||||
// Res = Src >> Shift
|
||||
OrderedNode *Res = _Lshr(Dest, Src);
|
||||
|
||||
// Res |= (Src << (Size - Shift + 1));
|
||||
OrderedNode *SrcShl = _Sub(_Constant(Size, Size + 1), Src);
|
||||
auto TmpHigher = _Lshl(Dest, SrcShl);
|
||||
auto TmpHigher = _Lshl(GetSrcSize(Op), Dest, SrcShl);
|
||||
|
||||
auto One = _Constant(Size, 1);
|
||||
auto Zero = _Constant(Size, 0);
|
||||
@@ -2669,7 +2641,7 @@ void OpDispatchBuilder::RCROp(OpcodeArgs) {
|
||||
|
||||
// CF only changes if we actually shifted
|
||||
// Our new CF will be bit (Shift - 1) of the source
|
||||
auto NewCF = _Lshr(Dest, _Sub(Src, One));
|
||||
auto NewCF = _Bfe(1, 0, _Lshr(Dest, _Sub(Src, One)));
|
||||
CompareResult = _Select(FEXCore::IR::COND_UGE,
|
||||
Src, One,
|
||||
NewCF, CF);
|
||||
@@ -2701,17 +2673,62 @@ void OpDispatchBuilder::RCRSmallerOp(OpcodeArgs) {
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op
|
||||
Src = _And(Src, _Constant(Size, 0x1F));
|
||||
|
||||
OrderedNode *Tmp = _Constant(64, 0);
|
||||
OrderedNode *Tmp{};
|
||||
|
||||
// Insert the incoming value across the temporary 64bit source
|
||||
// Make sure to insert at <BitSize> + 1 offsets
|
||||
// We need to cover 32bits plus the amount that could rotate in
|
||||
for (size_t i = 0; i < (32 + Size + 1); i += (Size + 1)) {
|
||||
// Insert incoming value
|
||||
Tmp = _Bfi(8, Size, i, Tmp, Dest);
|
||||
|
||||
// Insert CF
|
||||
Tmp = _Bfi(8, 1, i + Size, Tmp, CF);
|
||||
if (Size == 8) {
|
||||
// 8-bit optimal cascade
|
||||
// Cascade: 0
|
||||
// Data: -> [7:0]
|
||||
// CF: -> [8:8]
|
||||
// Cascade: 1
|
||||
// Data: -> [16:9]
|
||||
// CF: -> [17:17]
|
||||
// Cascade: 2
|
||||
// Data: -> [25:18]
|
||||
// CF: -> [26:26]
|
||||
// Cascade: 3
|
||||
// Data: -> [34:27]
|
||||
// CF: -> [35:35]
|
||||
// Cascade: 4
|
||||
// Data: -> [43:36]
|
||||
// CF: -> [44:44]
|
||||
|
||||
// Insert CF, Destination already at [7:0]
|
||||
Tmp = _Bfi(8, 1, 8, Dest, CF);
|
||||
|
||||
// First Cascade, copies 9 bits from itself.
|
||||
Tmp = _Bfi(8, 9, 9, Tmp, Tmp);
|
||||
|
||||
// Second cascade, copies 18 bits from itself.
|
||||
Tmp = _Bfi(8, 18, 18, Tmp, Tmp);
|
||||
|
||||
// Final cascade, copies 9 bits again from itself.
|
||||
Tmp = _Bfi(8, 9, 36, Tmp, Tmp);
|
||||
}
|
||||
else {
|
||||
// 16-bit optimal cascade
|
||||
// Cascade: 0
|
||||
// Data: -> [15:0]
|
||||
// CF: -> [16:16]
|
||||
// Cascade: 1
|
||||
// Data: -> [32:17]
|
||||
// CF: -> [33:33]
|
||||
// Cascade: 2
|
||||
// Data: -> [49:34]
|
||||
// CF: -> [50:50]
|
||||
|
||||
// Insert CF, Destination already at [15:0]
|
||||
Tmp = _Bfi(8, 1, 16, Dest, CF);
|
||||
|
||||
// First Cascade, copies 17 bits from itself.
|
||||
Tmp = _Bfi(8, 17, 17, Tmp, Tmp);
|
||||
|
||||
// Final Cascade, copies 17 bits from itself again.
|
||||
Tmp = _Bfi(8, 17, 34, Tmp, Tmp);
|
||||
}
|
||||
|
||||
// Entire bitfield has been setup
|
||||
@@ -2723,7 +2740,7 @@ void OpDispatchBuilder::RCRSmallerOp(OpcodeArgs) {
|
||||
// CF only changes if we actually shifted
|
||||
// Our new CF will be bit (Shift - 1) of the source
|
||||
auto One = _Constant(Size, 1);
|
||||
auto NewCF = _Lshr(Tmp, _Sub(Src, One));
|
||||
auto NewCF = _Bfe(1, 0, _Lshr(Tmp, _Sub(Src, One)));
|
||||
auto CompareResult = _Select(FEXCore::IR::COND_UGE,
|
||||
Src, One,
|
||||
NewCF, CF);
|
||||
@@ -2780,15 +2797,8 @@ void OpDispatchBuilder::RCLOp(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC);
|
||||
|
||||
// x86 masks the shift by 0x3F or 0x1F depending on size of op
|
||||
if (Size == 64) {
|
||||
Src = _And(Src, _Constant(Size, 0x3F));
|
||||
} else {
|
||||
Src = _And(Src, _Constant(Size, 0x1F));
|
||||
}
|
||||
|
||||
// Res = Src << Shift
|
||||
OrderedNode *Res = _Lshl(Dest, Src);
|
||||
OrderedNode *Res = _Lshl(GetSrcSize(Op), Dest, Src);
|
||||
|
||||
// Res |= (Src << (Size - Shift + 1));
|
||||
OrderedNode *SrcShl = _Sub(_Constant(Size, Size + 1), Src);
|
||||
@@ -2819,7 +2829,7 @@ void OpDispatchBuilder::RCLOp(OpcodeArgs) {
|
||||
{
|
||||
// CF only changes if we actually shifted
|
||||
// Our new CF will be bit (Shift - 1) of the source
|
||||
auto NewCF = _Lshr(Dest, _Sub(_Constant(Size, Size), Src));
|
||||
auto NewCF = _Bfe(1, 0, _Lshr(Dest, _Sub(_Constant(Size, Size), Src)));
|
||||
CompareResult = _Select(FEXCore::IR::COND_UGE,
|
||||
Src, One,
|
||||
NewCF, CF);
|
||||
@@ -2878,7 +2888,7 @@ void OpDispatchBuilder::RCLSmallerOp(OpcodeArgs) {
|
||||
// Our new CF is now at the bit position that we are shifting
|
||||
// Either 0 if CF hasn't changed (CF is living in bit 0)
|
||||
// or higher
|
||||
auto NewCF = _Ror(Tmp, _Sub(_Constant(63), Src));
|
||||
auto NewCF = _Bfe(1, 0, _Ror(Tmp, _Sub(_Constant(63), Src)));
|
||||
auto CompareResult = _Select(FEXCore::IR::COND_UGE,
|
||||
Src, _Constant(1),
|
||||
NewCF, CF);
|
||||
@@ -2955,7 +2965,7 @@ void OpDispatchBuilder::BTOp(OpcodeArgs) {
|
||||
// Now shift in to the correct bit location
|
||||
Result = _Lshr(Result, BitSelect);
|
||||
}
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(Result);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(1, 0, Result));
|
||||
}
|
||||
|
||||
template<uint32_t SrcIndex>
|
||||
@@ -3033,7 +3043,7 @@ void OpDispatchBuilder::BTROp(OpcodeArgs) {
|
||||
_StoreMemAutoTSO(GPRClass, 1, MemoryLocation, Value, 1);
|
||||
}
|
||||
}
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(Result);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(1, 0, Result));
|
||||
}
|
||||
|
||||
template<uint32_t SrcIndex>
|
||||
@@ -3107,7 +3117,7 @@ void OpDispatchBuilder::BTSOp(OpcodeArgs) {
|
||||
_StoreMemAutoTSO(GPRClass, 1, MemoryLocation, Value, 1);
|
||||
}
|
||||
}
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(Result);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(1, 0, Result));
|
||||
}
|
||||
|
||||
template<uint32_t SrcIndex>
|
||||
@@ -3181,7 +3191,7 @@ void OpDispatchBuilder::BTCOp(OpcodeArgs) {
|
||||
_StoreMemAutoTSO(GPRClass, 1, MemoryLocation, Value, 1);
|
||||
}
|
||||
}
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(Result);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(1, 0, Result));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::IMUL1SrcOp(OpcodeArgs) {
|
||||
@@ -3407,12 +3417,14 @@ void OpDispatchBuilder::DAAOp(OpcodeArgs) {
|
||||
auto AL = LoadGPRRegister(X86State::REG_RAX, 1);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(0));
|
||||
CalculateDeferredFlags();
|
||||
|
||||
auto Cond = _Or(AF, _Select(FEXCore::IR::COND_UGT, _And(AL, _Constant(0xF)), _Constant(9), _Constant(1), _Constant(0)));
|
||||
auto FalseBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
auto TrueBlock = CreateNewCodeBlockAfter(FalseBlock);
|
||||
auto EndBlock = CreateNewCodeBlockAfter(TrueBlock);
|
||||
|
||||
CalculateDeferredFlags();
|
||||
_CondJump(Cond, TrueBlock, FalseBlock);
|
||||
SetCurrentCodeBlock(FalseBlock);
|
||||
{
|
||||
@@ -3425,8 +3437,13 @@ void OpDispatchBuilder::DAAOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RAX, NewAL, 1);
|
||||
CalculateDeferredFlags();
|
||||
auto NewCF = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC);
|
||||
// XXX: I don't think this is correct. Needs Investigation.
|
||||
// The `CF` variable is the original CF from the start of the operation
|
||||
// The `NewCF` will be _Constant(0) stored aboved.
|
||||
// So Or(CF, _Constant(0)) ill mean CF gets updated to the old value in the true case?
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Or(CF, NewCF));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(1));
|
||||
CalculateDeferredFlags();
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(EndBlock);
|
||||
@@ -3439,6 +3456,7 @@ void OpDispatchBuilder::DAAOp(OpcodeArgs) {
|
||||
SetCurrentCodeBlock(FalseBlock);
|
||||
{
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(0));
|
||||
CalculateDeferredFlags();
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(TrueBlock);
|
||||
@@ -3448,6 +3466,7 @@ void OpDispatchBuilder::DAAOp(OpcodeArgs) {
|
||||
auto NewAL = _Add(AL, _Constant(0x60));
|
||||
StoreGPRRegister(X86State::REG_RAX, NewAL, 1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(1));
|
||||
CalculateDeferredFlags();
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(EndBlock);
|
||||
@@ -3456,10 +3475,7 @@ void OpDispatchBuilder::DAAOp(OpcodeArgs) {
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(_Select(FEXCore::IR::COND_UGE, _And(AL, _Constant(0x80)), _Constant(0), _Constant(1), _Constant(0)));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(_Select(FEXCore::IR::COND_EQ, _And(AL, _Constant(0xFF)), _Constant(0), _Constant(1), _Constant(0)));
|
||||
auto EightBitMask = _Constant(0xFF);
|
||||
auto PopCountOp = _Popcount(_And(AL, EightBitMask));
|
||||
auto XorOp = _Xor(PopCountOp, _Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(XorOp);
|
||||
CalculatePFUncheckedABI(AL);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::DASOp(OpcodeArgs) {
|
||||
@@ -3469,12 +3485,14 @@ void OpDispatchBuilder::DASOp(OpcodeArgs) {
|
||||
auto AL = LoadGPRRegister(X86State::REG_RAX, 1);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(0));
|
||||
CalculateDeferredFlags();
|
||||
|
||||
auto Cond = _Or(AF, _Select(FEXCore::IR::COND_UGT, _And(AL, _Constant(0xf)), _Constant(9), _Constant(1), _Constant(0)));
|
||||
auto FalseBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
auto TrueBlock = CreateNewCodeBlockAfter(FalseBlock);
|
||||
auto EndBlock = CreateNewCodeBlockAfter(TrueBlock);
|
||||
|
||||
CalculateDeferredFlags();
|
||||
_CondJump(Cond, TrueBlock, FalseBlock);
|
||||
SetCurrentCodeBlock(FalseBlock);
|
||||
{
|
||||
@@ -3487,8 +3505,13 @@ void OpDispatchBuilder::DASOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RAX, NewAL, 1);
|
||||
CalculateDeferredFlags();
|
||||
auto NewCF = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC);
|
||||
// XXX: I don't think this is correct. Needs Investigation.
|
||||
// The `CF` variable is the original CF from the start of the operation
|
||||
// The `NewCF` will be _Constant(0) stored aboved.
|
||||
// So Or(CF, _Constant(0)) ill mean CF gets updated to the old value in the true case?
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Or(CF, NewCF));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(1));
|
||||
CalculateDeferredFlags();
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(EndBlock);
|
||||
@@ -3501,6 +3524,7 @@ void OpDispatchBuilder::DASOp(OpcodeArgs) {
|
||||
SetCurrentCodeBlock(FalseBlock);
|
||||
{
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(0));
|
||||
CalculateDeferredFlags();
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(TrueBlock);
|
||||
@@ -3509,6 +3533,7 @@ void OpDispatchBuilder::DASOp(OpcodeArgs) {
|
||||
auto NewAL = _Sub(AL, _Constant(0x60));
|
||||
StoreGPRRegister(X86State::REG_RAX, NewAL, 1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(1));
|
||||
CalculateDeferredFlags();
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(EndBlock);
|
||||
@@ -3516,13 +3541,12 @@ void OpDispatchBuilder::DASOp(OpcodeArgs) {
|
||||
AL = LoadGPRRegister(X86State::REG_RAX, 1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(_Select(FEXCore::IR::COND_UGE, _And(AL, _Constant(0x80)), _Constant(0), _Constant(1), _Constant(0)));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(_Select(FEXCore::IR::COND_EQ, _And(AL, _Constant(0xFF)), _Constant(0), _Constant(1), _Constant(0)));
|
||||
auto EightBitMask = _Constant(0xFF);
|
||||
auto PopCountOp = _Popcount(_And(AL, EightBitMask));
|
||||
auto XorOp = _Xor(PopCountOp, _Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(XorOp);
|
||||
CalculatePFUncheckedABI(AL);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AAAOp(OpcodeArgs) {
|
||||
InvalidateDeferredFlags();
|
||||
|
||||
auto AF = GetRFLAG(FEXCore::X86State::RFLAG_AF_LOC);
|
||||
auto AL = LoadGPRRegister(X86State::REG_RAX, 1);
|
||||
auto AX = LoadGPRRegister(X86State::REG_RAX, 2);
|
||||
@@ -3539,6 +3563,7 @@ void OpDispatchBuilder::AAAOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RAX, NewAX, 2);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(0));
|
||||
CalculateDeferredFlags();
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(TrueBlock);
|
||||
@@ -3548,12 +3573,15 @@ void OpDispatchBuilder::AAAOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RAX, Result, 2);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(1));
|
||||
CalculateDeferredFlags();
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(EndBlock);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AASOp(OpcodeArgs) {
|
||||
InvalidateDeferredFlags();
|
||||
|
||||
auto AF = GetRFLAG(FEXCore::X86State::RFLAG_AF_LOC);
|
||||
auto AL = LoadGPRRegister(X86State::REG_RAX, 1);
|
||||
auto AX = LoadGPRRegister(X86State::REG_RAX, 2);
|
||||
@@ -3570,6 +3598,7 @@ void OpDispatchBuilder::AASOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RAX, NewAX, 2);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(0));
|
||||
CalculateDeferredFlags();
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(TrueBlock);
|
||||
@@ -3580,12 +3609,15 @@ void OpDispatchBuilder::AASOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RAX, Result, 2);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(1));
|
||||
CalculateDeferredFlags();
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(EndBlock);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AAMOp(OpcodeArgs) {
|
||||
InvalidateDeferredFlags();
|
||||
|
||||
auto AL = LoadGPRRegister(X86State::REG_RAX, 1);
|
||||
auto Imm8 = _Constant(Op->Src[0].Data.Literal.Value & 0xFF);
|
||||
auto UDivOp = _UDiv(AL, Imm8);
|
||||
@@ -3598,13 +3630,12 @@ void OpDispatchBuilder::AAMOp(OpcodeArgs) {
|
||||
AL = LoadGPRRegister(X86State::REG_RAX, 1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(_Select(FEXCore::IR::COND_UGE, _And(AL, _Constant(0x80)), _Constant(0), _Constant(1), _Constant(0)));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(_Select(FEXCore::IR::COND_EQ, _And(AL, _Constant(0xFF)), _Constant(0), _Constant(1), _Constant(0)));
|
||||
auto EightBitMask = _Constant(0xFF);
|
||||
auto PopCountOp = _Popcount(_And(AL, EightBitMask));
|
||||
auto XorOp = _Xor(PopCountOp, _Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(XorOp);
|
||||
CalculatePFUncheckedABI(AL);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AADOp(OpcodeArgs) {
|
||||
InvalidateDeferredFlags();
|
||||
|
||||
auto AL = LoadGPRRegister(X86State::REG_RAX, 1);
|
||||
auto AH = _Lshr(LoadGPRRegister(X86State::REG_RAX, 2), _Constant(8));
|
||||
auto Imm8 = _Constant(Op->Src[0].Data.Literal.Value & 0xFF);
|
||||
@@ -3616,10 +3647,7 @@ void OpDispatchBuilder::AADOp(OpcodeArgs) {
|
||||
AL = LoadGPRRegister(X86State::REG_RAX, 1);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(_Select(FEXCore::IR::COND_UGE, _And(AL, _Constant(0x80)), _Constant(0), _Constant(1), _Constant(0)));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(_Select(FEXCore::IR::COND_EQ, _And(AL, _Constant(0xFF)), _Constant(0), _Constant(1), _Constant(0)));
|
||||
auto EightBitMask = _Constant(0xFF);
|
||||
auto PopCountOp = _Popcount(_And(AL, EightBitMask));
|
||||
auto XorOp = _Xor(PopCountOp, _Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(XorOp);
|
||||
CalculatePFUncheckedABI(AL);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::XLATOp(OpcodeArgs) {
|
||||
@@ -4020,6 +4048,7 @@ void OpDispatchBuilder::CMPSOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RSI, Dest_RSI);
|
||||
|
||||
OrderedNode *ZF = GetRFLAG(FEXCore::X86State::RFLAG_ZF_LOC);
|
||||
CalculateDeferredFlags();
|
||||
InternalCondJump = _CondJump(ZF, {REPE ? COND_NEQ : COND_EQ});
|
||||
|
||||
// Jump back to the start if we have more work to do
|
||||
@@ -4238,6 +4267,7 @@ void OpDispatchBuilder::SCASOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RDI, TailDest_RDI);
|
||||
|
||||
OrderedNode *ZF = GetRFLAG(FEXCore::X86State::RFLAG_ZF_LOC);
|
||||
CalculateDeferredFlags();
|
||||
InternalCondJump = _CondJump(ZF, {REPE ? COND_NEQ : COND_EQ});
|
||||
|
||||
// Jump back to the start if we have more work to do
|
||||
@@ -4269,7 +4299,7 @@ void OpDispatchBuilder::BSWAPOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::PUSHFOp(OpcodeArgs) {
|
||||
const uint8_t Size = GetSrcSize(Op);
|
||||
|
||||
OrderedNode *Src = GetPackedRFLAG(false);
|
||||
OrderedNode *Src = GetPackedRFLAG();
|
||||
if (Size != 8) {
|
||||
Src = _Bfe(Size * 8, 0, Src);
|
||||
}
|
||||
@@ -4659,18 +4689,16 @@ void OpDispatchBuilder::CMPXCHGPairOp(OpcodeArgs) {
|
||||
OrderedNode *Result_Upper = _ExtractElementPair(CASResult, 1);
|
||||
|
||||
// Set ZF if memory result was expected
|
||||
OrderedNode *EOR_Lower = _Xor(Result_Lower, Expected_Lower);
|
||||
OrderedNode *EOR_Upper = _Xor(Result_Upper, Expected_Upper);
|
||||
OrderedNode *Orr_Result = _Or(EOR_Lower, EOR_Upper);
|
||||
|
||||
auto OneConst = _Constant(1);
|
||||
auto ZeroConst = _Constant(0);
|
||||
|
||||
OrderedNode *ZFResult = _Select(FEXCore::IR::COND_EQ,
|
||||
Orr_Result, ZeroConst,
|
||||
CASResult, Expected,
|
||||
OneConst, ZeroConst);
|
||||
|
||||
// Set ZF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(ZFResult);
|
||||
CalculateDeferredFlags();
|
||||
|
||||
auto CondJump = _CondJump(ZFResult);
|
||||
|
||||
@@ -4707,8 +4735,6 @@ void OpDispatchBuilder::CreateJumpBlocks(fextl::vector<FEXCore::Frontend::Decode
|
||||
void OpDispatchBuilder::BeginFunction(uint64_t RIP, fextl::vector<FEXCore::Frontend::Decoder::DecodedBlocks> const *Blocks) {
|
||||
Entry = RIP;
|
||||
auto IRHeader = _IRHeader(InvalidNode, 0);
|
||||
Current_Header = IRHeader.first;
|
||||
Current_HeaderNode = IRHeader;
|
||||
CreateJumpBlocks(Blocks);
|
||||
|
||||
auto Block = GetNewJumpBlock(RIP);
|
||||
@@ -4737,6 +4763,7 @@ void OpDispatchBuilder::Finalize() {
|
||||
|
||||
// We haven't emitted. Dump out to the dispatcher
|
||||
SetCurrentCodeBlock(Handler.second.BlockEntry);
|
||||
CalculateDeferredFlags();
|
||||
_ExitFunction(_EntrypointOffset(Handler.first - Entry, GPRSize));
|
||||
}
|
||||
}
|
||||
@@ -4942,8 +4969,19 @@ OrderedNode *OpDispatchBuilder::LoadSource_WithOpSize(FEXCore::IR::RegisterClass
|
||||
LoadableType = true;
|
||||
}
|
||||
else if (Operand.IsSIB()) {
|
||||
OrderedNode *Tmp {};
|
||||
if (Operand.Data.SIB.Index != FEXCore::X86State::REG_INVALID) {
|
||||
const bool IsVSIB = (Op->Flags & X86Tables::DecodeFlags::FLAG_VSIB_BYTE) != 0;
|
||||
OrderedNode *Tmp{};
|
||||
|
||||
// NOTE: VSIB cannot have the index * scale portion calculated ahead of time,
|
||||
// since the index in this case is a vector. So, we can't just apply the scale
|
||||
// to it, since this needs to be applied to each element in the index register
|
||||
// after said element has been sign extended. So, we pass this through for the
|
||||
// instruction implementation to handle.
|
||||
//
|
||||
// What we do handle though, is the applying the displacement value to
|
||||
// the base register (if a base register is provided), since this is a
|
||||
// part of the address calculation that can be done ahead of time.
|
||||
if (Operand.Data.SIB.Index != FEXCore::X86State::REG_INVALID && !IsVSIB) {
|
||||
Tmp = LoadGPRRegister(Operand.Data.SIB.Index, GPRSize);
|
||||
|
||||
if (Operand.Data.SIB.Scale != 1) {
|
||||
@@ -5106,7 +5144,6 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
if (OpSize != VectorSize) {
|
||||
// Partial writes can come from FPRs.
|
||||
// TODO: Fix the instructions doing partial writes rather than dealing with it here.
|
||||
auto SrcVector = LoadXMMRegister(gprIndex);
|
||||
|
||||
LOGMAN_THROW_A_FMT(Class != IR::GPRClass, "Partial writes from GPR not allowed. Instruction: {}",
|
||||
Op->TableInfo->Name);
|
||||
@@ -5116,6 +5153,7 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
|
||||
if (VectorSize == Core::CPUState::XMM_AVX_REG_SIZE && OpSize == Core::CPUState::XMM_SSE_REG_SIZE) {
|
||||
Result = _VMov(OpSize, Src);
|
||||
} else {
|
||||
auto SrcVector = LoadXMMRegister(gprIndex);
|
||||
Result = _VInsElement(VectorSize, OpSize, 0, 0, SrcVector, Src);
|
||||
}
|
||||
}
|
||||
@@ -5315,11 +5353,25 @@ void OpDispatchBuilder::ALUOpImpl(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCor
|
||||
else {
|
||||
Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
|
||||
auto ALUOp = _Add(Dest, Src);
|
||||
// Overwrite our IR's op type
|
||||
ALUOp.first->Header.Op = ALUIROp;
|
||||
/* On x86, the canonical way to zero a register is XOR with itself...
|
||||
* because modern x86 detects this pattern in hardware. arm64 does not
|
||||
* detect this pattern, we should do it like the x86 hardware would. On
|
||||
* arm64, "mov x0, #0" is faster than "eor x0, x0, x0". Additionally this
|
||||
* lets more constant folding kick in for flags.
|
||||
*/
|
||||
if (ALUIROp == FEXCore::IR::IROps::OP_XOR &&
|
||||
Op->Dest.IsGPR() && Op->Src[0].IsGPR() &&
|
||||
Op->Dest.Data.GPR == Op->Src[0].Data.GPR) {
|
||||
|
||||
Result = _Constant(0);
|
||||
} else {
|
||||
auto ALUOp = _Add(Dest, Src);
|
||||
// Overwrite our IR's op type
|
||||
ALUOp.first->Header.Op = ALUIROp;
|
||||
|
||||
Result = ALUOp;
|
||||
}
|
||||
|
||||
Result = ALUOp;
|
||||
StoreResult(GPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
@@ -5432,6 +5484,7 @@ void OpDispatchBuilder::INTOp(OpcodeArgs) {
|
||||
|
||||
if (Op->OP == 0xCE) { // Conditional to only break if Overflow == 1
|
||||
auto Flag = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC);
|
||||
CalculateDeferredFlags();
|
||||
|
||||
// If condition doesn't hold then keep going
|
||||
auto CondJump = _CondJump(Flag, {COND_EQ});
|
||||
|
||||
+213
-62
@@ -28,13 +28,6 @@ class OpDispatchBuilder final : public IREmitter {
|
||||
friend class FEXCore::IR::Pass;
|
||||
friend class FEXCore::IR::PassManager;
|
||||
|
||||
enum class SelectionFlag {
|
||||
Nothing, // must rely on x86 flags
|
||||
CMP, // flags were set by a CMP between flagsOpDest/flagsOpDestSigned and flagsOpSrc/flagsOpSrcSigned with flagsOpSize size
|
||||
AND, // flags were set by an AND/TEST, flagsOpDest contains the resulting value of flagsOpSize size
|
||||
FCMP, // flags were set by a ucomis* / comis*
|
||||
};
|
||||
|
||||
public:
|
||||
enum class FlagsGenerationType : uint8_t {
|
||||
TYPE_NONE,
|
||||
@@ -49,6 +42,7 @@ public:
|
||||
TYPE_LSHLI,
|
||||
TYPE_LSHR,
|
||||
TYPE_LSHRI,
|
||||
TYPE_LSHRDI,
|
||||
TYPE_ASHR,
|
||||
TYPE_ASHRI,
|
||||
TYPE_ROR,
|
||||
@@ -68,25 +62,6 @@ public:
|
||||
TYPE_RDRAND,
|
||||
};
|
||||
|
||||
SelectionFlag flagsOp{};
|
||||
uint8_t flagsOpSize{};
|
||||
OrderedNode* flagsOpDest{};
|
||||
OrderedNode* flagsOpSrc{};
|
||||
OrderedNode* flagsOpDestSigned{};
|
||||
OrderedNode* flagsOpSrcSigned{};
|
||||
|
||||
FEXCore::Context::ContextImpl *CTX{};
|
||||
|
||||
// Used during new op bringup
|
||||
bool ShouldDump {false};
|
||||
|
||||
struct JumpTargetInfo {
|
||||
OrderedNode* BlockEntry;
|
||||
bool HaveEmitted;
|
||||
};
|
||||
|
||||
fextl::map<uint64_t, JumpTargetInfo> JumpTargets;
|
||||
|
||||
OrderedNode* GetNewJumpBlock(uint64_t RIP) {
|
||||
auto it = JumpTargets.find(RIP);
|
||||
LOGMAN_THROW_A_FMT(it != JumpTargets.end(), "Couldn't find block generated for 0x{:x}", RIP);
|
||||
@@ -108,6 +83,10 @@ public:
|
||||
|
||||
void StartNewBlock() {
|
||||
flagsOp = SelectionFlag::Nothing;
|
||||
|
||||
// If we loaded flags but didn't change them, invalidate the cached copy and move on.
|
||||
// Changes get stored out by CalculateDeferredFlags.
|
||||
CachedNZCV = nullptr;
|
||||
}
|
||||
|
||||
bool FinishOp(uint64_t NextRIP, bool LastOp) {
|
||||
@@ -182,6 +161,12 @@ public:
|
||||
bool HadDecodeFailure() const { return DecodeFailure; }
|
||||
bool NeedsBlockEnder() const { return NeedsBlockEnd; }
|
||||
|
||||
void ResetHandledLock() { HandledLock = false; }
|
||||
bool HasHandledLock() const { return HandledLock; }
|
||||
|
||||
void SetDumpIR(bool DumpIR) { ShouldDump = DumpIR; }
|
||||
bool ShouldDumpIR() const { return ShouldDump; }
|
||||
|
||||
void BeginFunction(uint64_t RIP, fextl::vector<FEXCore::Frontend::Decoder::DecodedBlocks> const *Blocks);
|
||||
void Finalize();
|
||||
|
||||
@@ -827,16 +812,54 @@ public:
|
||||
void InvalidOp(OpcodeArgs);
|
||||
|
||||
void SetPackedRFLAG(bool Lower8, OrderedNode *Src);
|
||||
OrderedNode *GetPackedRFLAG(bool Lower8);
|
||||
OrderedNode *GetPackedRFLAG(uint32_t FlagsMask = ~0U);
|
||||
|
||||
void SetMultiblock(bool _Multiblock) { Multiblock = _Multiblock; }
|
||||
|
||||
bool HandledLock = false;
|
||||
private:
|
||||
enum class SelectionFlag {
|
||||
Nothing, // must rely on x86 flags
|
||||
CMP, // flags were set by a CMP between flagsOpDest/flagsOpDestSigned and flagsOpSrc/flagsOpSrcSigned with flagsOpSize size
|
||||
AND, // flags were set by an AND/TEST, flagsOpDest contains the resulting value of flagsOpSize size
|
||||
FCMP, // flags were set by a ucomis* / comis*
|
||||
};
|
||||
|
||||
struct JumpTargetInfo {
|
||||
OrderedNode* BlockEntry;
|
||||
bool HaveEmitted;
|
||||
};
|
||||
|
||||
FEXCore::Context::ContextImpl *CTX{};
|
||||
|
||||
SelectionFlag flagsOp{};
|
||||
uint8_t flagsOpSize{};
|
||||
OrderedNode* flagsOpDest{};
|
||||
OrderedNode* flagsOpSrc{};
|
||||
OrderedNode* flagsOpDestSigned{};
|
||||
OrderedNode* flagsOpSrcSigned{};
|
||||
|
||||
static bool IsNZCV(unsigned BitOffset) {
|
||||
switch (BitOffset) {
|
||||
case FEXCore::X86State::RFLAG_CF_LOC:
|
||||
case FEXCore::X86State::RFLAG_ZF_LOC:
|
||||
case FEXCore::X86State::RFLAG_SF_LOC:
|
||||
case FEXCore::X86State::RFLAG_OF_LOC:
|
||||
return true;
|
||||
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode* CachedNZCV = {};
|
||||
uint32_t PossiblySetNZCVBits = 0;
|
||||
|
||||
fextl::map<uint64_t, JumpTargetInfo> JumpTargets;
|
||||
bool HandledLock{false};
|
||||
bool DecodeFailure{false};
|
||||
bool NeedsBlockEnd{false};
|
||||
FEXCore::IR::IROp_IRHeader *Current_Header{};
|
||||
OrderedNode *Current_HeaderNode{};
|
||||
// Used during new op bringup
|
||||
bool ShouldDump{false};
|
||||
|
||||
void ALUOpImpl(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp, bool RequiresMask);
|
||||
|
||||
@@ -1038,19 +1061,123 @@ private:
|
||||
[[nodiscard]] uint32_t GetDstBitSize(X86Tables::DecodedOp Op) const;
|
||||
[[nodiscard]] uint32_t GetSrcBitSize(X86Tables::DecodedOp Op) const;
|
||||
|
||||
static inline constexpr unsigned IndexNZCV(unsigned BitOffset) {
|
||||
switch (BitOffset) {
|
||||
case FEXCore::X86State::RFLAG_OF_LOC: return 28;
|
||||
case FEXCore::X86State::RFLAG_CF_LOC: return 29;
|
||||
case FEXCore::X86State::RFLAG_ZF_LOC: return 30;
|
||||
case FEXCore::X86State::RFLAG_SF_LOC: return 31;
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode *GetNZCV() {
|
||||
if (!CachedNZCV) {
|
||||
CachedNZCV = _LoadFlag(FEXCore::X86State::RFLAG_NZCV_LOC);
|
||||
|
||||
// We don't know what's set
|
||||
PossiblySetNZCVBits = ~0;
|
||||
}
|
||||
|
||||
return CachedNZCV;
|
||||
}
|
||||
|
||||
void SetNZCV(OrderedNode *Value) {
|
||||
CachedNZCV = Value;
|
||||
}
|
||||
|
||||
void ZeroNZCV() {
|
||||
CachedNZCV = _Constant(0);
|
||||
PossiblySetNZCVBits = 0;
|
||||
}
|
||||
|
||||
void ZeroCV() {
|
||||
// Get old NZCV before we mess with PossiblySetNZCVBits
|
||||
auto OldNZCV = GetNZCV();
|
||||
|
||||
// Mask out the NZ bits, clearing CV. Even if the code sets CV after, this can end up faster
|
||||
// moves by allowing orlshl to be used instead of bfi.
|
||||
PossiblySetNZCVBits = (1u << IndexNZCV(FEXCore::X86State::RFLAG_SF_LOC)) |
|
||||
(1u << IndexNZCV(FEXCore::X86State::RFLAG_ZF_LOC));
|
||||
SetNZCV(_And(OldNZCV, _Constant(PossiblySetNZCVBits)));
|
||||
}
|
||||
|
||||
void SetN_ZeroZCV(unsigned SrcSize, OrderedNode *Res) {
|
||||
static_assert(IndexNZCV(FEXCore::X86State::RFLAG_SF_LOC) == 31);
|
||||
|
||||
unsigned NBit = 31;
|
||||
unsigned SignBit = (SrcSize * 8) - 1;
|
||||
|
||||
OrderedNode *Shifted;
|
||||
|
||||
// Shift the sign bit into the N bit
|
||||
if (SignBit > NBit)
|
||||
Shifted = _Ashr(Res, _Constant(SignBit - NBit));
|
||||
else if (SignBit < NBit)
|
||||
Shifted = _Lshl(Res, _Constant(NBit - SignBit));
|
||||
else
|
||||
Shifted = Res;
|
||||
|
||||
// Mask off just the N bit, which now equals the sign bit
|
||||
CachedNZCV = _And(Shifted, _Constant(1u << NBit));
|
||||
PossiblySetNZCVBits = (1u << NBit);
|
||||
}
|
||||
|
||||
void SetNZ_ZeroCV(unsigned SrcSize, OrderedNode *Res) {
|
||||
// The TestNZ opcode does this operation natively for 32-bit or 64-bit.
|
||||
// Otherwise we can implement the functionality ourselves with some bit math.
|
||||
if (CTX->BackendFeatures.SupportsFlags && SrcSize >= 4) {
|
||||
CachedNZCV = _TestNZ(SrcSize, Res);
|
||||
PossiblySetNZCVBits = (1u << 31) | (1u << 30);
|
||||
} else {
|
||||
// N
|
||||
SetN_ZeroZCV(SrcSize, Res);
|
||||
|
||||
// Z
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
auto SelectOp = _Select(FEXCore::IR::COND_EQ, Res, Zero, One, Zero);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(SelectOp);
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode *InsertNZCV(OrderedNode *NZCV, unsigned BitOffset, OrderedNode *Value) {
|
||||
unsigned Bit = IndexNZCV(BitOffset);
|
||||
|
||||
uint32_t SetBits = PossiblySetNZCVBits;
|
||||
PossiblySetNZCVBits |= (1u << Bit);
|
||||
|
||||
if (SetBits == 0)
|
||||
return _Lshl(Value, _Constant(Bit));
|
||||
else if (CTX->BackendFeatures.SupportsShiftedBitwise && (SetBits & (1u << Bit)) == 0)
|
||||
return _Orlshl(NZCV, Value, Bit);
|
||||
else
|
||||
return _Bfi(4, 1, Bit, NZCV, Value);
|
||||
}
|
||||
|
||||
template<unsigned BitOffset>
|
||||
void SetRFLAG(OrderedNode *Value) {
|
||||
flagsOp = SelectionFlag::Nothing;
|
||||
_StoreFlag(_Bfe(1, 0, Value), BitOffset);
|
||||
SetRFLAG(Value, BitOffset);
|
||||
}
|
||||
|
||||
void SetRFLAG(OrderedNode *Value, unsigned BitOffset) {
|
||||
flagsOp = SelectionFlag::Nothing;
|
||||
_StoreFlag(_Bfe(1, 0, Value), BitOffset);
|
||||
|
||||
if (IsNZCV(BitOffset))
|
||||
SetNZCV(InsertNZCV(GetNZCV(), BitOffset, Value));
|
||||
else
|
||||
_StoreFlag(Value, BitOffset);
|
||||
}
|
||||
|
||||
OrderedNode *GetRFLAG(unsigned BitOffset) {
|
||||
return _LoadFlag(BitOffset);
|
||||
if (IsNZCV(BitOffset)) {
|
||||
if (!CachedNZCV || (PossiblySetNZCVBits & (1u << IndexNZCV(BitOffset))))
|
||||
return _Bfe(1, 1, IndexNZCV(BitOffset), GetNZCV());
|
||||
else
|
||||
return _Constant(0);
|
||||
} else {
|
||||
return _LoadFlag(BitOffset);
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode *SelectCC(uint8_t OP, OrderedNode *TrueValue, OrderedNode *FalseValue);
|
||||
@@ -1155,34 +1282,41 @@ private:
|
||||
/**
|
||||
* @name These functions are used by the deferred flag handling while it is calculating and storing flags in to RFLAGs.
|
||||
* @{ */
|
||||
void CalculcateFlags_ADC(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF);
|
||||
void CalculcateFlags_SBB(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF);
|
||||
void CalculcateFlags_SUB(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true);
|
||||
void CalculcateFlags_ADD(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true);
|
||||
void CalculcateFlags_MUL(uint8_t SrcSize, OrderedNode *Res, OrderedNode *High);
|
||||
void CalculcateFlags_UMUL(OrderedNode *High);
|
||||
void CalculcateFlags_Logical(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_ShiftLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_ShiftLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculcateFlags_ShiftRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_ShiftRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculcateFlags_SignShiftRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_SignShiftRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculcateFlags_RotateRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_RotateLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_RotateRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculcateFlags_RotateLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculcateFlags_FCMP(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_BEXTR(OrderedNode *Src);
|
||||
void CalculcateFlags_BLSI(uint8_t SrcSize, OrderedNode *Src);
|
||||
void CalculcateFlags_BLSMSK(OrderedNode *Src);
|
||||
void CalculcateFlags_BLSR(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src);
|
||||
void CalculcateFlags_POPCOUNT(OrderedNode *Src);
|
||||
void CalculcateFlags_BZHI(uint8_t SrcSize, OrderedNode *Result, OrderedNode *Src);
|
||||
void CalculcateFlags_TZCNT(OrderedNode *Src);
|
||||
void CalculcateFlags_LZCNT(uint8_t SrcSize, OrderedNode *Src);
|
||||
void CalculcateFlags_BITSELECT(OrderedNode *Src);
|
||||
void CalculcateFlags_RDRAND(OrderedNode *Src);
|
||||
OrderedNode *LoadPF();
|
||||
void CalculatePFUncheckedABI(OrderedNode *Res, OrderedNode *condition = nullptr);
|
||||
void CalculatePF(OrderedNode *Res, OrderedNode *condition = nullptr);
|
||||
|
||||
void CalculateOF_Add(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculateFlags_ADC(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF);
|
||||
void CalculateFlags_SBB(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF);
|
||||
void CalculateFlags_SUB(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true);
|
||||
void CalculateFlags_ADD(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true);
|
||||
void CalculateFlags_MUL(uint8_t SrcSize, OrderedNode *Res, OrderedNode *High);
|
||||
void CalculateFlags_UMUL(OrderedNode *High);
|
||||
void CalculateFlags_Logical(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculateFlags_ShiftLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculateFlags_ShiftRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRightDoubleImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculateFlags_SignShiftRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculateFlags_RotateRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculateFlags_RotateLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculateFlags_RotateRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculateFlags_RotateLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculateFlags_FCMP(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculateFlags_BEXTR(OrderedNode *Src);
|
||||
void CalculateFlags_BLSI(uint8_t SrcSize, OrderedNode *Src);
|
||||
void CalculateFlags_BLSMSK(OrderedNode *Src);
|
||||
void CalculateFlags_BLSR(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src);
|
||||
void CalculateFlags_POPCOUNT(OrderedNode *Src);
|
||||
void CalculateFlags_BZHI(uint8_t SrcSize, OrderedNode *Result, OrderedNode *Src);
|
||||
void CalculateFlags_TZCNT(OrderedNode *Src);
|
||||
void CalculateFlags_LZCNT(uint8_t SrcSize, OrderedNode *Src);
|
||||
void CalculateFlags_BITSELECT(OrderedNode *Src);
|
||||
void CalculateFlags_RDRAND(OrderedNode *Src);
|
||||
/** @} */
|
||||
|
||||
/**
|
||||
@@ -1395,6 +1529,23 @@ private:
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_ShiftRightDoubleImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero.
|
||||
if (Shift == 0) return;
|
||||
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_LSHRDI,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.OneSrcImmediate = {
|
||||
.Src1 = Src1,
|
||||
.Imm = Shift,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_RotateRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
// Doesn't set all the flags, needs to calculate.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
+405
-496
File diff suppressed because it is too large.
Load diff
+40
-119
@@ -204,8 +204,8 @@ void OpDispatchBuilder::VMOVSLDUPOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::MOVSSOp(OpcodeArgs) {
|
||||
if (Op->Dest.IsGPR() && Op->Src[0].IsGPR()) {
|
||||
// MOVSS xmm1, xmm2
|
||||
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, 16, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 4, Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
auto Result = _VInsElement(16, 4, 0, 0, Dest, Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
@@ -217,7 +217,7 @@ void OpDispatchBuilder::MOVSSOp(OpcodeArgs) {
|
||||
}
|
||||
else {
|
||||
// MOVSS mem32, xmm1
|
||||
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 4, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, 4, -1);
|
||||
}
|
||||
}
|
||||
@@ -2257,40 +2257,26 @@ template
|
||||
void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<8, true>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::MASKMOVOp(OpcodeArgs) {
|
||||
// Until we get correct PHI nodes this is required to be a loop unroll
|
||||
const auto Size = uint32_t{GetSrcSize(Op)} * 8;
|
||||
const auto Size = GetSrcSize(Op);
|
||||
|
||||
OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *MaskSrc = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
// Mask only cares about the top bit of each byte
|
||||
MaskSrc = _VCMPLTZ(Size, 1, MaskSrc);
|
||||
|
||||
// Vector that will overwrite byte elements.
|
||||
OrderedNode *VectorSrc = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
|
||||
// RDI source
|
||||
auto MemDest = LoadGPRRegister(X86State::REG_RDI);
|
||||
|
||||
const size_t NumElements = Size / 64;
|
||||
for (size_t Element = 0; Element < NumElements; ++Element) {
|
||||
// Extract the current element
|
||||
auto SrcElement = _VExtractToGPR(GetSrcSize(Op), 8, Src, Element);
|
||||
auto DestElement = _VExtractToGPR(GetSrcSize(Op), 8, Dest, Element);
|
||||
// DS prefix by default.
|
||||
MemDest = AppendSegmentOffset(MemDest, Op->Flags, FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX);
|
||||
|
||||
constexpr size_t NumSelectBits = 64 / 8;
|
||||
for (size_t Select = 0; Select < NumSelectBits; ++Select) {
|
||||
auto SelectMask = _Bfe(1, 8 * Select + 7, SrcElement);
|
||||
auto CondJump = _CondJump(SelectMask, {COND_EQ});
|
||||
auto StoreBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
SetFalseJumpTarget(CondJump, StoreBlock);
|
||||
SetCurrentCodeBlock(StoreBlock);
|
||||
{
|
||||
auto DestByte = _Bfe(8, 8 * Select, DestElement);
|
||||
auto MemLocation = _Add(MemDest, _Constant(Element * 8 + Select));
|
||||
// MASKMOVDQU/MASKMOVQ is explicitly weakly-ordered on its store
|
||||
_StoreMem(GPRClass, 1, MemLocation, DestByte, 1);
|
||||
}
|
||||
auto Jump = _Jump();
|
||||
auto NextJumpTarget = CreateNewCodeBlockAfter(StoreBlock);
|
||||
SetJumpTarget(Jump, NextJumpTarget);
|
||||
SetTrueJumpTarget(CondJump, NextJumpTarget);
|
||||
SetCurrentCodeBlock(NextJumpTarget);
|
||||
}
|
||||
}
|
||||
OrderedNode *XMMReg = _LoadMem(FPRClass, Size, MemDest, 1);
|
||||
|
||||
// If the Mask element high bit is set then overwrite the element with the source, else keep the memory variant
|
||||
XMMReg = _VBSL(Size, MaskSrc, VectorSrc, XMMReg);
|
||||
_StoreMem(FPRClass, Size, MemDest, XMMReg, 1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VMASKMOVOpImpl(OpcodeArgs, size_t ElementSize, size_t DataSize, bool IsStore,
|
||||
@@ -3641,36 +3627,19 @@ void OpDispatchBuilder::VPHSUBOp<2>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VPHSUBOp<4>(OpcodeArgs);
|
||||
|
||||
OrderedNode* OpDispatchBuilder::PHADDSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1,
|
||||
const X86Tables::DecodedOperand& Src2) {
|
||||
OrderedNode* OpDispatchBuilder::PHADDSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op) {
|
||||
const auto Size = GetSrcSize(Op);
|
||||
const uint8_t ElementSize = 2;
|
||||
|
||||
OrderedNode *Src1Node = LoadSource(FPRClass, Op, Src1, Op->Flags, -1);
|
||||
OrderedNode *Src2Node = LoadSource(FPRClass, Op, Src2, Op->Flags, -1);
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Src1Op, Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Src2Op, Op->Flags, -1);
|
||||
|
||||
if (Size == 8) {
|
||||
// Implementation is more efficient for 8byte registers
|
||||
OrderedNode *Src1_Larger = _VSXTL(Size * 2, 2, Src1Node);
|
||||
OrderedNode *Src2_Larger = _VSXTL(Size * 2, 2, Src2Node);
|
||||
|
||||
OrderedNode *AddRes = _VAddP(Size * 2, 4, Src1_Larger, Src2_Larger);
|
||||
|
||||
// Saturate back down to the result
|
||||
return _VSQXTN(Size * 2, 4, AddRes);
|
||||
}
|
||||
|
||||
OrderedNode *Src1_Larger = _VSXTL(Size, 2, Src1Node);
|
||||
OrderedNode *Src1_Larger_H = _VSXTL2(Size, 2, Src1Node);
|
||||
|
||||
OrderedNode *Src2_Larger = _VSXTL(Size, 2, Src2Node);
|
||||
OrderedNode *Src2_Larger_H = _VSXTL2(Size, 2, Src2Node);
|
||||
|
||||
OrderedNode *AddRes_L = _VAddP(Size, 4, Src1_Larger, Src1_Larger_H);
|
||||
OrderedNode *AddRes_H = _VAddP(Size, 4, Src2_Larger, Src2_Larger_H);
|
||||
auto Even = _VUnZip(Size, ElementSize, Src1, Src2);
|
||||
auto Odd = _VUnZip2(Size, ElementSize, Src1, Src2);
|
||||
|
||||
// Saturate back down to the result
|
||||
OrderedNode *Res = _VSQXTN(Size, 4, AddRes_L);
|
||||
return _VSQXTN2(Size, 4, Res, AddRes_H);
|
||||
return _VSQAdd(Size, ElementSize, Even, Odd);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PHADDS(OpcodeArgs) {
|
||||
@@ -3697,51 +3666,15 @@ OrderedNode* OpDispatchBuilder::PHSUBSOpImpl(OpcodeArgs, const X86Tables::Decode
|
||||
const X86Tables::DecodedOperand& Src2Op) {
|
||||
const auto Size = GetSrcSize(Op);
|
||||
const uint8_t ElementSize = 2;
|
||||
const uint8_t NumElements = Size / ElementSize;
|
||||
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Src1Op, Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Src2Op, Op->Flags, -1);
|
||||
|
||||
// This is a bit complicated since AArch64 doesn't support a pairwise subtract
|
||||
OrderedNode *Src1_Neg = _VNeg(Size, ElementSize, Src1);
|
||||
OrderedNode *Src2_Neg = _VNeg(Size, ElementSize, Src2);
|
||||
|
||||
// Now we need to swizzle the values
|
||||
OrderedNode *Swizzle_Src1 = Src1;
|
||||
OrderedNode *Swizzle_Src2 = Src2;
|
||||
|
||||
// Odd elements turn in to negated elements
|
||||
for (size_t i = 1; i < NumElements; i += 2) {
|
||||
Swizzle_Src1 = _VInsElement(Size, ElementSize, i, i, Swizzle_Src1, Src1_Neg);
|
||||
Swizzle_Src2 = _VInsElement(Size, ElementSize, i, i, Swizzle_Src2, Src2_Neg);
|
||||
}
|
||||
|
||||
Src1 = Swizzle_Src1;
|
||||
Src2 = Swizzle_Src2;
|
||||
|
||||
if (Size == 8) {
|
||||
// Implementation is more efficient for 8byte registers
|
||||
OrderedNode *Src1_Larger = _VSXTL(Size * 2, 2, Src1);
|
||||
OrderedNode *Src2_Larger = _VSXTL(Size * 2, 2, Src2);
|
||||
|
||||
OrderedNode *AddRes = _VAddP(Size * 2, 4, Src1_Larger, Src2_Larger);
|
||||
|
||||
// Saturate back down to the result
|
||||
return _VSQXTN(Size * 2, 4, AddRes);
|
||||
}
|
||||
|
||||
OrderedNode *Src1_Larger = _VSXTL(Size, 2, Src1);
|
||||
OrderedNode *Src1_Larger_H = _VSXTL2(Size, 2, Src1);
|
||||
|
||||
OrderedNode *Src2_Larger = _VSXTL(Size, 2, Src2);
|
||||
OrderedNode *Src2_Larger_H = _VSXTL2(Size, 2, Src2);
|
||||
|
||||
OrderedNode *AddRes_L = _VAddP(Size, 4, Src1_Larger, Src1_Larger_H);
|
||||
OrderedNode *AddRes_H = _VAddP(Size, 4, Src2_Larger, Src2_Larger_H);
|
||||
auto Even = _VUnZip(Size, ElementSize, Src1, Src2);
|
||||
auto Odd = _VUnZip2(Size, ElementSize, Src1, Src2);
|
||||
|
||||
// Saturate back down to the result
|
||||
OrderedNode *Res = _VSQXTN(Size, 4, AddRes_L);
|
||||
return _VSQXTN2(Size, 4, Res, AddRes_H);
|
||||
return _VSQSub(Size, ElementSize, Even, Odd);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PHSUBS(OpcodeArgs) {
|
||||
@@ -3778,33 +3711,19 @@ OrderedNode* OpDispatchBuilder::PSADBWOpImpl(OpcodeArgs,
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Src2Op, Op->Flags, -1);
|
||||
|
||||
if (Size == 8) {
|
||||
OrderedNode *Src1_Low = _VUXTL(Size*2, 1, Src1);
|
||||
OrderedNode *Src2_Low = _VUXTL(Size*2, 1, Src2);
|
||||
|
||||
OrderedNode *SubResult = _VSub(Size*2, 2, Src1_Low, Src2_Low);
|
||||
OrderedNode *AbsResult = _VAbs(Size*2, 2, SubResult);
|
||||
auto AbsResult = _VUABDL(Size * 2, 1, Src1, Src2);
|
||||
|
||||
// Now vector-wide add the results for each
|
||||
return _VAddV(Size * 2, 2, AbsResult);
|
||||
}
|
||||
|
||||
|
||||
OrderedNode *Src1_Low = _VUXTL(Size, 1, Src1);
|
||||
OrderedNode *Src1_High = _VUXTL2(Size, 1, Src1);
|
||||
|
||||
OrderedNode *Src2_Low = _VUXTL(Size, 1, Src2);
|
||||
OrderedNode *Src2_High = _VUXTL2(Size, 1, Src2);
|
||||
|
||||
OrderedNode *SubResult_Low = _VSub(Size, 2, Src1_Low, Src2_Low);
|
||||
OrderedNode *SubResult_High = _VSub(Size, 2, Src1_High, Src2_High);
|
||||
|
||||
OrderedNode *AbsResult_Low = _VAbs(Size, 2, SubResult_Low);
|
||||
OrderedNode *AbsResult_High = _VAbs(Size, 2, SubResult_High);
|
||||
auto AbsResult_Low = _VUABDL(Size, 1, Src1, Src2);
|
||||
auto AbsResult_High = _VUABDL2(Size, 1, Src1, Src2);
|
||||
|
||||
OrderedNode *Result_Low = _VAddV(16, 2, AbsResult_Low);
|
||||
OrderedNode *Result_High = _VAddV(16, 2, AbsResult_High);
|
||||
auto Low = _VZip(Size, 8, Result_Low, Result_High);
|
||||
|
||||
OrderedNode *Low = _VInsElement(Size, 8, 1, 0, Result_Low, Result_High);
|
||||
if (Is128Bit) {
|
||||
return Low;
|
||||
}
|
||||
@@ -4235,11 +4154,8 @@ OrderedNode* OpDispatchBuilder::PHMINPOSUWOpImpl(OpcodeArgs) {
|
||||
Element, MinGPR, Indexes[i - 1], Pos);
|
||||
}
|
||||
|
||||
// Insert the minimum in to bits [15:0]
|
||||
OrderedNode *Result = _VMov(2, Min);
|
||||
|
||||
// Insert position in to bits [18:16]
|
||||
return _VInsGPR(16, 2, 1, Result, Pos);
|
||||
return _VInsGPR(16, 2, 1, Min, Pos);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PHMINPOSUWOp(OpcodeArgs) {
|
||||
@@ -4895,7 +4811,12 @@ void OpDispatchBuilder::PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask
|
||||
OrderedNode *Result = _Select(IR::COND_EQ, ResultNoFlags, ZeroConst,
|
||||
IfZero, IfNotZero);
|
||||
|
||||
StoreGPRRegister(X86State::REG_RCX, Result, 4);
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
if (GPRSize == 8) {
|
||||
// If being stored to an 8-byte register, zero extend the 4-byte result.
|
||||
Result = _Bfe(8, 32, 0, Result);
|
||||
}
|
||||
StoreGPRRegister(X86State::REG_RCX, Result);
|
||||
}
|
||||
|
||||
// Set all of the necessary flags.
|
||||
|
||||
+14
-14
@@ -171,12 +171,14 @@ void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
auto zero = _Constant(0);
|
||||
|
||||
// Sign extend to 64bits
|
||||
if (read_width != 8)
|
||||
if (read_width != 8) {
|
||||
data = _Sext(read_width * 8, data);
|
||||
}
|
||||
|
||||
// Extract sign and make interger absolute
|
||||
auto sign = _Select(COND_SLT, data, zero, _Constant(0x8000), zero);
|
||||
auto absolute = _Select(COND_SLT, data, zero, _Sub(zero, data), data);
|
||||
|
||||
auto absolute = _Abs(data);
|
||||
|
||||
// left justify the absolute interger
|
||||
auto shift = _Sub(_Constant(63), _FindMSB(absolute));
|
||||
@@ -621,18 +623,19 @@ void OpDispatchBuilder::FXTRACT(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FNINIT(OpcodeArgs) {
|
||||
auto Zero = _Constant(0);
|
||||
// Init FCW to 0x037F
|
||||
auto NewFCW = _Constant(16, 0x037F);
|
||||
_F80LoadFCW(NewFCW);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
// Init FSW to 0
|
||||
SetX87Top(_Constant(0));
|
||||
SetX87Top(Zero);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(Zero);
|
||||
|
||||
// Tags all get set to 0b11
|
||||
_StoreContext(2, GPRClass, _Constant(0xFFFF), offsetof(FEXCore::Core::CPUState, FTW));
|
||||
@@ -1278,7 +1281,7 @@ void OpDispatchBuilder::X87FXAM(OpcodeArgs) {
|
||||
OrderedNode *Result = _VExtractToGPR(16, 8, a, 1);
|
||||
|
||||
// Extract the sign bit
|
||||
Result = _Lshr(Result, _Constant(15));
|
||||
Result = _Bfe(1, 15, Result);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Result);
|
||||
|
||||
// Claim this is a normal number
|
||||
@@ -1354,20 +1357,17 @@ void OpDispatchBuilder::X87FCMOV(OpcodeArgs) {
|
||||
break;
|
||||
}
|
||||
|
||||
auto MaskConst = _Constant(FLAGMask);
|
||||
auto RFLAG = GetPackedRFLAG(FLAGMask);
|
||||
|
||||
auto RFLAG = GetPackedRFLAG(false);
|
||||
|
||||
auto AndOp = _And(RFLAG, MaskConst);
|
||||
switch (Type) {
|
||||
case COMPARE_ZERO: {
|
||||
SrcCond = _Select(FEXCore::IR::COND_EQ,
|
||||
AndOp, ZeroConst, OneConst, ZeroConst);
|
||||
RFLAG, ZeroConst, OneConst, ZeroConst);
|
||||
break;
|
||||
}
|
||||
case COMPARE_NOTZERO: {
|
||||
SrcCond = _Select(FEXCore::IR::COND_EQ,
|
||||
AndOp, ZeroConst, ZeroConst, OneConst);
|
||||
RFLAG, ZeroConst, ZeroConst, OneConst);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -50,12 +50,12 @@ void OpDispatchBuilder::FNINITF64(OpcodeArgs) {
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
// Init FSW to 0
|
||||
SetX87Top(_Constant(0));
|
||||
SetX87Top(Zero);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(Zero);
|
||||
|
||||
// Tags all get set to 0b11
|
||||
_StoreContext(2, GPRClass, _Constant(0xFFFF), offsetof(FEXCore::Core::CPUState, FTW));
|
||||
@@ -1119,7 +1119,7 @@ void OpDispatchBuilder::X87FXAMF64(OpcodeArgs) {
|
||||
OrderedNode *Result = _VExtractToGPR(8, 8, a, 0);
|
||||
|
||||
// Extract the sign bit
|
||||
Result = _Lshr(Result, _Constant(63));
|
||||
Result = _Bfe(1, 63, Result);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Result);
|
||||
|
||||
// Claim this is a normal number
|
||||
|
||||
@@ -6,26 +6,6 @@
|
||||
#include <signal.h>
|
||||
|
||||
namespace FEXCore {
|
||||
struct ThreadState {
|
||||
FEXCore::Core::InternalThreadState *Thread{};
|
||||
};
|
||||
|
||||
thread_local ThreadState ThreadData{};
|
||||
|
||||
FEXCore::Core::InternalThreadState *SignalDelegator::GetTLSThread() {
|
||||
return ThreadData.Thread;
|
||||
}
|
||||
|
||||
void SignalDelegator::RegisterTLSState(FEXCore::Core::InternalThreadState *Thread) {
|
||||
ThreadData.Thread = Thread;
|
||||
RegisterFrontendTLSState(Thread);
|
||||
}
|
||||
|
||||
void SignalDelegator::UninstallTLSState(FEXCore::Core::InternalThreadState *Thread) {
|
||||
UninstallFrontendTLSState(Thread);
|
||||
ThreadData.Thread = nullptr;
|
||||
}
|
||||
|
||||
void SignalDelegator::RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
SetHostSignalHandler(Signal, Func, Required);
|
||||
FrontendRegisterHostSignalHandler(Signal, Func, Required);
|
||||
|
||||
@@ -48,7 +48,7 @@ void InitializeSecondaryModRMTables() {
|
||||
{((3 << 3) | 1), 1, X86InstInfo{"RDTSCP", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{((3 << 3) | 2), 1, X86InstInfo{"MONITORX", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{((3 << 3) | 3), 1, X86InstInfo{"MWAITX", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{((3 << 3) | 4), 1, X86InstInfo{"CLZERO", TYPE_INST, FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{((3 << 3) | 4), 1, X86InstInfo{"CLZERO", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SF_SRC_RAX | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{((3 << 3) | 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{((3 << 3) | 6), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{((3 << 3) | 7), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
@@ -342,10 +342,10 @@ void InitializeVEXTables() {
|
||||
{OPD(2, 0b01, 0x8C), 1, X86InstInfo{"VPMASKMOV", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x8E), 1, X86InstInfo{"VPMASKMOV", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SF_MOD_DST | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(2, 0b01, 0x90), 1, X86InstInfo{"VPGATHERD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x91), 1, X86InstInfo{"VPGATHERQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x92), 1, X86InstInfo{"VPGATHERD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x93), 1, X86InstInfo{"VPGATHERQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x90), 1, X86InstInfo{"VPGATHERDD/Q", TYPE_UNDEC, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_2ND_SRC | FLAGS_VEX_VSIB | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x91), 1, X86InstInfo{"VPGATHERQD/Q", TYPE_UNDEC, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_2ND_SRC | FLAGS_VEX_VSIB | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x92), 1, X86InstInfo{"VGATHERDPS/D", TYPE_UNDEC, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_2ND_SRC | FLAGS_VEX_VSIB | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x93), 1, X86InstInfo{"VGATHERQPS/D", TYPE_UNDEC, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_2ND_SRC | FLAGS_VEX_VSIB | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(2, 0b01, 0x96), 1, X86InstInfo{"VFMADDSUB132", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x97), 1, X86InstInfo{"VFMSUBADD132", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
@@ -25,7 +25,7 @@ constexpr uint32_t FLAG_ADDRESS_SIZE = (1 << 1);
|
||||
constexpr uint32_t FLAG_LOCK = (1 << 2);
|
||||
constexpr uint32_t FLAG_LEGACY_PREFIX = (1 << 3);
|
||||
constexpr uint32_t FLAG_REX_PREFIX = (1 << 4);
|
||||
// Hole where 1 << 5 is
|
||||
constexpr uint32_t FLAG_VSIB_BYTE = (1 << 5);
|
||||
// Hole where 1 << 6 is
|
||||
constexpr uint32_t FLAG_REX_WIDENING = (1 << 7);
|
||||
constexpr uint32_t FLAG_REX_XGPR_B = (1 << 8);
|
||||
@@ -138,9 +138,10 @@ struct DecodedOperand {
|
||||
}
|
||||
|
||||
union TypeUnion {
|
||||
struct {
|
||||
struct GPRType {
|
||||
bool HighBits;
|
||||
uint8_t GPR;
|
||||
auto operator<=>(const GPRType&) const = default;
|
||||
} GPR;
|
||||
|
||||
struct {
|
||||
@@ -155,9 +156,10 @@ struct DecodedOperand {
|
||||
} Value;
|
||||
} RIPLiteral;
|
||||
|
||||
struct {
|
||||
struct LiteralType {
|
||||
uint64_t Value;
|
||||
uint8_t Size;
|
||||
auto operator<=>(const LiteralType&) const = default;
|
||||
} Literal;
|
||||
|
||||
struct {
|
||||
@@ -347,6 +349,8 @@ constexpr InstFlagType FLAGS_VEX_1ST_SRC = (1ULL << 24);
|
||||
constexpr InstFlagType FLAGS_VEX_2ND_SRC = (1ULL << 25);
|
||||
// Whether or not the instruction has a VEX prefix for the destination
|
||||
constexpr InstFlagType FLAGS_VEX_DST = (1ULL << 26);
|
||||
// Whether or not the instruction has a VSIB byte
|
||||
constexpr InstFlagType FLAGS_VEX_VSIB = (1ULL << 27);
|
||||
|
||||
constexpr InstFlagType FLAGS_SIZE_DST_OFF = 58;
|
||||
constexpr InstFlagType FLAGS_SIZE_SRC_OFF = FLAGS_SIZE_DST_OFF + 3;
|
||||
|
||||
+3
-3
@@ -206,7 +206,7 @@ namespace FEXCore::IR {
|
||||
FEXCore::Allocator::YesIKnowImNotSupposedToUseTheGlibcAllocator glibc;
|
||||
|
||||
AOTIRCaptureCacheWriteoutLock.lock();
|
||||
std::function<void()> fn = std::move(AOTIRCaptureCacheWriteoutQueue.front());
|
||||
WriteOutFn fn = std::move(AOTIRCaptureCacheWriteoutQueue.front());
|
||||
bool MaybeEmpty = false;
|
||||
AOTIRCaptureCacheWriteoutQueue.pop();
|
||||
MaybeEmpty = AOTIRCaptureCacheWriteoutQueue.size() == 0;
|
||||
@@ -225,7 +225,7 @@ namespace FEXCore::IR {
|
||||
LOGMAN_MSG_A_FMT("Must never get here");
|
||||
}
|
||||
|
||||
void AOTIRCaptureCache::AOTIRCaptureCacheWriteoutQueue_Append(const std::function<void()> &fn) {
|
||||
void AOTIRCaptureCache::AOTIRCaptureCacheWriteoutQueue_Append(const WriteOutFn &fn) {
|
||||
bool Flush = false;
|
||||
|
||||
{
|
||||
@@ -242,7 +242,7 @@ namespace FEXCore::IR {
|
||||
}
|
||||
}
|
||||
|
||||
void AOTIRCaptureCache::WriteFilesWithCode(std::function<void(const fextl::string& fileid, const fextl::string& filename)> Writer) {
|
||||
void AOTIRCaptureCache::WriteFilesWithCode(const Context::AOTIRCodeFileWriterFn &Writer) {
|
||||
std::shared_lock lk(AOTIRCacheLock);
|
||||
for( const auto &Entry: AOTIRCache) {
|
||||
if (Entry.second.ContainsCode) {
|
||||
|
||||
+13
-12
@@ -89,13 +89,14 @@ namespace FEXCore::IR {
|
||||
|
||||
class AOTIRCaptureCache final {
|
||||
public:
|
||||
using WriteOutFn = std::function<void()>;
|
||||
|
||||
AOTIRCaptureCache(FEXCore::Context::ContextImpl *ctx) : CTX {ctx} {}
|
||||
|
||||
void FinalizeAOTIRCache();
|
||||
void AOTIRCaptureCacheWriteoutQueue_Flush();
|
||||
void AOTIRCaptureCacheWriteoutQueue_Append(const std::function<void()> &fn);
|
||||
void WriteFilesWithCode(std::function<void(const fextl::string& fileid, const fextl::string& filename)> Writer);
|
||||
void AOTIRCaptureCacheWriteoutQueue_Append(const WriteOutFn &fn);
|
||||
void WriteFilesWithCode(const Context::AOTIRCodeFileWriterFn &Writer);
|
||||
|
||||
struct PreGenerateIRFetchResult {
|
||||
FEXCore::IR::IRListView *IRList {};
|
||||
@@ -121,16 +122,16 @@ namespace FEXCore::IR {
|
||||
void UnloadAOTIRCacheEntry(AOTIRCacheEntry *Entry);
|
||||
|
||||
// Callbacks
|
||||
void SetAOTIRLoader(std::function<int(const fextl::string&)> CacheReader) {
|
||||
AOTIRLoader = CacheReader;
|
||||
void SetAOTIRLoader(Context::AOTIRLoaderCBFn CacheReader) {
|
||||
AOTIRLoader = std::move(CacheReader);
|
||||
}
|
||||
|
||||
void SetAOTIRWriter(std::function<fextl::unique_ptr<FEXCore::Context::AOTIRWriter>(const fextl::string&)> CacheWriter) {
|
||||
AOTIRWriter = CacheWriter;
|
||||
void SetAOTIRWriter(Context::AOTIRWriterCBFn CacheWriter) {
|
||||
AOTIRWriter = std::move(CacheWriter);
|
||||
}
|
||||
|
||||
void SetAOTIRRenamer(std::function<void(const fextl::string&)> CacheRenamer) {
|
||||
AOTIRRenamer = CacheRenamer;
|
||||
void SetAOTIRRenamer(Context::AOTIRRenamerCBFn CacheRenamer) {
|
||||
AOTIRRenamer = std::move(CacheRenamer);
|
||||
}
|
||||
|
||||
private:
|
||||
@@ -140,13 +141,13 @@ namespace FEXCore::IR {
|
||||
std::shared_mutex AOTIRCaptureCacheWriteoutLock;
|
||||
std::atomic<bool> AOTIRCaptureCacheWriteoutFlusing;
|
||||
|
||||
fextl::queue<std::function<void()>> AOTIRCaptureCacheWriteoutQueue;
|
||||
fextl::queue<WriteOutFn> AOTIRCaptureCacheWriteoutQueue;
|
||||
|
||||
FEXCore::IR::AOTCacheType AOTIRCache;
|
||||
|
||||
std::function<int(const fextl::string&)> AOTIRLoader;
|
||||
std::function<fextl::unique_ptr<FEXCore::Context::AOTIRWriter>(const fextl::string&)> AOTIRWriter;
|
||||
std::function<void(const fextl::string&)> AOTIRRenamer;
|
||||
Context::AOTIRLoaderCBFn AOTIRLoader;
|
||||
Context::AOTIRWriterCBFn AOTIRWriter;
|
||||
Context::AOTIRRenamerCBFn AOTIRRenamer;
|
||||
fextl::unordered_map<fextl::string, FEXCore::IR::AOTIRCaptureCacheEntry> AOTIRCaptureCacheMap;
|
||||
};
|
||||
}
|
||||
+46
-9
@@ -717,6 +717,13 @@
|
||||
],
|
||||
"DestSize": "std::max<uint8_t>(4, GetOpSize(_Src))"
|
||||
},
|
||||
"GPR = Abs GPR:$Src": {
|
||||
"Desc": ["Integer 2's complement absolute value",
|
||||
"Dest = std::abs(Src)",
|
||||
"Will truncate to 64 or 32bits"
|
||||
],
|
||||
"DestSize": "std::max<uint8_t>(4, GetOpSize(_Src))"
|
||||
},
|
||||
"GPR = Not GPR:$Src": {
|
||||
"Desc": ["Integer binary not",
|
||||
"op:",
|
||||
@@ -775,6 +782,14 @@
|
||||
"Desc": ["Integer binary or"
|
||||
]
|
||||
},
|
||||
"GPR = Orlshl GPR:$Src1, GPR:$Src2, u8:$BitShift": {
|
||||
"Desc": ["Integer binary or with logical shift left"
|
||||
]
|
||||
},
|
||||
"GPR = Orlshr GPR:$Src1, GPR:$Src2, u8:$BitShift": {
|
||||
"Desc": ["Integer binary or with logical shift right"
|
||||
]
|
||||
},
|
||||
"GPR = Xor GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer binary exclusive or"
|
||||
]
|
||||
@@ -787,20 +802,33 @@
|
||||
"Desc": ["Integer binary AND NOT. Performs the equivalent of Src1 & ~Src2"],
|
||||
"DestSize": "std::max<uint8_t>(4, GetOpSize(_Src1))"
|
||||
},
|
||||
"GPR = Lshl GPR:$Src1, GPR:$Src2": {
|
||||
"GPR = TestNZ u8:$Size, GPR:$Src1": {
|
||||
"Desc": ["Return NZCV for a GPR, setting N and Z accordingly and zeroing C and V"],
|
||||
"DestSize": "4"
|
||||
},
|
||||
"GPR = Lshl u8:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer logical shift left"
|
||||
],
|
||||
"DestSize": "std::max<uint8_t>(4, GetOpSize(_Src1))"
|
||||
"EmitValidation": [
|
||||
"Size >= 4"
|
||||
],
|
||||
"DestSize": "Size"
|
||||
},
|
||||
"GPR = Lshr GPR:$Src1, GPR:$Src2": {
|
||||
"GPR = Lshr u8:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer logical shift right"
|
||||
],
|
||||
"DestSize": "std::max<uint8_t>(4, GetOpSize(_Src1))"
|
||||
"EmitValidation": [
|
||||
"Size >= 4"
|
||||
],
|
||||
"DestSize": "Size"
|
||||
},
|
||||
"GPR = Ashr GPR:$Src1, GPR:$Src2": {
|
||||
"GPR = Ashr u8:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer arithmetic shift right"
|
||||
],
|
||||
"DestSize": "std::max<uint8_t>(4, GetOpSize(_Src1))"
|
||||
"EmitValidation": [
|
||||
"Size >= 4"
|
||||
],
|
||||
"DestSize": "Size"
|
||||
},
|
||||
"GPR = Ror GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer rotate right"
|
||||
@@ -869,12 +897,15 @@
|
||||
],
|
||||
"DestSize": "8"
|
||||
},
|
||||
"GPR = Select CondClass:$Cond, GPR:$Cmp1, GPR:$Cmp2, GPR:$TrueVal, GPR:$FalseVal, u8:$CompareSize": {
|
||||
"GPR = Select CondClass:$Cond, SSA:$Cmp1, SSA:$Cmp2, GPR:$TrueVal, GPR:$FalseVal, u8:$CompareSize": {
|
||||
"Desc": ["Ternary selection of GPRs",
|
||||
"op:",
|
||||
"Dest = Cmp1 <Cond> Cmp2 ? TrueVal : FalseVal"
|
||||
],
|
||||
"DestSize": "std::max<uint8_t>(4, std::max<uint8_t>(GetOpSize(_TrueVal), GetOpSize(_FalseVal)))"
|
||||
"DestSize": "std::max<uint8_t>(4, std::max<uint8_t>(GetOpSize(_TrueVal), GetOpSize(_FalseVal)))",
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Cmp1) == WalkFindRegClass($Cmp2)"
|
||||
]
|
||||
},
|
||||
"GPR = Extr GPR:$Upper, GPR:$Lower, u8:$LSB": {
|
||||
"Desc": ["Concats the two GPRs to create a value that is the size of the full two GPRs",
|
||||
@@ -1301,7 +1332,13 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
},
|
||||
|
||||
"FPR = VUABDL2 u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Desc": ["Unsigned Absolute Difference Long",
|
||||
"Using the high elements of the source vectors"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
},
|
||||
"FPR = VUShl u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, FPR:$ShiftVector": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
|
||||
+8
-8
@@ -103,7 +103,7 @@ static void PrintArg(fextl::stringstream *out, IRListView const* IR, OrderedNode
|
||||
if (ArgID.IsInvalid()) {
|
||||
*out << "%Invalid";
|
||||
} else {
|
||||
*out << "%ssa" << std::dec << ArgID;
|
||||
*out << "%" << std::dec << ArgID;
|
||||
if (RAData) {
|
||||
auto PhyReg = RAData->GetNodeRegister(ArgID);
|
||||
|
||||
@@ -202,8 +202,8 @@ void Dump(fextl::stringstream *out, IRListView const* IR, IR::RegisterAllocation
|
||||
|
||||
++CurrentIndent;
|
||||
AddIndent();
|
||||
*out << "(%ssa0) " << "IRHeader ";
|
||||
*out << "%ssa" << HeaderOp->Blocks.ID() << ", ";
|
||||
*out << "(%0) " << "IRHeader ";
|
||||
*out << "%" << HeaderOp->Blocks.ID() << ", ";
|
||||
*out << "#" << std::dec << HeaderOp->BlockCount << std::endl;
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
|
||||
@@ -211,10 +211,10 @@ void Dump(fextl::stringstream *out, IRListView const* IR, IR::RegisterAllocation
|
||||
auto BlockIROp = BlockHeader->C<FEXCore::IR::IROp_CodeBlock>();
|
||||
|
||||
AddIndent();
|
||||
*out << "(%ssa" << IR->GetID(BlockNode) << ") " << "CodeBlock ";
|
||||
*out << "(%" << IR->GetID(BlockNode) << ") " << "CodeBlock ";
|
||||
|
||||
*out << "%ssa" << BlockIROp->Begin.ID() << ", ";
|
||||
*out << "%ssa" << BlockIROp->Last.ID() << std::endl;
|
||||
*out << "%" << BlockIROp->Begin.ID() << ", ";
|
||||
*out << "%" << BlockIROp->Last.ID() << std::endl;
|
||||
}
|
||||
|
||||
++CurrentIndent;
|
||||
@@ -244,7 +244,7 @@ void Dump(fextl::stringstream *out, IRListView const* IR, IR::RegisterAllocation
|
||||
NumElements /= ElementSize;
|
||||
}
|
||||
|
||||
*out << "%ssa" << std::dec << ID;
|
||||
*out << "%" << std::dec << ID;
|
||||
|
||||
if (RAData) {
|
||||
auto PhyReg = RAData->GetNodeRegister(ID);
|
||||
@@ -284,7 +284,7 @@ void Dump(fextl::stringstream *out, IRListView const* IR, IR::RegisterAllocation
|
||||
NumElements = IROp->Size / ElementSize;
|
||||
}
|
||||
|
||||
*out << "(%ssa" << std::dec << ID << ' ';
|
||||
*out << "(%" << std::dec << ID << ' ';
|
||||
*out << 'i' << std::dec << (ElementSize * 8);
|
||||
if (NumElements > 1) {
|
||||
*out << 'v' << std::dec << NumElements;
|
||||
|
||||
+1
-1
@@ -413,7 +413,7 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
}
|
||||
|
||||
// Check if we are pulling in some IR from the IR Printer
|
||||
// Prints (%ssa%d) at the start of lines without a definition
|
||||
// Prints (%%d) at the start of lines without a definition
|
||||
if (Line[0] == '(') {
|
||||
size_t DefinitionEnd = fextl::string::npos;
|
||||
if ((DefinitionEnd = Line.find_first_of(')', CurrentPos)) != fextl::string::npos) {
|
||||
|
||||
+146
-2
@@ -561,6 +561,44 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
}
|
||||
*/
|
||||
|
||||
case OP_LOADMEMTSO: {
|
||||
auto Op = IROp->CW<IR::IROp_LoadMemTSO>();
|
||||
auto AddressHeader = IREmit->GetOpHeader(Op->Addr);
|
||||
|
||||
if (Op->Class == FEXCore::IR::FPRClass && AddressHeader->Op == OP_ADD && AddressHeader->Size == 8) {
|
||||
// TODO: LRCPC3 supports a vector unscaled offset like LRCPC2.
|
||||
// Support once hardware is available to use this.
|
||||
auto [OffsetType, OffsetScale, Arg0, Arg1] = MemExtendedAddressing(IREmit, IROp->Size, AddressHeader);
|
||||
|
||||
Op->OffsetType = OffsetType;
|
||||
Op->OffsetScale = OffsetScale;
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Addr_Index, Arg0); // Addr
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Offset_Index, Arg1); // Offset
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_STOREMEMTSO: {
|
||||
auto Op = IROp->CW<IR::IROp_StoreMemTSO>();
|
||||
auto AddressHeader = IREmit->GetOpHeader(Op->Addr);
|
||||
|
||||
if (Op->Class == FEXCore::IR::FPRClass && AddressHeader->Op == OP_ADD && AddressHeader->Size == 8) {
|
||||
// TODO: LRCPC3 supports a vector unscaled offset like LRCPC2.
|
||||
// Support once hardware is available to use this.
|
||||
auto [OffsetType, OffsetScale, Arg0, Arg1] = MemExtendedAddressing(IREmit, IROp->Size, AddressHeader);
|
||||
|
||||
Op->OffsetType = OffsetType;
|
||||
Op->OffsetScale = OffsetScale;
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Addr_Index, Arg0); // Addr
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Offset_Index, Arg1); // Offset
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_LOADMEM: {
|
||||
auto Op = IROp->CW<IR::IROp_LoadMem>();
|
||||
auto AddressHeader = IREmit->GetOpHeader(Op->Addr);
|
||||
@@ -652,6 +690,20 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_TESTNZ: {
|
||||
auto Op = IROp->CW<IR::IROp_TestNZ>();
|
||||
uint64_t Constant1{};
|
||||
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1)) {
|
||||
bool N = Constant1 & (1ull << ((Op->Size * 8) - 1));
|
||||
bool Z = Constant1 == 0;
|
||||
uint32_t NZVC = (N ? (1u << 31) : 0) | (Z ? (1u << 30) : 0);
|
||||
|
||||
IREmit->ReplaceWithConstant(CodeNode, NZVC);
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_OR: {
|
||||
auto Op = IROp->CW<IR::IROp_Or>();
|
||||
uint64_t Constant1{};
|
||||
@@ -669,6 +721,32 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_ORLSHL: {
|
||||
auto Op = IROp->CW<IR::IROp_Orlshl>();
|
||||
uint64_t Constant1{};
|
||||
uint64_t Constant2{};
|
||||
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) &&
|
||||
IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
uint64_t NewConstant = Constant1 | (Constant2 << Op->BitShift);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_ORLSHR: {
|
||||
auto Op = IROp->CW<IR::IROp_Orlshr>();
|
||||
uint64_t Constant1{};
|
||||
uint64_t Constant2{};
|
||||
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) &&
|
||||
IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
uint64_t NewConstant = Constant1 | (Constant2 >> Op->BitShift);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_XOR: {
|
||||
auto Op = IROp->C<IR::IROp_Xor>();
|
||||
uint64_t Constant1{};
|
||||
@@ -694,7 +772,9 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) &&
|
||||
IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
uint64_t NewConstant = (Constant1 << Constant2) & getMask(Op);
|
||||
// Shifts mask the shift amount by 63 or 31 depending on operating size;
|
||||
uint64_t ShiftMask = IROp->Size == 8 ? 63 : 31;
|
||||
uint64_t NewConstant = (Constant1 << (Constant2 & ShiftMask)) & getMask(Op);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
Changed = true;
|
||||
}
|
||||
@@ -720,7 +800,9 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) &&
|
||||
IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
uint64_t NewConstant = (Constant1 >> Constant2) & getMask(Op);
|
||||
// Shifts mask the shift amount by 63 or 31 depending on operating size;
|
||||
uint64_t ShiftMask = IROp->Size == 8 ? 63 : 31;
|
||||
uint64_t NewConstant = (Constant1 >> (Constant2 & ShiftMask)) & getMask(Op);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
Changed = true;
|
||||
}
|
||||
@@ -775,6 +857,45 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_SBFE: {
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
uint64_t Constant;
|
||||
if (IREmit->IsValueConstant(Op->Src, &Constant)) {
|
||||
// SBFE of a constant can be converted to a constant.
|
||||
uint64_t SourceMask = (1ULL << Op->Width) - 1;
|
||||
if (Op->Width == 64)
|
||||
SourceMask = ~0ULL;
|
||||
SourceMask <<= Op->lsb;
|
||||
|
||||
int64_t NewConstant = (Constant & SourceMask) >> Op->lsb;
|
||||
NewConstant <<= 64 - Op->Width;
|
||||
NewConstant >>= 64 - Op->Width;
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_BFI: {
|
||||
auto Op = IROp->C<IR::IROp_Bfi>();
|
||||
uint64_t ConstantDest{};
|
||||
uint64_t ConstantSrc{};
|
||||
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &ConstantDest) &&
|
||||
IREmit->IsValueConstant(Op->Header.Args[1], &ConstantSrc)) {
|
||||
|
||||
uint64_t SourceMask = (1ULL << Op->Width) - 1;
|
||||
if (Op->Width == 64)
|
||||
SourceMask = ~0ULL;
|
||||
|
||||
uint64_t NewConstant = ConstantDest & ~(SourceMask << Op->lsb);
|
||||
NewConstant |= (ConstantSrc & SourceMask) << Op->lsb;
|
||||
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_MUL: {
|
||||
auto Op = IROp->C<IR::IROp_Mul>();
|
||||
uint64_t Constant1{};
|
||||
@@ -797,7 +918,25 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_SELECT: {
|
||||
auto Op = IROp->C<IR::IROp_Select>();
|
||||
uint64_t Constant1{};
|
||||
uint64_t Constant2{};
|
||||
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) &&
|
||||
IREmit->IsValueConstant(Op->Header.Args[1], &Constant2) &&
|
||||
Op->Cond == COND_EQ) {
|
||||
|
||||
Constant1 &= getMask(Op);
|
||||
Constant2 &= getMask(Op);
|
||||
|
||||
bool is_true = Constant1 == Constant2;
|
||||
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(Op->Header.Args[is_true ? 2 : 3]));
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_CONDJUMP: {
|
||||
auto Op = IROp->CW<IR::IROp_CondJump>();
|
||||
|
||||
@@ -807,6 +946,11 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
// Fold the select into the CondJump if possible. Could handle more complex cases, too.
|
||||
if (Op->Cond.Val == COND_NEQ && IREmit->IsValueConstant(Op->Cmp2, &Constant) && Constant == 0 && Select->Op == OP_SELECT) {
|
||||
|
||||
const auto SelectCmpClass = IREmit->WalkFindRegClass(Select->Args[0]);
|
||||
if (SelectCmpClass == GPRPairClass) {
|
||||
// If the comparison class is a GPRPair then don't fold the select since it isn't free.
|
||||
break;
|
||||
}
|
||||
uint64_t Constant1{};
|
||||
uint64_t Constant2{};
|
||||
|
||||
|
||||
+57
-55
@@ -69,7 +69,9 @@ namespace {
|
||||
FEXCore::IR::RegisterClassType AccessRegClass;
|
||||
uint32_t AccessOffset;
|
||||
uint8_t AccessSize;
|
||||
FEXCore::IR::OrderedNode *Node;
|
||||
///< The last value that was loaded or stored.
|
||||
FEXCore::IR::OrderedNode *ValueNode;
|
||||
///< With a store access, the store node that is doing the operation.
|
||||
FEXCore::IR::OrderedNode *StoreNode;
|
||||
};
|
||||
|
||||
@@ -464,7 +466,7 @@ ContextMemberInfo *RCLSE::FindMemberInfo(ContextInfo *ContextClassificationInfo,
|
||||
return ContextClassificationInfo->Lookup.at(Offset);
|
||||
}
|
||||
|
||||
ContextMemberInfo *RCLSE::RecordAccess(ContextMemberInfo *Info, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size, LastAccessType AccessType, FEXCore::IR::OrderedNode *Node, FEXCore::IR::OrderedNode *StoreNode) {
|
||||
ContextMemberInfo *RCLSE::RecordAccess(ContextMemberInfo *Info, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size, LastAccessType AccessType, FEXCore::IR::OrderedNode *ValueNode, FEXCore::IR::OrderedNode *StoreNode) {
|
||||
LOGMAN_THROW_AA_FMT((Offset + Size) <= (Info->Class.Offset + Info->Class.Size), "Access to context item went over member size");
|
||||
LOGMAN_THROW_AA_FMT(Info->Accessed != ACCESS_INVALID, "Tried to access invalid member");
|
||||
|
||||
@@ -480,15 +482,15 @@ ContextMemberInfo *RCLSE::RecordAccess(ContextMemberInfo *Info, FEXCore::IR::Reg
|
||||
Info->AccessRegClass = RegClass;
|
||||
Info->AccessOffset = Offset;
|
||||
Info->AccessSize = Size;
|
||||
Info->Node = Node;
|
||||
Info->ValueNode = ValueNode;
|
||||
if (StoreNode != nullptr)
|
||||
Info->StoreNode = StoreNode;
|
||||
return Info;
|
||||
}
|
||||
|
||||
ContextMemberInfo *RCLSE::RecordAccess(ContextInfo *ClassifiedInfo, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size, LastAccessType AccessType, FEXCore::IR::OrderedNode *Node, FEXCore::IR::OrderedNode *StoreNode) {
|
||||
ContextMemberInfo *RCLSE::RecordAccess(ContextInfo *ClassifiedInfo, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size, LastAccessType AccessType, FEXCore::IR::OrderedNode *ValueNode, FEXCore::IR::OrderedNode *StoreNode) {
|
||||
ContextMemberInfo *Info = FindMemberInfo(ClassifiedInfo, Offset, Size);
|
||||
return RecordAccess(Info, RegClass, Offset, Size, AccessType, Node, StoreNode);
|
||||
return RecordAccess(Info, RegClass, Offset, Size, AccessType, ValueNode, StoreNode);
|
||||
}
|
||||
|
||||
void RCLSE::CalculateControlFlowInfo(FEXCore::IR::IREmitter *IREmit) {
|
||||
@@ -546,32 +548,32 @@ void RCLSE::CalculateControlFlowInfo(FEXCore::IR::IREmitter *IREmit) {
|
||||
* @brief This pass removes redundant pairs of storecontext and loadcontext ops
|
||||
*
|
||||
* eg.
|
||||
* %ssa26 i128 = LoadMem %ssa25 i64, 0x10
|
||||
* (%%ssa27) StoreContext %ssa26 i128, 0x10, 0xb0
|
||||
* %ssa28 i128 = LoadContext 0x10, 0x90
|
||||
* %ssa29 i128 = LoadContext 0x10, 0xb0
|
||||
* %26 i128 = LoadMem %25 i64, 0x10
|
||||
* (%%27) StoreContext %26 i128, 0x10, 0xb0
|
||||
* %28 i128 = LoadContext 0x10, 0x90
|
||||
* %29 i128 = LoadContext 0x10, 0xb0
|
||||
* Converts to
|
||||
* %ssa26 i128 = LoadMem %ssa25 i64, 0x10
|
||||
* (%%ssa27) StoreContext %ssa26 i128, 0x10, 0xb0
|
||||
* %ssa28 i128 = LoadContext 0x10, 0x90
|
||||
* %26 i128 = LoadMem %25 i64, 0x10
|
||||
* (%%27) StoreContext %26 i128, 0x10, 0xb0
|
||||
* %28 i128 = LoadContext 0x10, 0x90
|
||||
*
|
||||
* eg.
|
||||
* %ssa6 i128 = LoadContext 0x10, 0x90
|
||||
* %ssa7 i128 = LoadContext 0x10, 0x90
|
||||
* %ssa8 i128 = VXor %ssa7 i128, %ssa6 i128
|
||||
* %6 i128 = LoadContext 0x10, 0x90
|
||||
* %7 i128 = LoadContext 0x10, 0x90
|
||||
* %8 i128 = VXor %7 i128, %6 i128
|
||||
* Converts to
|
||||
* %ssa6 i128 = LoadContext 0x10, 0x90
|
||||
* %ssa7 i128 = VXor %ssa6 i128, %ssa6 i128
|
||||
* %6 i128 = LoadContext 0x10, 0x90
|
||||
* %7 i128 = VXor %6 i128, %6 i128
|
||||
*
|
||||
* eg.
|
||||
* (%%ssa189) StoreContext %ssa188 i128, 0x10, 0xa0
|
||||
* %ssa190 i128 = LoadContext 0x10, 0x90
|
||||
* %ssa192 i128 = VAdd %ssa188 i128, %ssa190 i128, 0x10, 0x4
|
||||
* (%%ssa193) StoreContext %ssa192 i128, 0x10, 0xa0
|
||||
* (%%189) StoreContext %188 i128, 0x10, 0xa0
|
||||
* %190 i128 = LoadContext 0x10, 0x90
|
||||
* %192 i128 = VAdd %188 i128, %190 i128, 0x10, 0x4
|
||||
* (%%193) StoreContext %192 i128, 0x10, 0xa0
|
||||
* Converts to
|
||||
* %ssa173 i128 = LoadContext 0x10, 0x90
|
||||
* %ssa175 i128 = VAdd %ssa172 i128, %ssa173 i128, 0x10, 0x4
|
||||
* (%%ssa176) StoreContext %ssa175 i128, 0x10, 0xa0
|
||||
* %173 i128 = LoadContext 0x10, 0x90
|
||||
* %175 i128 = VAdd %172 i128, %173 i128, 0x10, 0x4
|
||||
* (%%176) StoreContext %175 i128, 0x10, 0xa0
|
||||
|
||||
*/
|
||||
bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
@@ -624,7 +626,7 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
uint32_t LastOffset = Info->AccessOffset;
|
||||
uint8_t LastSize = Info->AccessSize;
|
||||
LastAccessType LastAccess = Info->Accessed;
|
||||
OrderedNode *LastNode = Info->Node;
|
||||
OrderedNode *LastValueNode = Info->ValueNode;
|
||||
OrderedNode *LastStoreNode = Info->StoreNode;
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, CodeNode);
|
||||
|
||||
@@ -637,7 +639,7 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
if (LastClass == GPRClass) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
|
||||
uint8_t TruncateSize = IREmit->GetOpSize(LastNode);
|
||||
uint8_t TruncateSize = IREmit->GetOpSize(LastValueNode);
|
||||
|
||||
// Did store context do an implicit truncation?
|
||||
if (IREmit->GetOpSize(LastStoreNode) < TruncateSize)
|
||||
@@ -647,20 +649,20 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
if (IROp->Size < TruncateSize)
|
||||
TruncateSize = IROp->Size;
|
||||
|
||||
if (TruncateSize != IREmit->GetOpSize(LastNode)) {
|
||||
if (TruncateSize != IREmit->GetOpSize(LastValueNode)) {
|
||||
// We need to insert an explict truncation
|
||||
LastNode = IREmit->_Bfe(Info->AccessSize, TruncateSize * 8, 0, LastNode);
|
||||
LastValueNode = IREmit->_Bfe(Info->AccessSize, TruncateSize * 8, 0, LastValueNode);
|
||||
}
|
||||
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastNode);
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
} else if (LastClass == FPRClass) {
|
||||
if (LastSize == IROp->Size && LastSize == IREmit->GetOpSize(LastNode)) {
|
||||
if (LastSize == IROp->Size && LastSize == IREmit->GetOpSize(LastValueNode)) {
|
||||
if (IsFullAccess(Info->Accessed)) {
|
||||
// LoadCtx matches StoreCtx and Node Size
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastNode);
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
}
|
||||
else {
|
||||
@@ -668,37 +670,37 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
// the vector element
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
// zext to size
|
||||
LastNode = IREmit->_VMov(IROp->Size, LastNode);
|
||||
LastValueNode = IREmit->_VMov(IROp->Size, LastValueNode);
|
||||
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastNode);
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
}
|
||||
} else if (LastSize >= IROp->Size &&
|
||||
IROp->Size == IREmit->GetOpSize(LastNode)) {
|
||||
IROp->Size == IREmit->GetOpSize(LastValueNode)) {
|
||||
// LoadCtx is <= StoreCtx and Node is LoadCtx
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastNode);
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
} else if (LastSize >= IROp->Size &&
|
||||
IROp->Size < IREmit->GetOpSize(LastNode)) {
|
||||
IROp->Size < IREmit->GetOpSize(LastValueNode)) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
// trucate to size
|
||||
LastNode = IREmit->_VMov(IROp->Size, LastNode);
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastNode);
|
||||
LastValueNode = IREmit->_VMov(IROp->Size, LastValueNode);
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
} else if (LastSize >= IROp->Size &&
|
||||
IROp->Size > IREmit->GetOpSize(LastNode)) {
|
||||
IROp->Size > IREmit->GetOpSize(LastValueNode)) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
// zext to size
|
||||
LastNode = IREmit->_VMov(IROp->Size, LastNode);
|
||||
LastValueNode = IREmit->_VMov(IROp->Size, LastValueNode);
|
||||
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastNode);
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
} else {
|
||||
//fmt::print("RCLSE: Not GPR class, missed, {}, lastS: {}, S: {}, Node S: {}\n", LastClass, LastSize, IROp->Size, IREmit->GetOpSize(LastNode));
|
||||
//fmt::print("RCLSE: Not GPR class, missed, {}, lastS: {}, S: {}, Node S: {}\n", LastClass, LastSize, IROp->Size, IREmit->GetOpSize(LastValueNode));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -708,8 +710,8 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
LastOffset == Op->Offset &&
|
||||
LastSize == IROp->Size) {
|
||||
// Did we read and then read again?
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastNode);
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
@@ -753,18 +755,18 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
auto Op = IROp->CW<IR::IROp_LoadFlag>();
|
||||
auto Info = FindMemberInfo(&LocalInfo, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag, 1);
|
||||
LastAccessType LastAccess = Info->Accessed;
|
||||
OrderedNode *LastNode = Info->Node;
|
||||
OrderedNode *LastValueNode = Info->ValueNode;
|
||||
|
||||
if (IsWriteAccess(LastAccess)) { // 1 byte so always a full write
|
||||
// If the last store matches this load value then we can replace the loaded value with the previous valid one
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, LastNode);
|
||||
RecordAccess(Info, FEXCore::IR::GPRClass, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag, 1, ACCESS_READ, LastNode);
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, LastValueNode);
|
||||
RecordAccess(Info, FEXCore::IR::GPRClass, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag, 1, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
}
|
||||
else if (IsReadAccess(LastAccess)) {
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, LastNode);
|
||||
RecordAccess(Info, FEXCore::IR::GPRClass, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag, 1, ACCESS_READ, LastNode);
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, LastValueNode);
|
||||
RecordAccess(Info, FEXCore::IR::GPRClass, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag, 1, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -169,6 +169,28 @@ bool DeadStoreElimination::Run(IREmitter *IREmit) {
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
|
||||
auto ClassifyRegisterStore = [this](Info &BlockInfo, uint32_t Offset, uint8_t Size) {
|
||||
//// GPR ////
|
||||
if (IsFullGPR(Offset, Size))
|
||||
BlockInfo.gpr.writes |= GPRBit(Offset);
|
||||
else
|
||||
BlockInfo.gpr.reads |= GPRBit(Offset);
|
||||
|
||||
//// FPR ////
|
||||
if (IsTrackedWriteFPR(Offset, Size))
|
||||
BlockInfo.fpr.writes |= FPRBit(Offset, Size);
|
||||
else
|
||||
BlockInfo.fpr.reads |= FPRBit(Offset, Size);
|
||||
};
|
||||
|
||||
auto ClassifyRegisterLoad = [this](Info &BlockInfo, uint32_t Offset, uint8_t Size) {
|
||||
//// GPR ////
|
||||
BlockInfo.gpr.reads |= GPRBit(Offset);
|
||||
|
||||
//// FPR ////
|
||||
BlockInfo.fpr.reads |= FPRBit(Offset, Size);
|
||||
};
|
||||
|
||||
//// Flags ////
|
||||
if (IROp->Op == OP_STOREFLAG) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
@@ -188,43 +210,16 @@ bool DeadStoreElimination::Run(IREmitter *IREmit) {
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
BlockInfo.flag.reads |= 1UL << Op->Flag;
|
||||
} else if (IROp->Op == OP_STORECONTEXT) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
} else if (IROp->Op == OP_STOREREGISTER) {
|
||||
auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
//// GPR ////
|
||||
if (IsFullGPR(Op->Offset, IROp->Size))
|
||||
BlockInfo.gpr.writes |= GPRBit(Op->Offset);
|
||||
else
|
||||
BlockInfo.gpr.reads |= GPRBit(Op->Offset);
|
||||
|
||||
//// FPR ////
|
||||
if (IsTrackedWriteFPR(Op->Offset, IROp->Size))
|
||||
BlockInfo.fpr.writes |= FPRBit(Op->Offset, IROp->Size);
|
||||
else
|
||||
BlockInfo.fpr.reads |= FPRBit(Op->Offset, IROp->Size);
|
||||
} else if (IROp->Op == OP_STORECONTEXTINDEXED ||
|
||||
IROp->Op == OP_LOADCONTEXTINDEXED) {
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
//// GPR ////
|
||||
// We can't track through these
|
||||
BlockInfo.gpr.reads = -1;
|
||||
|
||||
//// FPR ////
|
||||
// We can't track through these
|
||||
BlockInfo.fpr.reads = -1;
|
||||
} else if (IROp->Op == OP_LOADCONTEXT) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
ClassifyRegisterStore(BlockInfo, Op->Offset, IROp->Size);
|
||||
} else if (IROp->Op == OP_LOADREGISTER) {
|
||||
auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
//// GPR ////
|
||||
BlockInfo.gpr.reads |= GPRBit(Op->Offset);
|
||||
|
||||
//// FPR ////
|
||||
BlockInfo.fpr.reads |= FPRBit(Op->Offset, IROp->Size);
|
||||
ClassifyRegisterLoad(BlockInfo, Op->Offset, IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -321,6 +316,25 @@ bool DeadStoreElimination::Run(IREmitter *IREmit) {
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
|
||||
auto RemoveDeadRegisterStore = [this](FEXCore::IR::IREmitter *IREmit, FEXCore::IR::OrderedNode *CodeNode, Info &BlockInfo, uint32_t Offset, uint8_t Size) -> bool {
|
||||
bool Changed{};
|
||||
//// GPRs ////
|
||||
// If this OP_STOREREGISTER is never read, remove it
|
||||
if (BlockInfo.gpr.kill & GPRBit(Offset)) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Changed = true;
|
||||
}
|
||||
|
||||
//// FPRs ////
|
||||
// If this OP_STOREREGISTER is never read, remove it
|
||||
if ((BlockInfo.fpr.kill & FPRBit(Offset, Size)) == FPRBit(Offset, Size) && (FPRBit(Offset, Size) != 0)) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Changed = true;
|
||||
}
|
||||
|
||||
return Changed;
|
||||
};
|
||||
|
||||
//// Flags ////
|
||||
if (IROp->Op == OP_STOREFLAG) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
@@ -332,24 +346,12 @@ bool DeadStoreElimination::Run(IREmitter *IREmit) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Changed = true;
|
||||
}
|
||||
} else if (IROp->Op == OP_STORECONTEXT) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
} else if (IROp->Op == OP_STOREREGISTER) {
|
||||
auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
//// GPRs ////
|
||||
// If this OP_STORECONTEXT is never read, remove it
|
||||
if (BlockInfo.gpr.kill & GPRBit(Op->Offset)) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Changed = true;
|
||||
}
|
||||
|
||||
//// FPRs ////
|
||||
// If this OP_STORECONTEXT is never read, remove it
|
||||
if ((BlockInfo.fpr.kill & FPRBit(Op->Offset, IROp->Size)) == FPRBit(Op->Offset, IROp->Size) && (FPRBit(Op->Offset, IROp->Size) != 0)) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Changed = true;
|
||||
}
|
||||
Changed |= RemoveDeadRegisterStore(IREmit, CodeNode, BlockInfo, Op->Offset, IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -183,7 +183,7 @@ bool IRCompaction::Run(IREmitter *IREmit) {
|
||||
|
||||
#ifndef NDEBUG
|
||||
LOGMAN_THROW_A_FMT(NewArg.Value != UINT32_MAX,
|
||||
"Tried remapping unfound node %ssa{}", OldArg);
|
||||
"Tried remapping unfound node %{}", OldArg);
|
||||
#endif
|
||||
|
||||
LocalIROp->Args[i].NodeOffset = NewArg.Value * sizeof(OrderedNode);
|
||||
|
||||
+14
-14
@@ -82,13 +82,13 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
HadError |= OpSize == 0;
|
||||
// Does the op have a destination of size 0?
|
||||
if (OpSize == 0) {
|
||||
Errors << "%ssa" << ID << ": Had destination but with no size" << std::endl;
|
||||
Errors << "%" << ID << ": Had destination but with no size" << std::endl;
|
||||
}
|
||||
|
||||
// Does the node have zero uses? Should have been DCE'd
|
||||
if (CodeNode->GetUses() == 0) {
|
||||
HadWarning |= true;
|
||||
Warnings << "%ssa" << ID << ": Destination created but had no uses" << std::endl;
|
||||
Warnings << "%" << ID << ": Destination created but had no uses" << std::endl;
|
||||
}
|
||||
|
||||
if (RAData) {
|
||||
@@ -101,20 +101,20 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
// If no register class was assigned
|
||||
if (AssignedClass == IR::InvalidClass) {
|
||||
HadError |= true;
|
||||
Errors << "%ssa" << ID << ": Had destination but with no register class assigned" << std::endl;
|
||||
Errors << "%" << ID << ": Had destination but with no register class assigned" << std::endl;
|
||||
}
|
||||
|
||||
// If no physical register was assigned
|
||||
if (PhyReg.Reg == IR::InvalidReg) {
|
||||
HadError |= true;
|
||||
Errors << "%ssa" << ID << ": Had destination but with no register assigned" << std::endl;
|
||||
Errors << "%" << ID << ": Had destination but with no register assigned" << std::endl;
|
||||
}
|
||||
|
||||
// Assigned class wasn't the expected class and it is a non-complex op
|
||||
if (AssignedClass != ExpectedClass &&
|
||||
ExpectedClass != IR::ComplexClass) {
|
||||
HadWarning |= true;
|
||||
Warnings << "%ssa" << ID << ": Destination had register class " << AssignedClass.Val << " When register class " << ExpectedClass.Val << " Was expected" << std::endl;
|
||||
Warnings << "%" << ID << ": Destination had register class " << AssignedClass.Val << " When register class " << ExpectedClass.Val << " Was expected" << std::endl;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -128,12 +128,12 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
// Was an argument defined after this node?
|
||||
if (ArgID >= ID) {
|
||||
HadError |= true;
|
||||
Errors << "%ssa" << ID << ": Arg[" << i << "] has definition after use at %ssa" << ArgID << std::endl;
|
||||
Errors << "%" << ID << ": Arg[" << i << "] has definition after use at %" << ArgID << std::endl;
|
||||
}
|
||||
|
||||
if (ArgID.IsValid() && !NodeIsLive.Get(ArgID.Value)) {
|
||||
HadError |= true;
|
||||
Errors << "%ssa" << ID << ": Arg[" << i << "] references dead %ssa" << ArgID << std::endl;
|
||||
Errors << "%" << ID << ": Arg[" << i << "] references dead %" << ArgID << std::endl;
|
||||
}
|
||||
|
||||
if (ArgID.IsValid()) {
|
||||
@@ -162,7 +162,7 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
|
||||
if (TrueTargetOp->Op != OP_CODEBLOCK) {
|
||||
HadError |= true;
|
||||
Errors << "CondJump %ssa" << ID << ": True Target Jumps to Op that isn't the begining of a block" << std::endl;
|
||||
Errors << "CondJump %" << ID << ": True Target Jumps to Op that isn't the begining of a block" << std::endl;
|
||||
}
|
||||
else {
|
||||
auto Block = OffsetToBlockMap.try_emplace(Op->TrueBlock.ID()).first;
|
||||
@@ -171,7 +171,7 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
|
||||
if (FalseTargetOp->Op != OP_CODEBLOCK) {
|
||||
HadError |= true;
|
||||
Errors << "CondJump %ssa" << ID << ": False Target Jumps to Op that isn't the begining of a block" << std::endl;
|
||||
Errors << "CondJump %" << ID << ": False Target Jumps to Op that isn't the begining of a block" << std::endl;
|
||||
}
|
||||
else {
|
||||
auto Block = OffsetToBlockMap.try_emplace(Op->FalseBlock.ID()).first;
|
||||
@@ -188,7 +188,7 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
FEXCore::IR::IROp_Header const *TargetOp = CurrentIR.GetOp<IROp_Header>(TargetNode);
|
||||
if (TargetOp->Op != OP_CODEBLOCK) {
|
||||
HadError |= true;
|
||||
Errors << "Jump %ssa" << ID << ": Jump to Op that isn't the begining of a block" << std::endl;
|
||||
Errors << "Jump %" << ID << ": Jump to Op that isn't the begining of a block" << std::endl;
|
||||
}
|
||||
else {
|
||||
auto Block = OffsetToBlockMap.try_emplace(Op->Header.Args[0].ID()).first;
|
||||
@@ -206,7 +206,7 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
size_t NumSuccessors = CurrentBlock->Successors.size();
|
||||
if (NumSuccessors > 2) {
|
||||
HadError |= true;
|
||||
Errors << "%ssa" << BlockID << " Has " << NumSuccessors << " successors which is too many" << std::endl;
|
||||
Errors << "%" << BlockID << " Has " << NumSuccessors << " successors which is too many" << std::endl;
|
||||
}
|
||||
|
||||
{
|
||||
@@ -222,7 +222,7 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
auto Op = GetOp(CodeCurrent);
|
||||
if (Op != IR::OP_ENDBLOCK) {
|
||||
HadError |= true;
|
||||
Errors << "%ssa" << BlockID << " Failed to end block with EndBlock" << std::endl;
|
||||
Errors << "%" << BlockID << " Failed to end block with EndBlock" << std::endl;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -233,7 +233,7 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
auto Op = GetOp(CodeCurrent);
|
||||
if (!IsBlockExit(Op)) {
|
||||
HadError |= true;
|
||||
Errors << "%ssa" << BlockID << " Didn't have a block exit IR op as its last instruction" << std::endl;
|
||||
Errors << "%" << BlockID << " Didn't have a block exit IR op as its last instruction" << std::endl;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -243,7 +243,7 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
auto [Node, IROp] = CurrentIR.at(IR::NodeID{i})();
|
||||
if (Node->NumUses != Uses[i] && IROp->Op != OP_CODEBLOCK && IROp->Op != OP_IRHEADER) {
|
||||
HadError |= true;
|
||||
Errors << "%ssa" << i << " Has " << Uses[i] << " Uses, but reports " << Node->NumUses << std::endl;
|
||||
Errors << "%" << i << " Has " << Uses[i] << " Uses, but reports " << Node->NumUses << std::endl;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -25,14 +25,9 @@ private:
|
||||
};
|
||||
|
||||
bool LongDivideEliminationPass::IsZeroOp(IREmitter *IREmit, OrderedNodeWrapper Arg) {
|
||||
auto IROp = IREmit->GetOpHeader(Arg);
|
||||
uint64_t Value;
|
||||
|
||||
// XOR based zero
|
||||
if (IROp->Op == OP_XOR) {
|
||||
return IROp->Args[0] == IROp->Args[1];
|
||||
}
|
||||
else if (IREmit->IsValueConstant(Arg, &Value)) {
|
||||
if (IREmit->IsValueConstant(Arg, &Value)) {
|
||||
// Zero constant based zero op
|
||||
return Value == 0;
|
||||
}
|
||||
@@ -87,8 +82,7 @@ bool LongDivideEliminationPass::Run(IREmitter *IREmit) {
|
||||
else if (IROp->Op == OP_LUDIV ||
|
||||
IROp->Op == OP_LUREM) {
|
||||
auto Op = IROp->C<IR::IROp_LUDiv>();
|
||||
// Check upper Op to see if it came from a xor zeroing op
|
||||
// XOR: Result = _Xor(Dest, Src);
|
||||
// Check upper Op to see if it came from a zeroing op
|
||||
// If it does then it we only need a 64bit UDIV
|
||||
if (IsZeroOp(IREmit, Op->Upper)) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
|
||||
@@ -47,7 +47,7 @@ bool PhiValidation::Run(IREmitter *IREmit) {
|
||||
// If we have found a non-phi IR op and then had a Phi or PhiValue value then this is a programming mistake
|
||||
// PHI values MUST be defined at the top of the block only
|
||||
HadError |= true;
|
||||
Errors << "Phi %ssa" << CurrentIR.GetID(CodeNode) << ": Was defined after non-phi operations. Which is invalid!" << std::endl;
|
||||
Errors << "Phi %" << CurrentIR.GetID(CodeNode) << ": Was defined after non-phi operations. Which is invalid!" << std::endl;
|
||||
}
|
||||
|
||||
// Check all the phi values to ensure they have the same type
|
||||
|
||||
@@ -297,28 +297,28 @@ bool RAValidation::Run(IREmitter *IREmit) {
|
||||
auto CurrentSSAAtReg = BlockRegState.Get(PhyReg);
|
||||
if (CurrentSSAAtReg == RegState::InvalidReg) {
|
||||
HadError |= true;
|
||||
Errors << fextl::fmt::format("%ssa{}: Arg[{}] unknown Reg: {}, class: {}\n", ID, i, PhyReg.Reg, PhyReg.Class);
|
||||
Errors << fextl::fmt::format("%{}: Arg[{}] unknown Reg: {}, class: {}\n", ID, i, PhyReg.Reg, PhyReg.Class);
|
||||
} else if (CurrentSSAAtReg == RegState::CorruptedPair) {
|
||||
HadError |= true;
|
||||
|
||||
auto Lower = BlockRegState.Get(PhysicalRegister(GPRClass, uint8_t(PhyReg.Reg*2) + 1));
|
||||
auto Upper = BlockRegState.Get(PhysicalRegister(GPRClass, PhyReg.Reg*2 + 1));
|
||||
|
||||
Errors << fextl::fmt::format("%ssa{}: Arg[{}] expects paired reg{} to contain %ssa{}, but it actually contains {{%ssa{}, %ssa{}}}\n",
|
||||
Errors << fextl::fmt::format("%{}: Arg[{}] expects paired reg{} to contain %{}, but it actually contains {{%{}, %{}}}\n",
|
||||
ID, i, PhyReg.Reg, ArgID, Lower, Upper);
|
||||
} else if (CurrentSSAAtReg == RegState::UninitializedValue) {
|
||||
HadError |= true;
|
||||
|
||||
Errors << fextl::fmt::format("%ssa{}: Arg[{}] expects reg{} to contain %ssa{}, but it is uninitialized\n",
|
||||
Errors << fextl::fmt::format("%{}: Arg[{}] expects reg{} to contain %{}, but it is uninitialized\n",
|
||||
ID, i, PhyReg.Reg, ArgID);
|
||||
} else if (CurrentSSAAtReg == RegState::ClobberedValue) {
|
||||
HadError |= true;
|
||||
|
||||
Errors << fextl::fmt::format("%ssa{}: Arg[{}] expects reg{} to contain %ssa{}, but contents vary depending on control flow\n",
|
||||
Errors << fextl::fmt::format("%{}: Arg[{}] expects reg{} to contain %{}, but contents vary depending on control flow\n",
|
||||
ID, i, PhyReg.Reg, ArgID);
|
||||
} else if (CurrentSSAAtReg != ArgID) {
|
||||
HadError |= true;
|
||||
Errors << fextl::fmt::format("%ssa{}: Arg[{}] expects reg{} to contain %ssa{}, but it actually contains %ssa{}\n",
|
||||
Errors << fextl::fmt::format("%{}: Arg[{}] expects reg{} to contain %{}, but it actually contains %{}\n",
|
||||
ID, i, PhyReg.Reg, ArgID, CurrentSSAAtReg);
|
||||
}
|
||||
};
|
||||
@@ -343,15 +343,15 @@ bool RAValidation::Run(IREmitter *IREmit) {
|
||||
|
||||
if (Value == RegState::UninitializedValue) {
|
||||
HadError |= true;
|
||||
Errors << fextl::fmt::format("%ssa{}: FillRegister expected %ssa{} in Slot {}, but was undefined in at least one control flow path\n",
|
||||
Errors << fextl::fmt::format("%{}: FillRegister expected %{} in Slot {}, but was undefined in at least one control flow path\n",
|
||||
ID, ExpectedValue, FillRegister->Slot);
|
||||
} else if (Value == RegState::ClobberedValue) {
|
||||
HadError |= true;
|
||||
Errors << fextl::fmt::format("%ssa{}: FillRegister expected %ssa{} in Slot {}, but contents vary depending on control flow\n",
|
||||
Errors << fextl::fmt::format("%{}: FillRegister expected %{} in Slot {}, but contents vary depending on control flow\n",
|
||||
ID, ExpectedValue, FillRegister->Slot);
|
||||
} else if (Value != ExpectedValue) {
|
||||
HadError |= true;
|
||||
Errors << fextl::fmt::format("%ssa{}: FillRegister expected %ssa{} in Slot {}, but it actually contains %ssa{}\n",
|
||||
Errors << fextl::fmt::format("%{}: FillRegister expected %{} in Slot {}, but it actually contains %{}\n",
|
||||
ID, ExpectedValue, FillRegister->Slot, Value);
|
||||
}
|
||||
break;
|
||||
|
||||
@@ -494,7 +494,7 @@ namespace {
|
||||
const auto ArgNode = Arg.ID();
|
||||
auto& ArgNodeLiveRange = LiveRanges[ArgNode.Value];
|
||||
LOGMAN_THROW_AA_FMT(ArgNodeLiveRange.Begin.Value != UINT32_MAX,
|
||||
"%ssa{} used by %ssa{} before defined?", ArgNode, Node);
|
||||
"%{} used by %{} before defined?", ArgNode, Node);
|
||||
|
||||
const auto ArgNodeBlockID = Graph->Nodes[ArgNode.Value].Head.BlockID;
|
||||
if (ArgNodeBlockID == BlockNodeID) {
|
||||
@@ -1311,7 +1311,7 @@ namespace {
|
||||
|
||||
if (!CurrentNodes.contains(InterferenceNode)) {
|
||||
InterferenceIdToSpill = InterferenceNode;
|
||||
LogMan::Msg::DFmt("Panic spilling %ssa{}, Live Range[{}, {})", InterferenceIdToSpill, InterferenceLiveRange->Begin, InterferenceLiveRange->End);
|
||||
LogMan::Msg::DFmt("Panic spilling %{}, Live Range[{}, {})", InterferenceIdToSpill, InterferenceLiveRange->Begin, InterferenceLiveRange->End);
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
@@ -1320,14 +1320,14 @@ namespace {
|
||||
|
||||
if (InterferenceIdToSpill.IsInvalid()) {
|
||||
int j = 0;
|
||||
LogMan::Msg::DFmt("node %ssa{}, was dumped in to virtual reg {}. Live Range[{}, {})",
|
||||
LogMan::Msg::DFmt("node %{}, was dumped in to virtual reg {}. Live Range[{}, {})",
|
||||
CurrentLocation, -1,
|
||||
OpLiveRange->Begin, OpLiveRange->End);
|
||||
|
||||
RegisterNode->Interferences.Iterate([&](IR::NodeID InterferenceNode) {
|
||||
auto *InterferenceLiveRange = &LiveRanges[InterferenceNode.Value];
|
||||
|
||||
LogMan::Msg::DFmt("\tInt{}: %ssa{} Remat: {} [{}, {})", j++, InterferenceNode, InterferenceLiveRange->RematCost, InterferenceLiveRange->Begin, InterferenceLiveRange->End);
|
||||
LogMan::Msg::DFmt("\tInt{}: %{} Remat: {} [{}, {})", j++, InterferenceNode, InterferenceLiveRange->RematCost, InterferenceLiveRange->Begin, InterferenceLiveRange->End);
|
||||
});
|
||||
}
|
||||
LOGMAN_THROW_A_FMT(InterferenceIdToSpill.IsValid(), "Couldn't find Node to spill");
|
||||
@@ -1395,7 +1395,7 @@ namespace {
|
||||
auto FirstUseLocation = FindFirstUse(IREmit, ConstantNode, NextIter, NodeIterator::Invalid());
|
||||
|
||||
LOGMAN_THROW_A_FMT(FirstUseLocation != IR::NodeIterator::Invalid(),
|
||||
"At %ssa{} Spilling Op %ssa{} but Failure to find op use",
|
||||
"At %{} Spilling Op %{} but Failure to find op use",
|
||||
Node, *InterferenceNode);
|
||||
|
||||
if (FirstUseLocation != IR::NodeIterator::Invalid()) {
|
||||
@@ -1454,7 +1454,7 @@ namespace {
|
||||
auto FirstUseLocation = FindFirstUse(IREmit, InterferenceOrderedNode, FirstIter, NodeIterator::Invalid());
|
||||
|
||||
LOGMAN_THROW_A_FMT(FirstUseLocation != NodeIterator::Invalid(),
|
||||
"At %ssa{} Spilling Op %ssa{} but Failure to find op use",
|
||||
"At %{} Spilling Op %{} but Failure to find op use",
|
||||
Node, *InterferenceNode);
|
||||
|
||||
if (FirstUseLocation != IR::NodeIterator::Invalid()) {
|
||||
|
||||
@@ -107,18 +107,18 @@ bool ValueDominanceValidation::Run(IREmitter *IREmit) {
|
||||
// then it must only be declared prior to this instruction
|
||||
// Eg: Valid
|
||||
// CodeBlock_1:
|
||||
// %ssa_1 = Load
|
||||
// %ssa_2 = Load
|
||||
// %ssa_3 = <Op> %ssa_1, %ssa_2
|
||||
// %_1 = Load
|
||||
// %_2 = Load
|
||||
// %_3 = <Op> %_1, %_2
|
||||
//
|
||||
// Eg: Invalid
|
||||
// CodeBlock_1:
|
||||
// %ssa_1 = Load
|
||||
// %ssa_2 = <Op> %ssa_1, %ssa_3
|
||||
// %ssa_3 = Load
|
||||
// %_1 = Load
|
||||
// %_2 = <Op> %_1, %_3
|
||||
// %_3 = Load
|
||||
if (Arg.ID() > CodeID) {
|
||||
HadError |= true;
|
||||
Errors << "Inst %ssa" << CodeID << ": Arg[" << i << "] %ssa" << Arg.ID() << " definition does not dominate this use!" << std::endl;
|
||||
Errors << "Inst %" << CodeID << ": Arg[" << i << "] %" << Arg.ID() << " definition does not dominate this use!" << std::endl;
|
||||
}
|
||||
}
|
||||
else if (Arg.ID() < BlockIROp->Begin.ID()) {
|
||||
@@ -127,21 +127,21 @@ bool ValueDominanceValidation::Run(IREmitter *IREmit) {
|
||||
|
||||
// Eg: Valid
|
||||
// CodeBlock_1:
|
||||
// %ssa_1 = Load
|
||||
// %ssa_2 = Load
|
||||
// %_1 = Load
|
||||
// %_2 = Load
|
||||
// Jump %CodeBlock_2
|
||||
//
|
||||
// CodeBlock_2:
|
||||
// %ssa_3 = <Op> %ssa_1, %ssa_2
|
||||
// %_3 = <Op> %_1, %_2
|
||||
//
|
||||
// Eg: Invalid
|
||||
// CodeBlock_1:
|
||||
// %ssa_1 = Load
|
||||
// %ssa_2 = Load
|
||||
// %_1 = Load
|
||||
// %_2 = Load
|
||||
// Jump %CodeBlock_3
|
||||
//
|
||||
// CodeBlock_2:
|
||||
// %ssa_3 = <Op> %ssa_1, %ssa_2
|
||||
// %_3 = <Op> %_1, %_2
|
||||
//
|
||||
// CodeBlock_3:
|
||||
// ...
|
||||
@@ -171,26 +171,26 @@ bool ValueDominanceValidation::Run(IREmitter *IREmit) {
|
||||
FoundPredDefine = true;
|
||||
break;
|
||||
}
|
||||
Errors << "\tChecking Pred %ssa" << CurrentIR.GetID(Pred) << std::endl;
|
||||
Errors << "\tChecking Pred %" << CurrentIR.GetID(Pred) << std::endl;
|
||||
}
|
||||
|
||||
if (!FoundPredDefine) {
|
||||
HadError |= true;
|
||||
Errors << "Inst %ssa" << CodeID << ": Arg[" << i << "] %ssa" << Arg.ID() << " definition does not dominate this use! But was defined before this block!" << std::endl;
|
||||
Errors << "Inst %" << CodeID << ": Arg[" << i << "] %" << Arg.ID() << " definition does not dominate this use! But was defined before this block!" << std::endl;
|
||||
}
|
||||
}
|
||||
else if (Arg.ID() > BlockIROp->Last.ID()) {
|
||||
// If this SSA argument is defined AFTER this block then it is just completely broken
|
||||
// Eg: Invalid
|
||||
// CodeBlock_1:
|
||||
// %ssa_1 = Load
|
||||
// %ssa_2 = <Op> %ssa_1, %ssa_3
|
||||
// %_1 = Load
|
||||
// %_2 = <Op> %_1, %_3
|
||||
// Jump %CodeBlock_2
|
||||
//
|
||||
// CodeBlock_2:
|
||||
// %ssa_3 = Load
|
||||
// %_3 = Load
|
||||
HadError |= true;
|
||||
Errors << "Inst %ssa" << CodeID << ": Arg[" << i << "] %ssa" << Arg.ID() << " definition does not dominate this use!" << std::endl;
|
||||
Errors << "Inst %" << CodeID << ": Arg[" << i << "] %" << Arg.ID() << " definition does not dominate this use!" << std::endl;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+8
-12
@@ -2060,12 +2060,11 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
LDR |= Size << 30;
|
||||
LDR |= AddrReg << 5;
|
||||
LDR |= DataReg;
|
||||
PC[-1] = DMB;
|
||||
PC[0] = LDR;
|
||||
PC[1] = DMB;
|
||||
PC[1] = DMB_LD; // Back-patch the half-barrier.
|
||||
ClearICache(&PC[-1], 16);
|
||||
// Back up one instruction and have another go
|
||||
return std::make_pair(true, -4);
|
||||
// With the instruction modified, now execute again.
|
||||
return std::make_pair(true, 0);
|
||||
}
|
||||
}
|
||||
else if ( (Instr & LDAXR_MASK) == STLR_INST) { // STLR*
|
||||
@@ -2084,9 +2083,8 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
STR |= Size << 30;
|
||||
STR |= AddrReg << 5;
|
||||
STR |= DataReg;
|
||||
PC[-1] = DMB;
|
||||
PC[-1] = DMB; // Back-patch the half-barrier.
|
||||
PC[0] = STR;
|
||||
PC[1] = DMB;
|
||||
ClearICache(&PC[-1], 16);
|
||||
// Back up one instruction and have another go
|
||||
return std::make_pair(true, -4);
|
||||
@@ -2111,12 +2109,11 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
LDUR |= AddrReg << 5;
|
||||
LDUR |= DataReg;
|
||||
LDUR |= Instr & (0b1'1111'1111 << 9);
|
||||
PC[-1] = DMB;
|
||||
PC[0] = LDUR;
|
||||
PC[1] = DMB;
|
||||
PC[1] = DMB_LD; // Back-patch the half-barrier.
|
||||
ClearICache(&PC[-1], 16);
|
||||
// Back up one instruction and have another go
|
||||
return std::make_pair(true, -4);
|
||||
// With the instruction modified, now execute again.
|
||||
return std::make_pair(true, 0);
|
||||
}
|
||||
}
|
||||
else if ((Instr & RCPC2_MASK) == STLUR_INST) { // STLUR*
|
||||
@@ -2138,9 +2135,8 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
STUR |= AddrReg << 5;
|
||||
STUR |= DataReg;
|
||||
STUR |= Instr & (0b1'1111'1111 << 9);
|
||||
PC[-1] = DMB;
|
||||
PC[-1] = DMB; // Back-patch the half-barrier.
|
||||
PC[0] = STUR;
|
||||
PC[1] = DMB;
|
||||
ClearICache(&PC[-1], 16);
|
||||
// Back up one instruction and have another go
|
||||
return std::make_pair(true, -4);
|
||||
|
||||
+1
-1
@@ -61,7 +61,7 @@ namespace FEXCore::Telemetry {
|
||||
}
|
||||
}
|
||||
|
||||
Value &GetObject(TelemetryType Type) {
|
||||
Value &GetTelemetryValue(TelemetryType Type) {
|
||||
return TelemetryValues.at(Type);
|
||||
}
|
||||
#endif
|
||||
|
||||
Vendored
+10
-10
@@ -32,17 +32,17 @@ SSA is quite nice to work with when translating the x86-64 code to the IR, when
|
||||
* Read the python generation file to determine the extent of what it can do
|
||||
|
||||
## IR function considerations
|
||||
The first SSA node is a special case node that is considered invalid. This means %ssa0 will always be invalid for "null" node checks
|
||||
The first real SSA node also has to be a IRHeader node. This means it is safe to assume that %ssa1 will always be an IRHeader.
|
||||
The first SSA node is a special case node that is considered invalid. This means %0 will always be invalid for "null" node checks
|
||||
The first real SSA node also has to be a IRHeader node. This means it is safe to assume that %1 will always be an IRHeader.
|
||||
|
||||
|
||||
```(%%ssa1) IRHeader 0x41a9a0, %%ssa2, 5```
|
||||
```(%%1) IRHeader 0x41a9a0, %%2, 5```
|
||||
|
||||
The header provides information about that function like the entry point address.
|
||||
Additionally it also points to the first `CodeBlock` IROp
|
||||
|
||||
|
||||
```(%%ssa2) CodeBlock %%ssa7, %%ssa168, %%ssa3```
|
||||
```(%%2) CodeBlock %%7, %%168, %%3```
|
||||
|
||||
|
||||
* The `CodeBlock` Op is a jump target and must be treated as if it'll be jumped to from other blocks
|
||||
@@ -54,12 +54,12 @@ Additionally it also points to the first `CodeBlock` IROp
|
||||
### Example code block
|
||||
|
||||
```
|
||||
(%%ssa3) CodeBlock %%ssa169, %%ssa173, %%ssa4
|
||||
(%%ssa169) BeginBlock %ssa3
|
||||
%ssa170 i64 = Constant 0x41a9e1
|
||||
(%%ssa171) StoreContext %ssa170 i64, 0x8, 0x0
|
||||
(%%ssa172) ExitFunction
|
||||
(%%ssa173) EndBlock %ssa3
|
||||
(%%3) CodeBlock %%169, %%173, %%4
|
||||
(%%169) BeginBlock %3
|
||||
%170 i64 = Constant 0x41a9e1
|
||||
(%%171) StoreContext %170 i64, 0x8, 0x0
|
||||
(%%172) ExitFunction
|
||||
(%%173) EndBlock %3
|
||||
```
|
||||
|
||||
* BeginBlock points back to the CodeBlock SSA which helps with iterating across multiple blocks
|
||||
|
||||
+17
-17
@@ -17,23 +17,23 @@ Ex:
|
||||
Translates to the IR of:
|
||||
```
|
||||
BeginBlock
|
||||
%ssa8 i32 = Constant 0x1
|
||||
StoreContext 0x8, 0x8, %ssa8
|
||||
%ssa64 i32 = Constant 0x1
|
||||
StoreContext 0x8, 0x30, %ssa64
|
||||
%ssa120 i32 = Constant 0x1f
|
||||
StoreContext 0x8, 0x28, %ssa120
|
||||
%ssa176 i32 = Constant 0x1
|
||||
StoreContext 0x8, 0x20, %ssa176
|
||||
%ssa232 i64 = LoadContext 0x8, 0x8
|
||||
%ssa264 i64 = LoadContext 0x8, 0x30
|
||||
%ssa296 i64 = LoadContext 0x8, 0x28
|
||||
%ssa328 i64 = LoadContext 0x8, 0x20
|
||||
%ssa360 i64 = LoadContext 0x8, 0x58
|
||||
%ssa392 i64 = LoadContext 0x8, 0x48
|
||||
%ssa424 i64 = LoadContext 0x8, 0x50
|
||||
%ssa456 i64 = Syscall%ssa232, %ssa264, %ssa296, %ssa328, %ssa360, %ssa392, %ssa424
|
||||
StoreContext 0x8, 0x8, %ssa456
|
||||
%8 i32 = Constant 0x1
|
||||
StoreContext 0x8, 0x8, %8
|
||||
%64 i32 = Constant 0x1
|
||||
StoreContext 0x8, 0x30, %64
|
||||
%120 i32 = Constant 0x1f
|
||||
StoreContext 0x8, 0x28, %120
|
||||
%176 i32 = Constant 0x1
|
||||
StoreContext 0x8, 0x20, %176
|
||||
%232 i64 = LoadContext 0x8, 0x8
|
||||
%264 i64 = LoadContext 0x8, 0x30
|
||||
%296 i64 = LoadContext 0x8, 0x28
|
||||
%328 i64 = LoadContext 0x8, 0x20
|
||||
%360 i64 = LoadContext 0x8, 0x58
|
||||
%392 i64 = LoadContext 0x8, 0x48
|
||||
%424 i64 = LoadContext 0x8, 0x50
|
||||
%456 i64 = Syscall%232, %264, %296, %328, %360, %392, %424
|
||||
StoreContext 0x8, 0x8, %456
|
||||
BeginBlock
|
||||
EndBlock 0x1e
|
||||
ExitFunction
|
||||
|
||||
+1
-1
@@ -30,7 +30,7 @@ Large amount of x86-64 instructions load or store registers in order from the co
|
||||
We can merge these in to loadstore pair ops to improve perf
|
||||
### Function level heuristic pass
|
||||
Once we know that a function is a true full recompile we can do some additional optimizations.
|
||||
Remove any final flag stores. We know that a compiler won't pass flags past a function call boundry(It doesn't exist in the ABI)
|
||||
Remove any final flag stores. We know that a compiler won't pass flags past a function call boundary(It doesn't exist in the ABI)
|
||||
Remove any loadstores to the context mid function, only do a final store at the end of the function and do loads at the start. Which means ops just map registers directly throughout the entire function.
|
||||
### SIMD coalescing pass?
|
||||
When operating on older MMX ops(64bit SIMD) and they may end up up generating some independent ops that can be coalesced in to a 128bit op
|
||||
@@ -2,12 +2,16 @@
|
||||
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/EnumOperators.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/list.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
|
||||
#include <charconv>
|
||||
#include <optional>
|
||||
#include <stdint.h>
|
||||
|
||||
@@ -52,6 +56,9 @@ namespace Handler {
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
};
|
||||
|
||||
#define ENUMDEFINES
|
||||
#include <FEXCore/Config/ConfigOptions.inl>
|
||||
|
||||
enum ConfigCore {
|
||||
CONFIG_INTERPRETER,
|
||||
CONFIG_IRJIT,
|
||||
@@ -83,6 +90,42 @@ namespace Handler {
|
||||
LAYER_TOP,
|
||||
};
|
||||
|
||||
template<typename PairTypes>
|
||||
static inline fextl::string EnumParser(PairTypes const &EnumPairs, std::string_view const View) {
|
||||
uint64_t EnumMask{};
|
||||
auto Results = std::from_chars(View.data(), View.data() + View.size(), EnumMask);
|
||||
if (Results.ec == std::errc()) {
|
||||
// If the data is a valid number, just pass it through.
|
||||
return View.data();
|
||||
}
|
||||
|
||||
auto Begin = 0;
|
||||
auto End = View.find_first_of(',');
|
||||
std::string_view Option = View.substr(Begin, End);
|
||||
while (Option.size() != 0) {
|
||||
auto EnumValue = std::find_if(EnumPairs.begin(), EnumPairs.end(),
|
||||
[Option](const DisassembleConfigPair &Value) -> bool {
|
||||
return Value.first == Option;
|
||||
});
|
||||
|
||||
if (EnumValue == EnumPairs.end()) {
|
||||
LogMan::Msg::IFmt("Skipping Unknown option: {}", Option);
|
||||
}
|
||||
else {
|
||||
EnumMask |= FEXCore::ToUnderlying(EnumValue->second);
|
||||
}
|
||||
|
||||
if (End == std::string::npos) {
|
||||
break;
|
||||
}
|
||||
Begin = End + 1;
|
||||
End = View.find_first_of(',', Begin);
|
||||
Option = View.substr(Begin, End);
|
||||
}
|
||||
|
||||
return fextl::fmt::format("{}", EnumMask);
|
||||
}
|
||||
|
||||
namespace DefaultValues {
|
||||
#define P(x) x
|
||||
#define OPT_BASE(type, group, enum, json, default) extern const P(type) P(enum);
|
||||
|
||||
@@ -35,6 +35,8 @@ namespace CodeSerialize {
|
||||
namespace CPU {
|
||||
struct CPUBackendFeatures {
|
||||
bool SupportsStaticRegisterAllocation = false;
|
||||
bool SupportsShiftedBitwise = false;
|
||||
bool SupportsFlags = false;
|
||||
};
|
||||
|
||||
class CPUBackend {
|
||||
|
||||
+18
-8
@@ -86,9 +86,17 @@ namespace FEXCore::Context {
|
||||
void *VDSO_kernel_rt_sigreturn;
|
||||
};
|
||||
|
||||
using CustomCPUFactoryType = std::function<fextl::unique_ptr<FEXCore::CPU::CPUBackend> (FEXCore::Context::Context*, FEXCore::Core::InternalThreadState *Thread)>;
|
||||
using CodeRangeInvalidationFn = std::function<void(uint64_t start, uint64_t Length)>;
|
||||
|
||||
using ExitHandler = std::function<void(uint64_t ThreadId, FEXCore::Context::ExitReason)>;
|
||||
using CustomCPUFactoryType = std::function<fextl::unique_ptr<CPU::CPUBackend>(Context*, Core::InternalThreadState *Thread)>;
|
||||
using CustomIREntrypointHandler = std::function<void(uintptr_t Entrypoint, IR::IREmitter *)>;
|
||||
|
||||
using ExitHandler = std::function<void(uint64_t ThreadId, ExitReason)>;
|
||||
|
||||
using AOTIRCodeFileWriterFn = std::function<void(const fextl::string& fileid, const fextl::string& filename)>;
|
||||
using AOTIRLoaderCBFn = std::function<int(const fextl::string&)>;
|
||||
using AOTIRRenamerCBFn = std::function<void(const fextl::string&)>;
|
||||
using AOTIRWriterCBFn = std::function<fextl::unique_ptr<AOTIRWriter>(const fextl::string&)>;
|
||||
|
||||
class Context {
|
||||
public:
|
||||
@@ -243,8 +251,10 @@ namespace FEXCore::Context {
|
||||
FEX_DEFAULT_VISIBILITY virtual void RunThread(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void StopThread(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void DestroyThread(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
#ifndef _WIN32
|
||||
FEX_DEFAULT_VISIBILITY virtual void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) {}
|
||||
FEX_DEFAULT_VISIBILITY virtual void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child) {}
|
||||
#endif
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetSignalDelegator(FEXCore::SignalDelegator *SignalDelegation) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetSyscallHandler(FEXCore::HLE::SyscallHandler *Handler) = 0;
|
||||
|
||||
@@ -255,18 +265,18 @@ namespace FEXCore::Context {
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(const fextl::string& Name) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void UnloadAOTIRCacheEntry(FEXCore::IR::AOTIRCacheEntry *Entry) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetAOTIRLoader(std::function<int(const fextl::string&)> CacheReader) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetAOTIRWriter(std::function<fextl::unique_ptr<AOTIRWriter>(const fextl::string&)> CacheWriter) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetAOTIRRenamer(std::function<void(const fextl::string&)> CacheRenamer) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetAOTIRLoader(AOTIRLoaderCBFn CacheReader) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetAOTIRWriter(AOTIRWriterCBFn CacheWriter) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetAOTIRRenamer(AOTIRRenamerCBFn CacheRenamer) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void FinalizeAOTIRCache() = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void WriteFilesWithCode(std::function<void(const fextl::string& fileid, const fextl::string& filename)> Writer) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void WriteFilesWithCode(AOTIRCodeFileWriterFn Writer) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, std::function<void(uint64_t start, uint64_t Length)> callback) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, CodeRangeInvalidationFn callback) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void MarkMemoryShared() = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, fextl::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)> Handler, void *Creator = nullptr, void *Data = nullptr) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void *Creator = nullptr, void *Data = nullptr) = 0;
|
||||
|
||||
/**
|
||||
* @brief Allows the frontend to register its own thunk handlers independent of what is controlled in the backend.
|
||||
|
||||
@@ -29,6 +29,7 @@ class HostFeatures final {
|
||||
bool SupportsBMI2{};
|
||||
bool SupportsCLWB{};
|
||||
bool SupportsPMULL_128Bit{};
|
||||
bool SupportsCSSC{};
|
||||
|
||||
// Float exception behaviour
|
||||
bool SupportsFlushInputsToZero{};
|
||||
|
||||
+3
-11
@@ -45,8 +45,8 @@ namespace Core {
|
||||
*
|
||||
* Required to know which thread has received the signal when it occurs
|
||||
*/
|
||||
void RegisterTLSState(FEXCore::Core::InternalThreadState *Thread);
|
||||
void UninstallTLSState(FEXCore::Core::InternalThreadState *Thread);
|
||||
virtual void RegisterTLSState(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
virtual void UninstallTLSState(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
|
||||
/**
|
||||
* @brief Registers a signal handler for the host to handle a signal
|
||||
@@ -120,17 +120,9 @@ namespace Core {
|
||||
protected:
|
||||
SignalDelegatorConfig Config;
|
||||
|
||||
FEXCore::Core::InternalThreadState *GetTLSThread();
|
||||
virtual FEXCore::Core::InternalThreadState *GetTLSThread() = 0;
|
||||
virtual void HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) = 0;
|
||||
|
||||
/**
|
||||
* @brief Registers an emulated thread's object to a TLS object
|
||||
*
|
||||
* Required to know which thread has received the signal when it occurs
|
||||
*/
|
||||
virtual void RegisterFrontendTLSState(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
virtual void UninstallFrontendTLSState(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
|
||||
/**
|
||||
* @brief Registers a signal handler for the host to handle a signal
|
||||
*
|
||||
|
||||
@@ -56,6 +56,7 @@ enum X86Reg : uint32_t {
|
||||
* @{ */
|
||||
enum X86RegLocation : uint32_t {
|
||||
RFLAG_CF_LOC = 0,
|
||||
RFLAG_RESERVED_LOC = 1, // Reserved Bit, Read-as-1
|
||||
RFLAG_PF_LOC = 2,
|
||||
RFLAG_AF_LOC = 4,
|
||||
RFLAG_ZF_LOC = 6,
|
||||
@@ -73,6 +74,13 @@ enum X86RegLocation : uint32_t {
|
||||
RFLAG_VIP_LOC = 20,
|
||||
RFLAG_ID_LOC = 21,
|
||||
|
||||
// So we can implement arm64-like flag manipulaton on the interpreter/x86 jit..
|
||||
// SF/ZF/CF/OF packed into a 32-bit word, matching arm64's NZCV structure (not semantics).
|
||||
RFLAG_NZCV_LOC = 24,
|
||||
RFLAG_NZCV_1_LOC = 25,
|
||||
RFLAG_NZCV_2_LOC = 26,
|
||||
RFLAG_NZCV_3_LOC = 27,
|
||||
|
||||
// So we can share flag handling logic, we put x87 flags after RFLAGS
|
||||
X87FLAG_BASE = 32,
|
||||
X87FLAG_IE_LOC = 32,
|
||||
|
||||
+19
-6
@@ -93,6 +93,15 @@ friend class FEXCore::IR::PassManager;
|
||||
IRPair<IROp_StoreMemTSO> _StoreMemTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *Addr, OrderedNode *Value, uint8_t Align = 1) {
|
||||
return _StoreMemTSO(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
IRPair<IROp_Lshl> _Lshl(OrderedNode *Src1, OrderedNode *Src2) {
|
||||
return _Lshl(std::max<uint8_t>(4, GetOpSize(Src1)), Src1, Src2);
|
||||
}
|
||||
IRPair<IROp_Lshr> _Lshr(OrderedNode *Src1, OrderedNode *Src2) {
|
||||
return _Lshr(std::max<uint8_t>(4, GetOpSize(Src1)), Src1, Src2);
|
||||
}
|
||||
IRPair<IROp_Ashr> _Ashr(OrderedNode *Src1, OrderedNode *Src2) {
|
||||
return _Ashr(std::max<uint8_t>(4, GetOpSize(Src1)), Src1, Src2);
|
||||
}
|
||||
OrderedNode *Invalid() {
|
||||
return InvalidNode;
|
||||
}
|
||||
@@ -115,7 +124,7 @@ friend class FEXCore::IR::PassManager;
|
||||
|
||||
void SetJumpTarget(IR::IROp_Jump *Op, OrderedNode *Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK,
|
||||
"Tried setting Jump target to %ssa{} {}",
|
||||
"Tried setting Jump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(),
|
||||
IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
|
||||
@@ -123,7 +132,7 @@ friend class FEXCore::IR::PassManager;
|
||||
}
|
||||
void SetTrueJumpTarget(IR::IROp_CondJump *Op, OrderedNode *Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK,
|
||||
"Tried setting CondJump target to %ssa{} {}",
|
||||
"Tried setting CondJump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(),
|
||||
IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
|
||||
@@ -131,7 +140,7 @@ friend class FEXCore::IR::PassManager;
|
||||
}
|
||||
void SetFalseJumpTarget(IR::IROp_CondJump *Op, OrderedNode *Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK,
|
||||
"Tried setting CondJump target to %ssa{} {}",
|
||||
"Tried setting CondJump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(),
|
||||
IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
|
||||
@@ -140,7 +149,7 @@ friend class FEXCore::IR::PassManager;
|
||||
|
||||
void SetJumpTarget(IRPair<IROp_Jump> Op, OrderedNode *Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK,
|
||||
"Tried setting Jump target to %ssa{} {}",
|
||||
"Tried setting Jump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(),
|
||||
IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
|
||||
@@ -148,20 +157,24 @@ friend class FEXCore::IR::PassManager;
|
||||
}
|
||||
void SetTrueJumpTarget(IRPair<IROp_CondJump> Op, OrderedNode *Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK,
|
||||
"Tried setting CondJump target to %ssa{} {}",
|
||||
"Tried setting CondJump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(),
|
||||
IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
Op.first->TrueBlock.NodeOffset = Target->Wrapped(DualListData.ListBegin()).NodeOffset;
|
||||
}
|
||||
void SetFalseJumpTarget(IRPair<IROp_CondJump> Op, OrderedNode *Target) {
|
||||
LOGMAN_THROW_A_FMT(Target->Op(DualListData.DataBegin())->Op == OP_CODEBLOCK,
|
||||
"Tried setting CondJump target to %ssa{} {}",
|
||||
"Tried setting CondJump target to %{} {}",
|
||||
Target->Wrapped(DualListData.ListBegin()).ID(),
|
||||
IR::GetName(Target->Op(DualListData.DataBegin())->Op));
|
||||
Op.first->FalseBlock.NodeOffset = Target->Wrapped(DualListData.ListBegin()).NodeOffset;
|
||||
}
|
||||
|
||||
/** @} */
|
||||
FEXCore::IR::RegisterClassType WalkFindRegClass(OrderedNodeWrapper ssa) {
|
||||
OrderedNode *RealNode = ssa.GetNode(DualListData.ListBegin());
|
||||
return WalkFindRegClass(RealNode);
|
||||
}
|
||||
|
||||
bool IsValueConstant(OrderedNodeWrapper ssa, uint64_t *Constant = nullptr) {
|
||||
OrderedNode *RealNode = ssa.GetNode(DualListData.ListBegin());
|
||||
|
||||
+2
-3
@@ -7,7 +7,6 @@
|
||||
#include <FEXCore/Utils/ThreadPoolAllocator.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <cassert>
|
||||
#include <cstddef>
|
||||
#include <cstring>
|
||||
#include <tuple>
|
||||
@@ -35,7 +34,7 @@ class DualIntrusiveAllocator {
|
||||
}
|
||||
|
||||
[[nodiscard]] void *DataAllocate(size_t Size) {
|
||||
assert(DataCheckSize(Size) &&
|
||||
LOGMAN_THROW_A_FMT(DataCheckSize(Size),
|
||||
"Ran out of space in DualIntrusiveAllocator during allocation");
|
||||
size_t NewOffset = DataCurrentOffset + Size;
|
||||
uintptr_t NewPointer = Data + DataCurrentOffset;
|
||||
@@ -44,7 +43,7 @@ class DualIntrusiveAllocator {
|
||||
}
|
||||
|
||||
[[nodiscard]] void *ListAllocate(size_t Size) {
|
||||
assert(ListCheckSize(Size) &&
|
||||
LOGMAN_THROW_A_FMT(ListCheckSize(Size),
|
||||
"Ran out of space in DualIntrusiveAllocator during allocation");
|
||||
size_t NewOffset = ListCurrentOffset + Size;
|
||||
uintptr_t NewPointer = List + ListCurrentOffset;
|
||||
|
||||
@@ -90,6 +90,9 @@ namespace FEXCore::ArchHelpers::Arm64 {
|
||||
constexpr uint32_t DMB = 0b1101'0101'0000'0011'0011'0000'1011'1111 |
|
||||
0b1011'0000'0000; // Inner shareable all
|
||||
|
||||
constexpr uint32_t DMB_LD = 0b1101'0101'0000'0011'0011'0000'1011'1111 |
|
||||
0b1101'0000'0000; // Inner shareable load
|
||||
|
||||
inline uint32_t GetRdReg(uint32_t Instr) {
|
||||
return (Instr >> RD_OFFSET) & REGISTER_MASK;
|
||||
}
|
||||
|
||||
+107
-17
@@ -7,7 +7,9 @@
|
||||
#include <mutex>
|
||||
#include <shared_mutex>
|
||||
#include <signal.h>
|
||||
#ifndef _WIN32
|
||||
#include <sys/syscall.h>
|
||||
#endif
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEXCore {
|
||||
@@ -93,8 +95,70 @@ namespace FEXCore {
|
||||
};
|
||||
#else
|
||||
// Windows doesn't support forking, so these can be standard mutexes.
|
||||
using ForkableUniqueMutex = std::mutex;
|
||||
using ForkableSharedMutex = std::shared_mutex;
|
||||
class ForkableUniqueMutex final {
|
||||
public:
|
||||
ForkableUniqueMutex() = default;
|
||||
|
||||
// Non-moveable
|
||||
ForkableUniqueMutex(const ForkableUniqueMutex&) = delete;
|
||||
ForkableUniqueMutex& operator=(const ForkableUniqueMutex&) = delete;
|
||||
ForkableUniqueMutex(ForkableUniqueMutex &&rhs) = delete;
|
||||
ForkableUniqueMutex& operator=(ForkableUniqueMutex &&) = delete;
|
||||
|
||||
void lock() {
|
||||
Mutex.lock();
|
||||
}
|
||||
void unlock() {
|
||||
Mutex.unlock();
|
||||
}
|
||||
// Initialize the internal pthread object to its default initializer state.
|
||||
// Should only ever be used in the child process when a Linux fork() has occured.
|
||||
void StealAndDropActiveLocks() {
|
||||
LogMan::Msg::AFmt("{} is unsupported on WIN32 builds!", __func__);
|
||||
}
|
||||
private:
|
||||
std::mutex Mutex;
|
||||
};
|
||||
|
||||
class ForkableSharedMutex final {
|
||||
public:
|
||||
ForkableSharedMutex() = default;
|
||||
|
||||
// Non-moveable
|
||||
ForkableSharedMutex(const ForkableSharedMutex&) = delete;
|
||||
ForkableSharedMutex& operator=(const ForkableSharedMutex&) = delete;
|
||||
ForkableSharedMutex(ForkableSharedMutex &&rhs) = delete;
|
||||
ForkableSharedMutex& operator=(ForkableSharedMutex &&) = delete;
|
||||
|
||||
void lock() {
|
||||
Mutex.lock();
|
||||
}
|
||||
void unlock() {
|
||||
Mutex.unlock();
|
||||
}
|
||||
void lock_shared() {
|
||||
Mutex.lock_shared();
|
||||
}
|
||||
|
||||
void unlock_shared() {
|
||||
Mutex.unlock_shared();
|
||||
}
|
||||
|
||||
bool try_lock() {
|
||||
return Mutex.try_lock();
|
||||
}
|
||||
|
||||
bool try_lock_shared() {
|
||||
return Mutex.try_lock_shared();
|
||||
}
|
||||
// Initialize the internal pthread object to its default initializer state.
|
||||
// Should only ever be used in the child process when a Linux fork() has occured.
|
||||
void StealAndDropActiveLocks() {
|
||||
LogMan::Msg::AFmt("{} is unsupported on WIN32 builds!", __func__);
|
||||
}
|
||||
private:
|
||||
std::shared_mutex Mutex;
|
||||
};
|
||||
#endif
|
||||
|
||||
template<typename MutexType, void (MutexType::*lock_fn)(), void (MutexType::*unlock_fn)()>
|
||||
@@ -165,10 +229,37 @@ namespace FEXCore {
|
||||
&FEXCore::ForkableSharedMutex::lock,
|
||||
&FEXCore::ForkableSharedMutex::unlock>;
|
||||
|
||||
class ScopedSignalMasker final {
|
||||
public:
|
||||
ScopedSignalMasker() = default;
|
||||
|
||||
void Mask(uint64_t Mask) {
|
||||
#ifndef _WIN32
|
||||
// Mask all signals, storing the original incoming mask
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &Mask, &OriginalMask, sizeof(OriginalMask));
|
||||
#endif
|
||||
}
|
||||
|
||||
// Move-only type
|
||||
ScopedSignalMasker(const ScopedSignalMasker&) = delete;
|
||||
ScopedSignalMasker& operator=(ScopedSignalMasker&) = delete;
|
||||
ScopedSignalMasker(ScopedSignalMasker &&rhs) = default;
|
||||
ScopedSignalMasker& operator=(ScopedSignalMasker &&) = default;
|
||||
|
||||
void Unmask() {
|
||||
#ifndef _WIN32
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &OriginalMask, nullptr, sizeof(OriginalMask));
|
||||
#endif
|
||||
}
|
||||
private:
|
||||
#ifndef _WIN32
|
||||
uint64_t OriginalMask{};
|
||||
#endif
|
||||
};
|
||||
|
||||
template<typename MutexType, void (MutexType::*lock_fn)(), void (MutexType::*unlock_fn)()>
|
||||
class ScopedPotentialDeferredSignalWithMutexBase final {
|
||||
public:
|
||||
|
||||
ScopedPotentialDeferredSignalWithMutexBase(MutexType &_Mutex, FEXCore::Core::InternalThreadState *Thread, uint64_t Mask = ~0ULL)
|
||||
: Mutex {&_Mutex}
|
||||
, Thread {Thread} {
|
||||
@@ -176,8 +267,7 @@ namespace FEXCore {
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Increment(1);
|
||||
}
|
||||
else {
|
||||
// Mask all signals, storing the original incoming mask
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &Mask, &OriginalMask, sizeof(OriginalMask));
|
||||
Masker.Mask(Mask);
|
||||
}
|
||||
// Lock the mutex
|
||||
(Mutex->*lock_fn)();
|
||||
@@ -201,29 +291,29 @@ namespace FEXCore {
|
||||
|
||||
if (Thread) {
|
||||
#ifdef _M_X86_64
|
||||
// Needs to be atomic so that operations can't end up getting reordered around this.
|
||||
// Without this, the refcount and the signal access could get reordered.
|
||||
auto Result = Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
|
||||
// Needs to be atomic so that operations can't end up getting reordered around this.
|
||||
// Without this, the refcount and the signal access could get reordered.
|
||||
auto Result = Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
|
||||
|
||||
// X86-64 must do an additional check around the store.
|
||||
if ((Result - 1) == 0) {
|
||||
// Must happen after the refcount store
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
|
||||
}
|
||||
// X86-64 must do an additional check around the store.
|
||||
if ((Result - 1) == 0) {
|
||||
// Must happen after the refcount store
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
|
||||
}
|
||||
#else
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1);
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress->Store(0);
|
||||
#endif
|
||||
}
|
||||
else {
|
||||
// Unmask back to the original signal mask
|
||||
::syscall(SYS_rt_sigprocmask, SIG_SETMASK, &OriginalMask, nullptr, sizeof(OriginalMask));
|
||||
Masker.Unmask();
|
||||
}
|
||||
}
|
||||
}
|
||||
private:
|
||||
MutexType *Mutex;
|
||||
uint64_t OriginalMask{};
|
||||
ScopedSignalMasker Masker;
|
||||
FEXCore::Core::InternalThreadState *Thread;
|
||||
};
|
||||
|
||||
|
||||
@@ -1,6 +1,10 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <bit>
|
||||
#include <cstdint>
|
||||
#include <type_traits>
|
||||
|
||||
namespace FEXCore {
|
||||
[[nodiscard]] constexpr uint64_t AlignUp(uint64_t value, uint64_t size) {
|
||||
@@ -10,4 +14,14 @@ namespace FEXCore {
|
||||
[[nodiscard]] constexpr uint64_t AlignDown(uint64_t value, uint64_t size) {
|
||||
return value - value % size;
|
||||
}
|
||||
|
||||
// Returns the ilog2 of a power-of-2 integer.
|
||||
// Asserts in the case that the passed in integer is not a power-of-2.
|
||||
template<typename T>
|
||||
requires(std::is_unsigned_v<T>)
|
||||
[[nodiscard]] constexpr T ilog2(T Value) {
|
||||
LOGMAN_THROW_A_FMT(std::has_single_bit(Value), "ilog2 requires popcount to be one");
|
||||
return std::countr_zero(Value);
|
||||
}
|
||||
|
||||
} // namespace FEXCore
|
||||
+3
-3
@@ -41,7 +41,7 @@ namespace FEXCore::Telemetry {
|
||||
TYPE_LAST,
|
||||
};
|
||||
|
||||
Value &GetObject(TelemetryType Type);
|
||||
FEX_DEFAULT_VISIBILITY Value &GetTelemetryValue(TelemetryType Type);
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void Initialize();
|
||||
FEX_DEFAULT_VISIBILITY void Shutdown(fextl::string const &ApplicationName);
|
||||
@@ -50,8 +50,8 @@ namespace FEXCore::Telemetry {
|
||||
// This returns the internal structure to the telemetry data structures
|
||||
// One must be careful with placing these in the hot path of code execution
|
||||
// It can be fairly costly, especially in the static version where it puts barriers in the code
|
||||
#define FEXCORE_TELEMETRY_STATIC_INIT(Name, Type) static FEXCore::Telemetry::Value &Name = FEXCore::Telemetry::GetObject(FEXCore::Telemetry::Type)
|
||||
#define FEXCORE_TELEMETRY_INIT(Name, Type) FEXCore::Telemetry::Value &Name = FEXCore::Telemetry::GetObject(FEXCore::Telemetry::Type)
|
||||
#define FEXCORE_TELEMETRY_STATIC_INIT(Name, Type) static FEXCore::Telemetry::Value &Name = FEXCore::Telemetry::GetTelemetryValue(FEXCore::Telemetry::Type)
|
||||
#define FEXCORE_TELEMETRY_INIT(Name, Type) FEXCore::Telemetry::Value &Name = FEXCore::Telemetry::GetTelemetryValue(FEXCore::Telemetry::Type)
|
||||
// Telemetry ALU operations
|
||||
// These are typically 3-4 instructions depending on what you're doing
|
||||
#define FEXCORE_TELEMETRY_SET(Name, Value) Name = Value
|
||||
|
||||
+1
-1
@@ -58,7 +58,7 @@ namespace fextl::fmt {
|
||||
FMT_INLINE auto print(::fmt::format_string<T...> fmt, T&&... args)
|
||||
-> void {
|
||||
auto String = fextl::fmt::vformat(fmt, ::fmt::make_format_args(args...));
|
||||
auto f = fextl::file::File::GetStdOUT();
|
||||
auto f = FEXCore::File::File::GetStdOUT();
|
||||
f.Write(String.c_str(), String.size());
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,21 @@
|
||||
file(GLOB_RECURSE TESTS CONFIGURE_DEPENDS *.cpp)
|
||||
|
||||
set (LIBS fmt::fmt vixl Catch2::Catch2WithMain FEXCore_Base)
|
||||
foreach(TEST ${TESTS})
|
||||
get_filename_component(TEST_NAME ${TEST} NAME_WLE)
|
||||
add_executable(FEXCore_Tests_${TEST_NAME} ${TEST})
|
||||
target_link_libraries(FEXCore_Tests_${TEST_NAME} PRIVATE ${LIBS})
|
||||
target_include_directories(FEXCore_Tests_${TEST_NAME} PUBLIC "${CMAKE_CURRENT_SOURCE_DIR}/../../Source/")
|
||||
set_target_properties(FEXCore_Tests_${TEST_NAME} PROPERTIES RUNTIME_OUTPUT_DIRECTORY "${CMAKE_BINARY_DIR}/FEXCore_Tests")
|
||||
catch_discover_tests(FEXCore_Tests_${TEST_NAME} TEST_SUFFIX ".${TEST_NAME}.FEXCore_Tests")
|
||||
endforeach()
|
||||
|
||||
execute_process(COMMAND "nproc" OUTPUT_VARIABLE CORES)
|
||||
string(STRIP ${CORES} CORES)
|
||||
|
||||
add_custom_target(
|
||||
fexcore_apitests
|
||||
WORKING_DIRECTORY "${CMAKE_BINARY_DIR}/"
|
||||
USES_TERMINAL
|
||||
COMMAND "ctest" "--timeout" "302" "-j${CORES}" "-R" "\.*.FEXCore_Tests$$")
|
||||
|
||||
@@ -0,0 +1,7 @@
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <catch2/catch.hpp>
|
||||
|
||||
TEST_CASE("ILog2") {
|
||||
auto i = GENERATE(range(0, 64));
|
||||
REQUIRE(FEXCore::ilog2(1ull << i) == i);
|
||||
}
|
||||
+1
@@ -1,3 +1,4 @@
|
||||
if (NOT MINGW_BUILD)
|
||||
add_subdirectory(Emitter/)
|
||||
add_subdirectory(APITests/)
|
||||
endif()
|
||||
+210
-1
@@ -259,6 +259,18 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Add/subtract immediate") {
|
||||
TEST_SINGLE(adds(Size::i64Bit, Reg::r29, Reg::r28, 4095, true), "adds x29, x28, #0xfff000 (16773120)");
|
||||
TEST_SINGLE(adds(Size::i64Bit, Reg::r29, Reg::r28, 16773120), "adds x29, x28, #0xfff000 (16773120)");
|
||||
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, 0, false), "cmn w28, #0x0 (0)");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, 4095, false), "cmn w28, #0xfff (4095)");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, 0, true), "cmn w28, #0x0 (0)");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, 4095, true), "cmn w28, #0xfff000 (16773120)");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, 16773120), "cmn w28, #0xfff000 (16773120)");
|
||||
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, 0, false), "cmn x28, #0x0 (0)");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, 4095, false), "cmn x28, #0xfff (4095)");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, 0, true), "cmn x28, #0x0 (0)");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, 4095, true), "cmn x28, #0xfff000 (16773120)");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, 16773120), "cmn x28, #0xfff000 (16773120)");
|
||||
|
||||
TEST_SINGLE(sub(Size::i32Bit, Reg::r29, Reg::r28, 0, false), "sub w29, w28, #0x0 (0)");
|
||||
TEST_SINGLE(sub(Size::i32Bit, Reg::r29, Reg::r28, 4095, false), "sub w29, w28, #0xfff (4095)");
|
||||
TEST_SINGLE(sub(Size::i32Bit, Reg::r29, Reg::r28, 0, true), "sub w29, w28, #0x0 (0)");
|
||||
@@ -358,6 +370,11 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Logical immediate") {
|
||||
TEST_SINGLE(eor(Size::i32Bit, Reg::r29, Reg::r28, -2), "eor w29, w28, #0xfffffffe");
|
||||
TEST_SINGLE(eor(Size::i64Bit, Reg::r29, Reg::r28, 1), "eor x29, x28, #0x1");
|
||||
TEST_SINGLE(eor(Size::i64Bit, Reg::r29, Reg::r28, -2), "eor x29, x28, #0xfffffffffffffffe");
|
||||
|
||||
TEST_SINGLE(tst(Size::i32Bit, Reg::r28, 1), "tst w28, #0x1");
|
||||
TEST_SINGLE(tst(Size::i32Bit, Reg::r28, -2), "tst w28, #0xfffffffe");
|
||||
TEST_SINGLE(tst(Size::i64Bit, Reg::r28, 1), "tst x28, #0x1");
|
||||
TEST_SINGLE(tst(Size::i64Bit, Reg::r28, -2), "tst x28, #0xfffffffffffffffe");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Move wide immediate") {
|
||||
TEST_SINGLE(movn(Size::i32Bit, Reg::r29, 0x4243, 0), "mov w29, #0xffffbdbc");
|
||||
@@ -424,6 +441,30 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Bitfield") {
|
||||
|
||||
TEST_SINGLE(asr(Size::i32Bit, Reg::r29, Reg::r28, 17), "asr w29, w28, #17");
|
||||
TEST_SINGLE(asr(Size::i64Bit, Reg::r29, Reg::r28, 17), "asr x29, x28, #17");
|
||||
|
||||
TEST_SINGLE(bfc(Size::i32Bit, Reg::r29, 4, 3), "bfc w29, #4, #3");
|
||||
TEST_SINGLE(bfc(Size::i32Bit, Reg::r29, 27, 3), "bfc w29, #27, #3");
|
||||
|
||||
TEST_SINGLE(bfc(Size::i64Bit, Reg::r29, 4, 3), "bfc x29, #4, #3");
|
||||
TEST_SINGLE(bfc(Size::i64Bit, Reg::r29, 57, 3), "bfc x29, #57, #3");
|
||||
|
||||
TEST_SINGLE(bfxil(Size::i32Bit, Reg::r29, Reg::r28, 4, 3), "bfxil w29, w28, #4, #3");
|
||||
TEST_SINGLE(bfxil(Size::i32Bit, Reg::r29, Reg::r28, 27, 3), "bfxil w29, w28, #27, #3");
|
||||
|
||||
TEST_SINGLE(bfxil(Size::i64Bit, Reg::r29, Reg::r28, 4, 3), "bfxil x29, x28, #4, #3");
|
||||
TEST_SINGLE(bfxil(Size::i64Bit, Reg::r29, Reg::r28, 57, 3), "bfxil x29, x28, #57, #3");
|
||||
|
||||
TEST_SINGLE(sbfiz(Size::i32Bit, Reg::r29, Reg::r28, 5, 3), "sbfiz w29, w28, #5, #3");
|
||||
TEST_SINGLE(sbfiz(Size::i32Bit, Reg::r29, Reg::r28, 27, 3), "sbfiz w29, w28, #27, #3");
|
||||
|
||||
TEST_SINGLE(sbfiz(Size::i64Bit, Reg::r29, Reg::r28, 5, 3), "sbfiz x29, x28, #5, #3");
|
||||
TEST_SINGLE(sbfiz(Size::i64Bit, Reg::r29, Reg::r28, 54, 3), "sbfiz x29, x28, #54, #3");
|
||||
|
||||
TEST_SINGLE(ubfiz(Size::i32Bit, Reg::r29, Reg::r28, 5, 3), "ubfiz w29, w28, #5, #3");
|
||||
TEST_SINGLE(ubfiz(Size::i32Bit, Reg::r29, Reg::r28, 27, 3), "ubfiz w29, w28, #27, #3");
|
||||
|
||||
TEST_SINGLE(ubfiz(Size::i64Bit, Reg::r29, Reg::r28, 5, 3), "ubfiz x29, x28, #5, #3");
|
||||
TEST_SINGLE(ubfiz(Size::i64Bit, Reg::r29, Reg::r28, 54, 3), "ubfiz x29, x28, #54, #3");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Extract") {
|
||||
TEST_SINGLE(extr(Size::i32Bit, Reg::r29, Reg::r28, Reg::r27, 0), "extr w29, w28, w27, #0");
|
||||
@@ -813,6 +854,38 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Logical - shifted register") {
|
||||
TEST_SINGLE(eon(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27, ShiftType::ROR, 0), "eon x29, x28, x27");
|
||||
TEST_SINGLE(eon(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27, ShiftType::ROR, 1), "eon x29, x28, x27, ror #1");
|
||||
TEST_SINGLE(eon(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27, ShiftType::ROR, 63), "eon x29, x28, x27, ror #63");
|
||||
|
||||
TEST_SINGLE(tst(Size::i32Bit, Reg::r28, Reg::r27, ShiftType::LSL, 0), "tst w28, w27");
|
||||
TEST_SINGLE(tst(Size::i32Bit, Reg::r28, Reg::r27, ShiftType::LSL, 1), "tst w28, w27, lsl #1");
|
||||
TEST_SINGLE(tst(Size::i32Bit, Reg::r28, Reg::r27, ShiftType::LSL, 31), "tst w28, w27, lsl #31");
|
||||
|
||||
TEST_SINGLE(tst(Size::i32Bit, Reg::r28, Reg::r27, ShiftType::LSR, 0), "tst w28, w27");
|
||||
TEST_SINGLE(tst(Size::i32Bit, Reg::r28, Reg::r27, ShiftType::LSR, 1), "tst w28, w27, lsr #1");
|
||||
TEST_SINGLE(tst(Size::i32Bit, Reg::r28, Reg::r27, ShiftType::LSR, 31), "tst w28, w27, lsr #31");
|
||||
|
||||
TEST_SINGLE(tst(Size::i32Bit, Reg::r28, Reg::r27, ShiftType::ASR, 0), "tst w28, w27");
|
||||
TEST_SINGLE(tst(Size::i32Bit, Reg::r28, Reg::r27, ShiftType::ASR, 1), "tst w28, w27, asr #1");
|
||||
TEST_SINGLE(tst(Size::i32Bit, Reg::r28, Reg::r27, ShiftType::ASR, 31), "tst w28, w27, asr #31");
|
||||
|
||||
TEST_SINGLE(tst(Size::i32Bit, Reg::r28, Reg::r27, ShiftType::ROR, 0), "tst w28, w27");
|
||||
TEST_SINGLE(tst(Size::i32Bit, Reg::r28, Reg::r27, ShiftType::ROR, 1), "tst w28, w27, ror #1");
|
||||
TEST_SINGLE(tst(Size::i32Bit, Reg::r28, Reg::r27, ShiftType::ROR, 31), "tst w28, w27, ror #31");
|
||||
|
||||
TEST_SINGLE(tst(Size::i64Bit, Reg::r28, Reg::r27, ShiftType::LSL, 0), "tst x28, x27");
|
||||
TEST_SINGLE(tst(Size::i64Bit, Reg::r28, Reg::r27, ShiftType::LSL, 1), "tst x28, x27, lsl #1");
|
||||
TEST_SINGLE(tst(Size::i64Bit, Reg::r28, Reg::r27, ShiftType::LSL, 63), "tst x28, x27, lsl #63");
|
||||
|
||||
TEST_SINGLE(tst(Size::i64Bit, Reg::r28, Reg::r27, ShiftType::LSR, 0), "tst x28, x27");
|
||||
TEST_SINGLE(tst(Size::i64Bit, Reg::r28, Reg::r27, ShiftType::LSR, 1), "tst x28, x27, lsr #1");
|
||||
TEST_SINGLE(tst(Size::i64Bit, Reg::r28, Reg::r27, ShiftType::LSR, 63), "tst x28, x27, lsr #63");
|
||||
|
||||
TEST_SINGLE(tst(Size::i64Bit, Reg::r28, Reg::r27, ShiftType::ASR, 0), "tst x28, x27");
|
||||
TEST_SINGLE(tst(Size::i64Bit, Reg::r28, Reg::r27, ShiftType::ASR, 1), "tst x28, x27, asr #1");
|
||||
TEST_SINGLE(tst(Size::i64Bit, Reg::r28, Reg::r27, ShiftType::ASR, 63), "tst x28, x27, asr #63");
|
||||
|
||||
TEST_SINGLE(tst(Size::i64Bit, Reg::r28, Reg::r27, ShiftType::ROR, 0), "tst x28, x27");
|
||||
TEST_SINGLE(tst(Size::i64Bit, Reg::r28, Reg::r27, ShiftType::ROR, 1), "tst x28, x27, ror #1");
|
||||
TEST_SINGLE(tst(Size::i64Bit, Reg::r28, Reg::r27, ShiftType::ROR, 63), "tst x28, x27, ror #63");
|
||||
}
|
||||
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: AddSub - shifted register") {
|
||||
@@ -868,6 +941,32 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: AddSub - shifted register") {
|
||||
// Unsupported
|
||||
}
|
||||
|
||||
{
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r29, Reg::r28), "cmn x29, x28");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r29, Reg::r28), "cmn w29, w28");
|
||||
|
||||
// LSL
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r29, Reg::r28, ShiftType::LSL, 1), "cmn x29, x28, lsl #1");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r29, Reg::r28, ShiftType::LSL, 1), "cmn w29, w28, lsl #1");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r29, Reg::r28, ShiftType::LSL, 63), "cmn x29, x28, lsl #63");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r29, Reg::r28, ShiftType::LSL, 31), "cmn w29, w28, lsl #31");
|
||||
|
||||
// LSR
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r29, Reg::r28, ShiftType::LSR, 1), "cmn x29, x28, lsr #1");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r29, Reg::r28, ShiftType::LSR, 1), "cmn w29, w28, lsr #1");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r29, Reg::r28, ShiftType::LSR, 63), "cmn x29, x28, lsr #63");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r29, Reg::r28, ShiftType::LSR, 31), "cmn w29, w28, lsr #31");
|
||||
|
||||
// ASR
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r29, Reg::r28, ShiftType::ASR, 1), "cmn x29, x28, asr #1");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r29, Reg::r28, ShiftType::ASR, 1), "cmn w29, w28, asr #1");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r29, Reg::r28, ShiftType::ASR, 63), "cmn x29, x28, asr #63");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r29, Reg::r28, ShiftType::ASR, 31), "cmn w29, w28, asr #31");
|
||||
|
||||
// ROR
|
||||
// Unsupported
|
||||
}
|
||||
|
||||
// FEX had a bug with this
|
||||
TEST_SINGLE(sub(Size::i64Bit, Reg::rsp, Reg::rsp, Reg::r0, ShiftType::LSL, 0), "neg xzr, x0");
|
||||
|
||||
@@ -1099,7 +1198,6 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: AddSub - extended register") {
|
||||
TEST_SINGLE(add(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27, ExtendedType::SXTX, 3), "add x29, x28, x27, sxtx #3");
|
||||
TEST_SINGLE(add(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27, ExtendedType::SXTX, 4), "add x29, x28, x27, sxtx #4");
|
||||
|
||||
|
||||
TEST_SINGLE(adds(Size::i32Bit, Reg::r29, Reg::r28, Reg::r27, ExtendedType::UXTB, 0), "adds w29, w28, w27, uxtb");
|
||||
TEST_SINGLE(adds(Size::i32Bit, Reg::r29, Reg::r28, Reg::r27, ExtendedType::UXTB, 1), "adds w29, w28, w27, uxtb #1");
|
||||
TEST_SINGLE(adds(Size::i32Bit, Reg::r29, Reg::r28, Reg::r27, ExtendedType::UXTB, 2), "adds w29, w28, w27, uxtb #2");
|
||||
@@ -1196,6 +1294,102 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: AddSub - extended register") {
|
||||
TEST_SINGLE(adds(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27, ExtendedType::SXTX, 3), "adds x29, x28, x27, sxtx #3");
|
||||
TEST_SINGLE(adds(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27, ExtendedType::SXTX, 4), "adds x29, x28, x27, sxtx #4");
|
||||
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::UXTB, 0), "cmn w28, w27, uxtb");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::UXTB, 1), "cmn w28, w27, uxtb #1");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::UXTB, 2), "cmn w28, w27, uxtb #2");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::UXTB, 3), "cmn w28, w27, uxtb #3");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::UXTB, 4), "cmn w28, w27, uxtb #4");
|
||||
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::UXTH, 0), "cmn w28, w27, uxth");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::UXTH, 1), "cmn w28, w27, uxth #1");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::UXTH, 2), "cmn w28, w27, uxth #2");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::UXTH, 3), "cmn w28, w27, uxth #3");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::UXTH, 4), "cmn w28, w27, uxth #4");
|
||||
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::LSL_32, 0), "cmn w28, w27");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::LSL_32, 1), "cmn w28, w27, lsl #1");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::LSL_32, 2), "cmn w28, w27, lsl #2");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::LSL_32, 3), "cmn w28, w27, lsl #3");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::LSL_32, 4), "cmn w28, w27, lsl #4");
|
||||
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::LSL_64, 0), "cmn w28, x27");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::LSL_64, 1), "cmn w28, x27, lsl #1");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::LSL_64, 2), "cmn w28, x27, lsl #2");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::LSL_64, 3), "cmn w28, x27, lsl #3");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::LSL_64, 4), "cmn w28, x27, lsl #4");
|
||||
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::SXTB, 0), "cmn w28, w27, sxtb");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::SXTB, 1), "cmn w28, w27, sxtb #1");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::SXTB, 2), "cmn w28, w27, sxtb #2");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::SXTB, 3), "cmn w28, w27, sxtb #3");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::SXTB, 4), "cmn w28, w27, sxtb #4");
|
||||
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::SXTH, 0), "cmn w28, w27, sxth");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::SXTH, 1), "cmn w28, w27, sxth #1");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::SXTH, 2), "cmn w28, w27, sxth #2");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::SXTH, 3), "cmn w28, w27, sxth #3");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::SXTH, 4), "cmn w28, w27, sxth #4");
|
||||
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::SXTW, 0), "cmn w28, w27, sxtw");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::SXTW, 1), "cmn w28, w27, sxtw #1");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::SXTW, 2), "cmn w28, w27, sxtw #2");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::SXTW, 3), "cmn w28, w27, sxtw #3");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::SXTW, 4), "cmn w28, w27, sxtw #4");
|
||||
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::SXTX, 0), "cmn w28, x27, sxtx");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::SXTX, 1), "cmn w28, x27, sxtx #1");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::SXTX, 2), "cmn w28, x27, sxtx #2");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::SXTX, 3), "cmn w28, x27, sxtx #3");
|
||||
TEST_SINGLE(cmn(Size::i32Bit, Reg::r28, Reg::r27, ExtendedType::SXTX, 4), "cmn w28, x27, sxtx #4");
|
||||
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::UXTB, 0), "cmn x28, w27, uxtb");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::UXTB, 1), "cmn x28, w27, uxtb #1");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::UXTB, 2), "cmn x28, w27, uxtb #2");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::UXTB, 3), "cmn x28, w27, uxtb #3");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::UXTB, 4), "cmn x28, w27, uxtb #4");
|
||||
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::UXTH, 0), "cmn x28, w27, uxth");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::UXTH, 1), "cmn x28, w27, uxth #1");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::UXTH, 2), "cmn x28, w27, uxth #2");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::UXTH, 3), "cmn x28, w27, uxth #3");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::UXTH, 4), "cmn x28, w27, uxth #4");
|
||||
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::UXTW, 0), "cmn x28, w27, uxtw");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::UXTW, 1), "cmn x28, w27, uxtw #1");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::UXTW, 2), "cmn x28, w27, uxtw #2");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::UXTW, 3), "cmn x28, w27, uxtw #3");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::UXTW, 4), "cmn x28, w27, uxtw #4");
|
||||
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::LSL_64, 0), "cmn x28, x27");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::LSL_64, 1), "cmn x28, x27, lsl #1");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::LSL_64, 2), "cmn x28, x27, lsl #2");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::LSL_64, 3), "cmn x28, x27, lsl #3");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::LSL_64, 4), "cmn x28, x27, lsl #4");
|
||||
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::SXTB, 0), "cmn x28, w27, sxtb");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::SXTB, 1), "cmn x28, w27, sxtb #1");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::SXTB, 2), "cmn x28, w27, sxtb #2");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::SXTB, 3), "cmn x28, w27, sxtb #3");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::SXTB, 4), "cmn x28, w27, sxtb #4");
|
||||
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::SXTH, 0), "cmn x28, w27, sxth");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::SXTH, 1), "cmn x28, w27, sxth #1");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::SXTH, 2), "cmn x28, w27, sxth #2");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::SXTH, 3), "cmn x28, w27, sxth #3");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::SXTH, 4), "cmn x28, w27, sxth #4");
|
||||
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::SXTW, 0), "cmn x28, w27, sxtw");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::SXTW, 1), "cmn x28, w27, sxtw #1");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::SXTW, 2), "cmn x28, w27, sxtw #2");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::SXTW, 3), "cmn x28, w27, sxtw #3");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::SXTW, 4), "cmn x28, w27, sxtw #4");
|
||||
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::SXTX, 0), "cmn x28, x27, sxtx");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::SXTX, 1), "cmn x28, x27, sxtx #1");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::SXTX, 2), "cmn x28, x27, sxtx #2");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::SXTX, 3), "cmn x28, x27, sxtx #3");
|
||||
TEST_SINGLE(cmn(Size::i64Bit, Reg::r28, Reg::r27, ExtendedType::SXTX, 4), "cmn x28, x27, sxtx #4");
|
||||
|
||||
TEST_SINGLE(sub(Size::i32Bit, Reg::r29, Reg::r28, Reg::r27, ExtendedType::UXTB, 0), "sub w29, w28, w27, uxtb");
|
||||
TEST_SINGLE(sub(Size::i32Bit, Reg::r29, Reg::r28, Reg::r27, ExtendedType::UXTB, 1), "sub w29, w28, w27, uxtb #1");
|
||||
TEST_SINGLE(sub(Size::i32Bit, Reg::r29, Reg::r28, Reg::r27, ExtendedType::UXTB, 2), "sub w29, w28, w27, uxtb #2");
|
||||
@@ -1496,6 +1690,12 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: AddSub - with carry") {
|
||||
|
||||
TEST_SINGLE(sbcs(Size::i32Bit, Reg::r29, Reg::r28, Reg::r27), "sbcs w29, w28, w27");
|
||||
TEST_SINGLE(sbcs(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27), "sbcs x29, x28, x27");
|
||||
|
||||
TEST_SINGLE(ngc(Size::i32Bit, Reg::r29, Reg::r27), "ngc w29, w27");
|
||||
TEST_SINGLE(ngc(Size::i64Bit, Reg::r29, Reg::r27), "ngc x29, x27");
|
||||
|
||||
TEST_SINGLE(ngcs(Size::i32Bit, Reg::r29, Reg::r27), "ngcs w29, w27");
|
||||
TEST_SINGLE(ngcs(Size::i64Bit, Reg::r29, Reg::r27), "ngcs x29, x27");
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Rotate right into flags") {
|
||||
TEST_SINGLE(rmif(XReg::x30, 63, 0b0000), "rmif x30, #63, #nzcv");
|
||||
@@ -1690,6 +1890,15 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: ALU: Conditional select") {
|
||||
TEST_SINGLE(csinv(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27, Condition::CC_EQ), "csinv x29, x28, x27, eq");
|
||||
TEST_SINGLE(csneg(Size::i32Bit, Reg::r29, Reg::r28, Reg::r27, Condition::CC_EQ), "csneg w29, w28, w27, eq");
|
||||
TEST_SINGLE(csneg(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27, Condition::CC_EQ), "csneg x29, x28, x27, eq");
|
||||
TEST_SINGLE(cneg(Size::i32Bit, Reg::r29, Reg::r28, Condition::CC_EQ), "cneg w29, w28, eq");
|
||||
TEST_SINGLE(cneg(Size::i64Bit, Reg::r29, Reg::r28, Condition::CC_EQ), "cneg x29, x28, eq");
|
||||
|
||||
TEST_SINGLE(cinc(Size::i32Bit, Reg::r29, Reg::r28, Condition::CC_EQ), "cinc w29, w28, eq");
|
||||
TEST_SINGLE(cinc(Size::i64Bit, Reg::r29, Reg::r28, Condition::CC_EQ), "cinc x29, x28, eq");
|
||||
TEST_SINGLE(cinv(Size::i32Bit, Reg::r29, Reg::r28, Condition::CC_EQ), "cinv w29, w28, eq");
|
||||
TEST_SINGLE(cinv(Size::i64Bit, Reg::r29, Reg::r28, Condition::CC_EQ), "cinv x29, x28, eq");
|
||||
TEST_SINGLE(csetm(Size::i32Bit, Reg::r29, Condition::CC_EQ), "csetm w29, eq");
|
||||
TEST_SINGLE(csetm(Size::i64Bit, Reg::r29, Condition::CC_EQ), "csetm x29, eq");
|
||||
|
||||
TEST_SINGLE(csel(Size::i32Bit, Reg::r29, Reg::r28, Reg::r27, Condition::CC_AL), "csel w29, w28, w27, al");
|
||||
TEST_SINGLE(csel(Size::i64Bit, Reg::r29, Reg::r28, Reg::r27, Condition::CC_AL), "csel x29, x28, x27, al");
|
||||
|
||||
+128
-8
@@ -1574,46 +1574,166 @@ TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Loadstore register unpri
|
||||
}
|
||||
}
|
||||
TEST_CASE_METHOD(TestDisassembler, "Emitter: Loadstore: Atomic memory operations") {
|
||||
TEST_SINGLE(stadd(SubRegSize::i8Bit, Reg::r30, Reg::r29), "staddb w30, [x29]");
|
||||
TEST_SINGLE(stadd(SubRegSize::i8Bit, Reg::r30, Reg::r29), "staddb w30, [x29]");
|
||||
TEST_SINGLE(stadd(SubRegSize::i16Bit, Reg::r30, Reg::r29), "staddh w30, [x29]");
|
||||
TEST_SINGLE(stadd(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stadd w30, [x29]");
|
||||
TEST_SINGLE(stadd(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stadd x30, [x29]");
|
||||
|
||||
TEST_SINGLE(staddl(SubRegSize::i8Bit, Reg::r30, Reg::r29), "staddlb w30, [x29]");
|
||||
TEST_SINGLE(staddl(SubRegSize::i8Bit, Reg::r30, Reg::r29), "staddlb w30, [x29]");
|
||||
TEST_SINGLE(staddl(SubRegSize::i16Bit, Reg::r30, Reg::r29), "staddlh w30, [x29]");
|
||||
TEST_SINGLE(staddl(SubRegSize::i32Bit, Reg::r30, Reg::r29), "staddl w30, [x29]");
|
||||
TEST_SINGLE(staddl(SubRegSize::i64Bit, Reg::r30, Reg::r29), "staddl x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stclr(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stclrb w30, [x29]");
|
||||
TEST_SINGLE(stadda(SubRegSize::i8Bit, Reg::r30, Reg::r29), "staddab w30, [x29]");
|
||||
TEST_SINGLE(stadda(SubRegSize::i16Bit, Reg::r30, Reg::r29), "staddah w30, [x29]");
|
||||
TEST_SINGLE(stadda(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stadda w30, [x29]");
|
||||
TEST_SINGLE(stadda(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stadda x30, [x29]");
|
||||
|
||||
TEST_SINGLE(staddal(SubRegSize::i8Bit, Reg::r30, Reg::r29), "staddalb w30, [x29]");
|
||||
TEST_SINGLE(staddal(SubRegSize::i16Bit, Reg::r30, Reg::r29), "staddalh w30, [x29]");
|
||||
TEST_SINGLE(staddal(SubRegSize::i32Bit, Reg::r30, Reg::r29), "staddal w30, [x29]");
|
||||
TEST_SINGLE(staddal(SubRegSize::i64Bit, Reg::r30, Reg::r29), "staddal x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stclr(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stclrb w30, [x29]");
|
||||
TEST_SINGLE(stclr(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stclrh w30, [x29]");
|
||||
TEST_SINGLE(stclr(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stclr w30, [x29]");
|
||||
TEST_SINGLE(stclr(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stclr x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stclrl(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stclrlb w30, [x29]");
|
||||
TEST_SINGLE(stclrl(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stclrlb w30, [x29]");
|
||||
TEST_SINGLE(stclrl(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stclrlh w30, [x29]");
|
||||
TEST_SINGLE(stclrl(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stclrl w30, [x29]");
|
||||
TEST_SINGLE(stclrl(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stclrl x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stset(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stsetb w30, [x29]");
|
||||
TEST_SINGLE(stclra(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stclrab w30, [x29]");
|
||||
TEST_SINGLE(stclra(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stclrah w30, [x29]");
|
||||
TEST_SINGLE(stclra(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stclra w30, [x29]");
|
||||
TEST_SINGLE(stclra(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stclra x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stclral(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stclralb w30, [x29]");
|
||||
TEST_SINGLE(stclral(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stclralh w30, [x29]");
|
||||
TEST_SINGLE(stclral(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stclral w30, [x29]");
|
||||
TEST_SINGLE(stclral(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stclral x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stset(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stsetb w30, [x29]");
|
||||
TEST_SINGLE(stset(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stseth w30, [x29]");
|
||||
TEST_SINGLE(stset(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stset w30, [x29]");
|
||||
TEST_SINGLE(stset(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stset x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stsetl(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stsetlb w30, [x29]");
|
||||
TEST_SINGLE(stsetl(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stsetlb w30, [x29]");
|
||||
TEST_SINGLE(stsetl(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stsetlh w30, [x29]");
|
||||
TEST_SINGLE(stsetl(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stsetl w30, [x29]");
|
||||
TEST_SINGLE(stsetl(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stsetl x30, [x29]");
|
||||
|
||||
TEST_SINGLE(steor(SubRegSize::i8Bit, Reg::r30, Reg::r29), "steorb w30, [x29]");
|
||||
TEST_SINGLE(stseta(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stsetab w30, [x29]");
|
||||
TEST_SINGLE(stseta(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stsetah w30, [x29]");
|
||||
TEST_SINGLE(stseta(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stseta w30, [x29]");
|
||||
TEST_SINGLE(stseta(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stseta x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stsetal(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stsetalb w30, [x29]");
|
||||
TEST_SINGLE(stsetal(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stsetalh w30, [x29]");
|
||||
TEST_SINGLE(stsetal(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stsetal w30, [x29]");
|
||||
TEST_SINGLE(stsetal(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stsetal x30, [x29]");
|
||||
|
||||
TEST_SINGLE(steor(SubRegSize::i8Bit, Reg::r30, Reg::r29), "steorb w30, [x29]");
|
||||
TEST_SINGLE(steor(SubRegSize::i16Bit, Reg::r30, Reg::r29), "steorh w30, [x29]");
|
||||
TEST_SINGLE(steor(SubRegSize::i32Bit, Reg::r30, Reg::r29), "steor w30, [x29]");
|
||||
TEST_SINGLE(steor(SubRegSize::i64Bit, Reg::r30, Reg::r29), "steor x30, [x29]");
|
||||
|
||||
TEST_SINGLE(steorl(SubRegSize::i8Bit, Reg::r30, Reg::r29), "steorlb w30, [x29]");
|
||||
TEST_SINGLE(steorl(SubRegSize::i8Bit, Reg::r30, Reg::r29), "steorlb w30, [x29]");
|
||||
TEST_SINGLE(steorl(SubRegSize::i16Bit, Reg::r30, Reg::r29), "steorlh w30, [x29]");
|
||||
TEST_SINGLE(steorl(SubRegSize::i32Bit, Reg::r30, Reg::r29), "steorl w30, [x29]");
|
||||
TEST_SINGLE(steorl(SubRegSize::i64Bit, Reg::r30, Reg::r29), "steorl x30, [x29]");
|
||||
|
||||
TEST_SINGLE(steora(SubRegSize::i8Bit, Reg::r30, Reg::r29), "steorab w30, [x29]");
|
||||
TEST_SINGLE(steora(SubRegSize::i16Bit, Reg::r30, Reg::r29), "steorah w30, [x29]");
|
||||
TEST_SINGLE(steora(SubRegSize::i32Bit, Reg::r30, Reg::r29), "steora w30, [x29]");
|
||||
TEST_SINGLE(steora(SubRegSize::i64Bit, Reg::r30, Reg::r29), "steora x30, [x29]");
|
||||
|
||||
TEST_SINGLE(steoral(SubRegSize::i8Bit, Reg::r30, Reg::r29), "steoralb w30, [x29]");
|
||||
TEST_SINGLE(steoral(SubRegSize::i16Bit, Reg::r30, Reg::r29), "steoralh w30, [x29]");
|
||||
TEST_SINGLE(steoral(SubRegSize::i32Bit, Reg::r30, Reg::r29), "steoral w30, [x29]");
|
||||
TEST_SINGLE(steoral(SubRegSize::i64Bit, Reg::r30, Reg::r29), "steoral x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stsmax(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stsmaxb w30, [x29]");
|
||||
TEST_SINGLE(stsmax(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stsmaxh w30, [x29]");
|
||||
TEST_SINGLE(stsmax(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stsmax w30, [x29]");
|
||||
TEST_SINGLE(stsmax(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stsmax x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stsmaxl(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stsmaxlb w30, [x29]");
|
||||
TEST_SINGLE(stsmaxl(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stsmaxlh w30, [x29]");
|
||||
TEST_SINGLE(stsmaxl(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stsmaxl w30, [x29]");
|
||||
TEST_SINGLE(stsmaxl(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stsmaxl x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stsmaxa(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stsmaxab w30, [x29]");
|
||||
TEST_SINGLE(stsmaxa(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stsmaxah w30, [x29]");
|
||||
TEST_SINGLE(stsmaxa(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stsmaxa w30, [x29]");
|
||||
TEST_SINGLE(stsmaxa(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stsmaxa x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stsmaxal(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stsmaxalb w30, [x29]");
|
||||
TEST_SINGLE(stsmaxal(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stsmaxalh w30, [x29]");
|
||||
TEST_SINGLE(stsmaxal(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stsmaxal w30, [x29]");
|
||||
TEST_SINGLE(stsmaxal(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stsmaxal x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stsmin(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stsminb w30, [x29]");
|
||||
TEST_SINGLE(stsmin(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stsminh w30, [x29]");
|
||||
TEST_SINGLE(stsmin(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stsmin w30, [x29]");
|
||||
TEST_SINGLE(stsmin(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stsmin x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stsminl(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stsminlb w30, [x29]");
|
||||
TEST_SINGLE(stsminl(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stsminlh w30, [x29]");
|
||||
TEST_SINGLE(stsminl(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stsminl w30, [x29]");
|
||||
TEST_SINGLE(stsminl(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stsminl x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stsmina(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stsminab w30, [x29]");
|
||||
TEST_SINGLE(stsmina(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stsminah w30, [x29]");
|
||||
TEST_SINGLE(stsmina(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stsmina w30, [x29]");
|
||||
TEST_SINGLE(stsmina(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stsmina x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stsminal(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stsminalb w30, [x29]");
|
||||
TEST_SINGLE(stsminal(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stsminalh w30, [x29]");
|
||||
TEST_SINGLE(stsminal(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stsminal w30, [x29]");
|
||||
TEST_SINGLE(stsminal(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stsminal x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stumax(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stumaxb w30, [x29]");
|
||||
TEST_SINGLE(stumax(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stumaxh w30, [x29]");
|
||||
TEST_SINGLE(stumax(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stumax w30, [x29]");
|
||||
TEST_SINGLE(stumax(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stumax x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stumaxl(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stumaxlb w30, [x29]");
|
||||
TEST_SINGLE(stumaxl(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stumaxlh w30, [x29]");
|
||||
TEST_SINGLE(stumaxl(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stumaxl w30, [x29]");
|
||||
TEST_SINGLE(stumaxl(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stumaxl x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stumaxa(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stumaxab w30, [x29]");
|
||||
TEST_SINGLE(stumaxa(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stumaxah w30, [x29]");
|
||||
TEST_SINGLE(stumaxa(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stumaxa w30, [x29]");
|
||||
TEST_SINGLE(stumaxa(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stumaxa x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stumaxal(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stumaxalb w30, [x29]");
|
||||
TEST_SINGLE(stumaxal(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stumaxalh w30, [x29]");
|
||||
TEST_SINGLE(stumaxal(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stumaxal w30, [x29]");
|
||||
TEST_SINGLE(stumaxal(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stumaxal x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stumin(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stuminb w30, [x29]");
|
||||
TEST_SINGLE(stumin(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stuminh w30, [x29]");
|
||||
TEST_SINGLE(stumin(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stumin w30, [x29]");
|
||||
TEST_SINGLE(stumin(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stumin x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stuminl(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stuminlb w30, [x29]");
|
||||
TEST_SINGLE(stuminl(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stuminlh w30, [x29]");
|
||||
TEST_SINGLE(stuminl(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stuminl w30, [x29]");
|
||||
TEST_SINGLE(stuminl(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stuminl x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stumina(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stuminab w30, [x29]");
|
||||
TEST_SINGLE(stumina(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stuminah w30, [x29]");
|
||||
TEST_SINGLE(stumina(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stumina w30, [x29]");
|
||||
TEST_SINGLE(stumina(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stumina x30, [x29]");
|
||||
|
||||
TEST_SINGLE(stuminal(SubRegSize::i8Bit, Reg::r30, Reg::r29), "stuminalb w30, [x29]");
|
||||
TEST_SINGLE(stuminal(SubRegSize::i16Bit, Reg::r30, Reg::r29), "stuminalh w30, [x29]");
|
||||
TEST_SINGLE(stuminal(SubRegSize::i32Bit, Reg::r30, Reg::r29), "stuminal w30, [x29]");
|
||||
TEST_SINGLE(stuminal(SubRegSize::i64Bit, Reg::r30, Reg::r29), "stuminal x30, [x29]");
|
||||
|
||||
TEST_SINGLE(ldswp(SubRegSize::i8Bit, Reg::r30, Reg::r28, Reg::r29), "swpb w30, w28, [x29]");
|
||||
TEST_SINGLE(ldswp(SubRegSize::i16Bit, Reg::r30, Reg::r28, Reg::r29), "swph w30, w28, [x29]");
|
||||
TEST_SINGLE(ldswp(SubRegSize::i32Bit, Reg::r30, Reg::r28, Reg::r29), "swp w30, w28, [x29]");
|
||||
|
||||
+932
-15
File diff suppressed because it is too large.
Load diff
@@ -33,6 +33,12 @@ namespace FHU::Filesystem {
|
||||
return access(Path.c_str(), F_OK) == 0;
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
inline bool ExistsAt(int FD, const fextl::string &Path) {
|
||||
return faccessat(FD, Path.c_str(), F_OK, 0) == 0;
|
||||
}
|
||||
#endif
|
||||
|
||||
enum class CreateDirectoryResult {
|
||||
CREATED,
|
||||
EXISTS,
|
||||
|
||||
@@ -7,6 +7,7 @@
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FHU::Symlinks {
|
||||
#ifndef _WIN32
|
||||
// Checks to see if a filepath is a symlink.
|
||||
inline bool IsSymlink(const fextl::string &Filename) {
|
||||
struct stat Buffer{};
|
||||
@@ -26,4 +27,5 @@ inline std::string_view ResolveSymlink(const fextl::string &Filename, std::span<
|
||||
|
||||
return std::string_view(ResultBuffer.data(), Result);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
@@ -98,8 +98,12 @@ inline int32_t pidfd_open(pid_t pid, unsigned int flags) {
|
||||
#else
|
||||
|
||||
inline int32_t getcpu(uint32_t *cpu, uint32_t *node) {
|
||||
*cpu = GetCurrentProcessorNumber();
|
||||
*node = 0;
|
||||
if (cpu) {
|
||||
*cpu = GetCurrentProcessorNumber();
|
||||
}
|
||||
if (node) {
|
||||
*node = 0;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
@@ -12,10 +12,14 @@ logger.setLevel(logging.WARNING)
|
||||
|
||||
# These defines are temporarily defined since python3-clang doesn't yet support these.
|
||||
# Once this tool gets switched over to C++ then this won't be an issue.
|
||||
# Type definitions redeclared from `clang/include/clang-c/Index.h`
|
||||
|
||||
# Expression that references a C++20 concept.
|
||||
CursorKind.CONCEPTSPECIALIZATIONEXPR = CursorKind(153),
|
||||
|
||||
# Expression that references a C++20 concept.
|
||||
CursorKind.REQUIRESEXPR = CursorKind(154),
|
||||
|
||||
# C++2a std::bit_cast expression.
|
||||
CursorKind.BUILTINBITCASTEXPR = CursorKind(280)
|
||||
|
||||
|
||||
@@ -2,13 +2,13 @@ add_subdirectory(cpp-optparse/)
|
||||
|
||||
set(NAME Common)
|
||||
set(SRCS
|
||||
Config.cpp
|
||||
ArgumentLoader.cpp
|
||||
EnvironmentLoader.cpp
|
||||
StringUtil.cpp)
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
list (APPEND SRCS
|
||||
Config.cpp
|
||||
FEXServerClient.cpp
|
||||
FileFormatCheck.cpp)
|
||||
endif()
|
||||
|
||||
@@ -10,9 +10,11 @@
|
||||
#include <FEXHeaderUtils/SymlinkChecks.h>
|
||||
|
||||
#include <cstring>
|
||||
#ifndef _WIN32
|
||||
#include <linux/limits.h>
|
||||
#include <list>
|
||||
#include <pwd.h>
|
||||
#endif
|
||||
#include <list>
|
||||
#include <utility>
|
||||
#include <json-maker.h>
|
||||
#include <tiny-json.h>
|
||||
@@ -103,12 +105,13 @@ namespace JSON {
|
||||
Dest = json_objClose(Dest);
|
||||
json_end(Dest);
|
||||
|
||||
constexpr int USER_PERMS = S_IRWXU | S_IRWXG | S_IRWXO;
|
||||
int FD = open(Filename.c_str(), O_CREAT | O_WRONLY | O_TRUNC | O_CLOEXEC, USER_PERMS);
|
||||
auto File = FEXCore::File::File(Filename.c_str(),
|
||||
FEXCore::File::FileModes::WRITE |
|
||||
FEXCore::File::FileModes::CREATE |
|
||||
FEXCore::File::FileModes::TRUNCATE);
|
||||
|
||||
if (FD != -1) {
|
||||
write(FD, Buffer, strlen(Buffer));
|
||||
close(FD);
|
||||
if (File.IsValid()) {
|
||||
File.Write(Buffer, strlen(Buffer));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -286,7 +289,7 @@ namespace JSON {
|
||||
// This is because we rewrite `/proc/self/exe` to the absolute program path calculated in here.
|
||||
if (!Program.starts_with('/')) {
|
||||
char ExistsTempPath[PATH_MAX];
|
||||
char *RealPath = realpath(Program.c_str(), ExistsTempPath);
|
||||
char *RealPath = FHU::Filesystem::Absolute(Program.c_str(), ExistsTempPath);
|
||||
if (RealPath) {
|
||||
Program = RealPath;
|
||||
}
|
||||
@@ -315,6 +318,7 @@ namespace JSON {
|
||||
// execveat binfmt_misc args layout: `FEXInterpreter <Path provided to execve pathname> <user provided argv[0]> <user provided argv[n]>...`
|
||||
// - Regular execveat with FD. FD points to file on disk that has been deleted.
|
||||
// execveat binfmt_misc args layout: `FEXInterpreter /dev/fd/<FD> <user provided argv[0]> <user provided argv[n]>...`
|
||||
#ifndef _WIN32
|
||||
if (ExecFDInterp || !ProgramFDFromEnv.empty()) {
|
||||
// Only in the case that FEX is executing an FD will the program argument potentially be a symlink.
|
||||
// This symlink will be in the style of `/dev/fd/<FD>`.
|
||||
@@ -331,6 +335,7 @@ namespace JSON {
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
return Program;
|
||||
}
|
||||
@@ -425,6 +430,7 @@ namespace JSON {
|
||||
return {};
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
char const* FindUserHomeThroughUID() {
|
||||
auto passwd = getpwuid(geteuid());
|
||||
if (passwd) {
|
||||
@@ -525,4 +531,10 @@ namespace JSON {
|
||||
FEXCore::Config::SetConfigFileLocation(GetConfigFileLocation(false), false);
|
||||
FEXCore::Config::SetConfigFileLocation(GetConfigFileLocation(true), true);
|
||||
}
|
||||
#else
|
||||
void InitializeConfigs() {
|
||||
// TODO: Find out how to set this up on WIN32.
|
||||
LogMan::Msg::EFmt("{} Unsupported on WIN32!", __func__);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
@@ -1,6 +1,7 @@
|
||||
#include "Common/Config.h"
|
||||
#include "Common/FileFormatCheck.h"
|
||||
#include "FEXCore/Config/Config.h"
|
||||
#include "FEXCore/Utils/EnumUtils.h"
|
||||
#include "Tools/CommonGUI/IMGui.h"
|
||||
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
@@ -75,6 +76,8 @@ namespace {
|
||||
#define OPT_STR(group, enum, json, default) \
|
||||
LoadedConfig->Set(FEXCore::Config::ConfigOption::CONFIG_##enum, default);
|
||||
#define OPT_STRARRAY(group, enum, json, default) // Do nothing
|
||||
#define OPT_STRENUM(group, enum, json, default) \
|
||||
LoadedConfig->Set(FEXCore::Config::ConfigOption::CONFIG_##enum, std::to_string(FEXCore::ToUnderlying(default)));
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
|
||||
// Erase unnamed options which shouldn't be set
|
||||
@@ -108,6 +111,8 @@ namespace {
|
||||
#define OPT_STR(group, enum, json, default) \
|
||||
if (!LoadedConfig->OptionExists(FEXCore::Config::ConfigOption::CONFIG_##enum)) { LoadedConfig->EraseSet(FEXCore::Config::ConfigOption::CONFIG_##enum, default); }
|
||||
#define OPT_STRARRAY(group, enum, json, default) // Do nothing
|
||||
#define OPT_STRENUM(group, enum, json, default) \
|
||||
if (!LoadedConfig->OptionExists(FEXCore::Config::ConfigOption::CONFIG_##enum)) { LoadedConfig->EraseSet(FEXCore::Config::ConfigOption::CONFIG_##enum, std::to_string(FEXCore::ToUnderlying(default))); }
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
|
||||
// Erase unnamed options which shouldn't be set
|
||||
|
||||
@@ -146,6 +146,8 @@ if (BUILD_TESTS)
|
||||
if (NOT MINGW_BUILD)
|
||||
list(APPEND SRCS TestHarnessRunner/HostRunner.cpp)
|
||||
list(APPEND LIBS LinuxEmulation)
|
||||
else()
|
||||
list(APPEND SRCS WindowsDummyHandlers.cpp)
|
||||
endif()
|
||||
|
||||
add_executable(TestHarnessRunner ${SRCS})
|
||||
|
||||
@@ -15,6 +15,7 @@ $end_info$
|
||||
#include "LinuxSyscalls/x64/Syscalls.h"
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/list.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
@@ -59,49 +60,62 @@ namespace JSON {
|
||||
}
|
||||
|
||||
namespace FEX::HLE {
|
||||
struct open_how;
|
||||
|
||||
static bool LoadFile(fextl::vector<char> &Data, const fextl::string &Filename) {
|
||||
int fd = open(Filename.c_str(), O_RDONLY | O_CLOEXEC);
|
||||
if (fd == -1) {
|
||||
return false;
|
||||
}
|
||||
|
||||
struct stat buf;
|
||||
if (fstat(fd, &buf) != 0) {
|
||||
LogMan::Msg::DFmt("Couldn't load configuration file: fstat");
|
||||
close(fd);
|
||||
return false;
|
||||
}
|
||||
|
||||
auto FileSize = buf.st_size;
|
||||
|
||||
if (FileSize <= 0) {
|
||||
LogMan::Msg::DFmt("FileSize less than or equal to zero specified");
|
||||
close(fd);
|
||||
return false;
|
||||
}
|
||||
|
||||
Data.resize(FileSize);
|
||||
const auto ReadSize = pread(fd, Data.data(), FileSize, 0);
|
||||
|
||||
close(fd);
|
||||
return ReadSize == FileSize;
|
||||
bool FileManager::RootFSPathExists(const char* Filepath) {
|
||||
LOGMAN_THROW_A_FMT(Filepath && Filepath[0] == '/', "Filepath needs to be absolute");
|
||||
return FHU::Filesystem::ExistsAt(RootFSFD, Filepath + 1);
|
||||
}
|
||||
|
||||
struct ThunkDBObject {
|
||||
fextl::string LibraryName;
|
||||
fextl::unordered_set<fextl::string> Depends;
|
||||
fextl::vector<fextl::string> Overlays;
|
||||
bool Enabled{};
|
||||
};
|
||||
|
||||
static void LoadThunkDatabase(fextl::unordered_map<fextl::string, ThunkDBObject>& ThunkDB, bool Is64BitMode, bool Global) {
|
||||
void FileManager::LoadThunkDatabase(fextl::unordered_map<fextl::string, ThunkDBObject>& ThunkDB, bool Global) {
|
||||
auto ThunkDBPath = FEXCore::Config::GetConfigDirectory(Global) + "ThunksDB.json";
|
||||
fextl::vector<char> FileData;
|
||||
if (LoadFile(FileData, ThunkDBPath)) {
|
||||
if (FEXCore::FileLoading::LoadFile(FileData, ThunkDBPath)) {
|
||||
FileData.push_back(0);
|
||||
|
||||
// If the thunksDB file exists then we need to check if the rootfs supports multi-arch or not.
|
||||
const bool RootFSIsMultiarch = RootFSPathExists("/usr/lib/x86_64-linux-gnu/") ||
|
||||
RootFSPathExists("/usr/lib/i386-linux-gnu/");
|
||||
|
||||
fextl::vector<fextl::string> PathPrefixes{};
|
||||
if (RootFSIsMultiarch) {
|
||||
// Multi-arch debian distros have a fairly complex arrangement of filepaths.
|
||||
// These fractal out to the combination of library prefixes with arch suffixes.
|
||||
constexpr static std::array<std::string_view, 4> LibPrefixes = {
|
||||
"/usr/lib",
|
||||
"/usr/local/lib",
|
||||
"/lib",
|
||||
"/usr/lib/pressure-vessel/overrides/lib",
|
||||
};
|
||||
|
||||
// We only need to generate 32-bit or 64-bit depending on the operating mode.
|
||||
const auto ArchPrefix = Is64BitMode() ?
|
||||
"x86_64-linux-gnu" :
|
||||
"i386-linux-gnu";
|
||||
|
||||
for (auto Prefix : LibPrefixes) {
|
||||
PathPrefixes.emplace_back(fextl::fmt::format("{}/{}", Prefix, ArchPrefix));
|
||||
}
|
||||
}
|
||||
else {
|
||||
// Non multi-arch supporting distros like Fedora and Debian have a much more simple layout.
|
||||
// lib/ folders refer to 32-bit library folders.
|
||||
// li64/ folders refer to 64-bit library folders.
|
||||
constexpr static std::array<std::string_view, 4> LibPrefixes = {
|
||||
"/usr",
|
||||
"/usr/local",
|
||||
"", // root, the '/' will be appended in the next step.
|
||||
"/usr/lib/pressure-vessel/overrides",
|
||||
};
|
||||
|
||||
// We only need to generate 32-bit or 64-bit depending on the operating mode.
|
||||
const auto ArchPrefix = Is64BitMode() ?
|
||||
"lib64" :
|
||||
"lib";
|
||||
|
||||
for (auto Prefix : LibPrefixes) {
|
||||
PathPrefixes.emplace_back(fextl::fmt::format("{}/{}", Prefix, ArchPrefix));
|
||||
}
|
||||
}
|
||||
|
||||
JSON::JsonAllocator Pool {
|
||||
.PoolObject = {
|
||||
.init = JSON::PoolInit,
|
||||
@@ -143,39 +157,24 @@ static void LoadThunkDatabase(fextl::unordered_map<fextl::string, ThunkDBObject>
|
||||
}
|
||||
}
|
||||
else if (ItemName == "Overlay") {
|
||||
auto AddWithReplacement = [Is64BitMode, HomeDirectory](ThunkDBObject& DBObject, fextl::string LibraryItem) {
|
||||
constexpr static std::array<std::string_view, 4> LibPrefixes = {
|
||||
"/usr/lib",
|
||||
"/usr/local/lib",
|
||||
"/lib",
|
||||
"/usr/lib/pressure-vessel/overrides/lib",
|
||||
};
|
||||
|
||||
constexpr static std::array<std::string_view, 2> ArchPrefixes = {
|
||||
"i386",
|
||||
"x86_64",
|
||||
};
|
||||
|
||||
auto AddWithReplacement = [HomeDirectory, &PathPrefixes](ThunkDBObject& DBObject, fextl::string LibraryItem) {
|
||||
// Walk through template string and fill in prefixes from right to left
|
||||
|
||||
using namespace std::string_view_literals;
|
||||
const std::pair PrefixArch { "@PREFIX_ARCH@"sv, LibraryItem.find("@PREFIX_ARCH@") };
|
||||
const std::pair PrefixHome { "@HOME@"sv, LibraryItem.find("@HOME@") };
|
||||
const std::pair PrefixLib { "@PREFIX_LIB@"sv, LibraryItem.find("@PREFIX_LIB@") };
|
||||
|
||||
fextl::string::size_type PrefixPositions[] = {
|
||||
PrefixArch.second, PrefixHome.second, PrefixLib.second,
|
||||
PrefixHome.second, PrefixLib.second,
|
||||
};
|
||||
// Sort offsets in descending order to enable safe in-place replacement
|
||||
std::sort(std::begin(PrefixPositions), std::end(PrefixPositions), std::greater<>{});
|
||||
|
||||
for (auto& LibPrefix : LibPrefixes) {
|
||||
for (auto& LibPrefix : PathPrefixes) {
|
||||
fextl::string Replacement = LibraryItem;
|
||||
for (auto PrefixPos : PrefixPositions) {
|
||||
if (PrefixPos == fextl::string::npos) {
|
||||
continue;
|
||||
} else if (PrefixPos == PrefixArch.second) {
|
||||
Replacement.replace(PrefixPos, PrefixArch.first.size(), ArchPrefixes[Is64BitMode]);
|
||||
} else if (PrefixPos == PrefixHome.second) {
|
||||
Replacement.replace(PrefixPos, PrefixHome.first.size(), HomeDirectory);
|
||||
} else if (PrefixPos == PrefixLib.second) {
|
||||
@@ -244,13 +243,20 @@ FileManager::FileManager(FEXCore::Context::Context *ctx)
|
||||
ConfigPaths.emplace_back(FEXCore::Config::GetApplicationConfig(AppName, false));
|
||||
}
|
||||
|
||||
if (!LDPath().empty()) {
|
||||
RootFSFD = open(LDPath().c_str(), O_DIRECTORY | O_PATH | O_CLOEXEC);
|
||||
if (RootFSFD == -1) {
|
||||
RootFSFD = AT_FDCWD;
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unordered_map<fextl::string, ThunkDBObject> ThunkDB;
|
||||
LoadThunkDatabase(ThunkDB, Is64BitMode(), true);
|
||||
LoadThunkDatabase(ThunkDB, Is64BitMode(), false);
|
||||
LoadThunkDatabase(ThunkDB, true);
|
||||
LoadThunkDatabase(ThunkDB, false);
|
||||
|
||||
for (const auto &Path : ConfigPaths) {
|
||||
fextl::vector<char> FileData;
|
||||
if (LoadFile(FileData, Path)) {
|
||||
if (FEXCore::FileLoading::LoadFile(FileData, Path)) {
|
||||
JSON::JsonAllocator Pool {
|
||||
.PoolObject = {
|
||||
.init = JSON::PoolInit,
|
||||
@@ -339,13 +345,6 @@ FileManager::FileManager(FEXCore::Context::Context *ctx)
|
||||
}
|
||||
|
||||
UpdatePID(::getpid());
|
||||
|
||||
if (!LDPath().empty()) {
|
||||
RootFSFD = open(LDPath().c_str(), O_DIRECTORY | O_PATH | O_CLOEXEC);
|
||||
if (RootFSFD == -1) {
|
||||
RootFSFD = AT_FDCWD;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
FileManager::~FileManager() {
|
||||
|
||||
Loaded 100 of 134 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user