mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 18:00:17 +02:00
Compare commits
440
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
a141d8bd93 | ||
|
|
52e21a6e02 | ||
|
|
7eb4520317 | ||
|
|
d4515c3a6c | ||
|
|
d214ebc8f2 | ||
|
|
c379eede3b | ||
|
|
8ea92ab9b6 | ||
|
|
40d9c66784 | ||
|
|
cc4da669c9 | ||
|
|
22a58925c7 | ||
|
|
25ed2578c2 | ||
|
|
90dcfab131 | ||
|
|
91ac4c9a3a | ||
|
|
0d86ee575b | ||
|
|
0409698783 | ||
|
|
ec40d53cc9 | ||
|
|
d46e6fac22 | ||
|
|
4593099882 | ||
|
|
21e29af48b | ||
|
|
892e44a900 | ||
|
|
bca29c2549 | ||
|
|
330e7f628c | ||
|
|
efe401c7a4 | ||
|
|
b3d88c043d | ||
|
|
edd36dac91 | ||
|
|
6de5fb2885 | ||
|
|
3b2ebafd83 | ||
|
|
dd4f508b77 | ||
|
|
3058a825b7 | ||
|
|
bdbeef3b9a | ||
|
|
8ead4a3c34 | ||
|
|
2d53a9b0dc | ||
|
|
8f47b46219 | ||
|
|
4f0d35bae0 | ||
|
|
de6d907e27 | ||
|
|
c9fc347de1 | ||
|
|
85e67f92c5 | ||
|
|
034a27ce5a | ||
|
|
056abb5901 | ||
|
|
af28406dbc | ||
|
|
376d6ba72c | ||
|
|
736a73453f | ||
|
|
6ced309c5c | ||
|
|
a30593efd6 | ||
|
|
3fa400bc55 | ||
|
|
90ea16325a | ||
|
|
be84a4332a | ||
|
|
cfeba859ac | ||
|
|
38ed7c9bb1 | ||
|
|
315ee90837 | ||
|
|
278574ce91 | ||
|
|
6c225d5469 | ||
|
|
d6a466ac0b | ||
|
|
58526d4faa | ||
|
|
3c90a82e75 | ||
|
|
06a27deebb | ||
|
|
f607f877c9 | ||
|
|
7273041314 | ||
|
|
997d041a84 | ||
|
|
226233ddce | ||
|
|
6c8edcbea5 | ||
|
|
e0f27bb855 | ||
|
|
4a14003f34 | ||
|
|
216a37348b | ||
|
|
ef7b2a9d4f | ||
|
|
ffb1cf4c7c | ||
|
|
964165eabe | ||
|
|
dca2d74eb6 | ||
|
|
e136ff52ab | ||
|
|
0df4944a7b | ||
|
|
2bb296df3b | ||
|
|
52380a3927 | ||
|
|
732f725a24 | ||
|
|
8887b16299 | ||
|
|
75c8b08d9b | ||
|
|
9741dc214d | ||
|
|
8b4b80c725 | ||
|
|
e3b6cccca1 | ||
|
|
d28ef1859d | ||
|
|
74b59f458d | ||
|
|
4ae04c2a9a | ||
|
|
7bd0789402 | ||
|
|
eb9fb9e834 | ||
|
|
c0ef0a7503 | ||
|
|
61719115e5 | ||
|
|
4a8cbe2ab8 | ||
|
|
c966f44189 | ||
|
|
193ef8b232 | ||
|
|
ad70a60d05 | ||
|
|
9dfd5b5526 | ||
|
|
09ec374eec | ||
|
|
8c1e9eda12 | ||
|
|
0eedd55dfd | ||
|
|
58c86d16f3 | ||
|
|
4a9170e981 | ||
|
|
b5b8ff01d5 | ||
|
|
6c86ffdeb7 | ||
|
|
70a1d92d9f | ||
|
|
0749477eb9 | ||
|
|
fb54341a1f | ||
|
|
db601d333b | ||
|
|
af1c2cccac | ||
|
|
bf1b76dcd2 | ||
|
|
48787ab460 | ||
|
|
7d4bf84304 | ||
|
|
5cf5d3e3a2 | ||
|
|
8ddb229447 | ||
|
|
9fdd96af61 | ||
|
|
7fbdd0607f | ||
|
|
0c0a1d8f12 | ||
|
|
30dd9ed267 | ||
|
|
5f0a1d55e5 | ||
|
|
17bef26708 | ||
|
|
58b5620f97 | ||
|
|
327b62ea45 | ||
|
|
91c6d693df | ||
|
|
aa3df3914b | ||
|
|
ad3a024b69 | ||
|
|
b9ec94264e | ||
|
|
fee9e91c4f | ||
|
|
1dc560d45a | ||
|
|
258e2b80f4 | ||
|
|
dcb889704d | ||
|
|
93096c27f9 | ||
|
|
3c6eba561f | ||
|
|
d2714f3338 | ||
|
|
b917db34c7 | ||
|
|
c45abaaa6b | ||
|
|
a2286cb00a | ||
|
|
481787d45c | ||
|
|
756002eef8 | ||
|
|
6789919fca | ||
|
|
d8c615ec65 | ||
|
|
6a1eb0e508 | ||
|
|
984b0260d5 | ||
|
|
785e59e889 | ||
|
|
671be5b191 | ||
|
|
b4e556e63a | ||
|
|
2128943b57 | ||
|
|
6ddb1094b6 | ||
|
|
b9d93b19d4 | ||
|
|
97a5da232d | ||
|
|
0d77af5366 | ||
|
|
a3435f2d22 | ||
|
|
154dd46b6d | ||
|
|
782952d55d | ||
|
|
a6bd293ae8 | ||
|
|
decd112a84 | ||
|
|
f9c15fa0a5 | ||
|
|
1121f2a1fb | ||
|
|
a7989eb79f | ||
|
|
80c199cefb | ||
|
|
317f92f4c8 | ||
|
|
dd344710d4 | ||
|
|
e4e8683c7b | ||
|
|
b2407352a9 | ||
|
|
e1b5e3e112 | ||
|
|
09793d295f | ||
|
|
e9f8c01e04 | ||
|
|
f69821d7db | ||
|
|
1e03219ca3 | ||
|
|
de8b4438e3 | ||
|
|
2cfd3330f0 | ||
|
|
1988b432ab | ||
|
|
6b2363d3f6 | ||
|
|
e76af56c86 | ||
|
|
b7173a3ae2 | ||
|
|
3f71517e1f | ||
|
|
17fbe52cf6 | ||
|
|
ecc526a9d8 | ||
|
|
436f4aa953 | ||
|
|
fd10ce0600 | ||
|
|
d9cd40eb48 | ||
|
|
9b9ab16b55 | ||
|
|
14256e7a90 | ||
|
|
d36c387c1f | ||
|
|
7e61b172b2 | ||
|
|
bd0cae9298 | ||
|
|
6ae94581bb | ||
|
|
ea8ae2bf2c | ||
|
|
694bc3312e | ||
|
|
f565939437 | ||
|
|
272d2747cd | ||
|
|
675956c92d | ||
|
|
03410ed4a2 | ||
|
|
ac88e7fabf | ||
|
|
71d42d7e73 | ||
|
|
2613762ac2 | ||
|
|
831b21cd39 | ||
|
|
7869d80073 | ||
|
|
ea39d80e83 | ||
|
|
62e481a608 | ||
|
|
bac7148d69 | ||
|
|
62f16a3c4b | ||
|
|
248ecd54e4 | ||
|
|
1f59accb4e | ||
|
|
dd7e4375cf | ||
|
|
3df3999138 | ||
|
|
1ca6443c9d | ||
|
|
a804cb5d07 | ||
|
|
a0883c9615 | ||
|
|
03c968a8fe | ||
|
|
e34de22cf4 | ||
|
|
3db1f1e01f | ||
|
|
5e467c737f | ||
|
|
dfc42fa7c5 | ||
|
|
f1016e82c7 | ||
|
|
5e099a5dbd | ||
|
|
8dceec9316 | ||
|
|
6f51092998 | ||
|
|
4e83ca67f5 | ||
|
|
76e6c6f8ea | ||
|
|
343a529e63 | ||
|
|
c6ccc68607 | ||
|
|
8009f4a6ee | ||
|
|
e126cecb16 | ||
|
|
5e19202540 | ||
|
|
06466bac0e | ||
|
|
156edcacb2 | ||
|
|
bf4b8d246a | ||
|
|
abc37ec35e | ||
|
|
dd735464a2 | ||
|
|
7041bdb144 | ||
|
|
19e8c3fcb6 | ||
|
|
06c5135d4e | ||
|
|
f66066cbd8 | ||
|
|
a144622a06 | ||
|
|
05c814b8c1 | ||
|
|
e0af51fb06 | ||
|
|
7b749558df | ||
|
|
1797c3c617 | ||
|
|
28a1c28cb4 | ||
|
|
20fcbb9c62 | ||
|
|
7c9c97f2b9 | ||
|
|
d72e121fd4 | ||
|
|
003fc3dab5 | ||
|
|
2c883d7cdc | ||
|
|
fcba49768c | ||
|
|
959af9c3af | ||
|
|
1edacf05e9 | ||
|
|
1f756fb386 | ||
|
|
5ebdc01783 | ||
|
|
aa29201e00 | ||
|
|
b98acdf97a | ||
|
|
7aeb4788ed | ||
|
|
7b1db40d4a | ||
|
|
dcaa90a855 | ||
|
|
0a74efe3f8 | ||
|
|
5a73134c7c | ||
|
|
43d535cf8d | ||
|
|
4a71edf7bc | ||
|
|
ae90040ed2 | ||
|
|
fbf982f001 | ||
|
|
839e3ac775 | ||
|
|
65d230f32b | ||
|
|
0882b0db0c | ||
|
|
3d92c1fb32 | ||
|
|
d4e679432c | ||
|
|
54c8cb9fd7 | ||
|
|
221ee1a122 | ||
|
|
2b29f90c92 | ||
|
|
d7ec0e570f | ||
|
|
86c2144943 | ||
|
|
ee85230db2 | ||
|
|
fde99a8dc1 | ||
|
|
2e692a1146 | ||
|
|
73f36fb5b3 | ||
|
|
f866ad52dc | ||
|
|
9566910574 | ||
|
|
4188d4d10b | ||
|
|
d75152cb74 | ||
|
|
add54b8089 | ||
|
|
7431495f13 | ||
|
|
5ed5566106 | ||
|
|
2d9a4412a4 | ||
|
|
982abd5719 | ||
|
|
f09db0cbe5 | ||
|
|
80cacc462e | ||
|
|
1a2b2f8870 | ||
|
|
e8c576047f | ||
|
|
995d3152eb | ||
|
|
ddc40cc273 | ||
|
|
5ee311b831 | ||
|
|
ea90296e22 | ||
|
|
58c1a8d148 | ||
|
|
87a19c7938 | ||
|
|
5b0709de2f | ||
|
|
87db37c65e | ||
|
|
100db13f99 | ||
|
|
d2061b27a2 | ||
|
|
002dfb9b84 | ||
|
|
0ab1241f09 | ||
|
|
aa63656ff1 | ||
|
|
6ba32da73a | ||
|
|
1129e71069 | ||
|
|
3b32fd5c49 | ||
|
|
7df1ed271f | ||
|
|
67e9b40bab | ||
|
|
62b80903f0 | ||
|
|
993d917478 | ||
|
|
ac20880a14 | ||
|
|
99cfe05ee5 | ||
|
|
7c187e82b7 | ||
|
|
2a760660a9 | ||
|
|
4b11826a7d | ||
|
|
887f586874 | ||
|
|
db94c04179 | ||
|
|
51e64c69f3 | ||
|
|
97c46843d2 | ||
|
|
16310c91c9 | ||
|
|
220e421c22 | ||
|
|
a156752fae | ||
|
|
5d1c574dc5 | ||
|
|
6310c20217 | ||
|
|
0dfc30fdd8 | ||
|
|
f050571696 | ||
|
|
6db5bc3721 | ||
|
|
4f84c281ef | ||
|
|
0ca438e081 | ||
|
|
2472aee473 | ||
|
|
714856f74c | ||
|
|
e98932df92 | ||
|
|
95623dac63 | ||
|
|
827015a8e3 | ||
|
|
a379818386 | ||
|
|
4516e4e9f6 | ||
|
|
83e44961fc | ||
|
|
bc2e959897 | ||
|
|
1b0b4ba416 | ||
|
|
25a0752e58 | ||
|
|
b0a7e0bbcc | ||
|
|
af561abc71 | ||
|
|
3bf323aa2b | ||
|
|
c9a33c638a | ||
|
|
8d1042859d | ||
|
|
9072571fc1 | ||
|
|
d8748b8216 | ||
|
|
2f930b201a | ||
|
|
26b4195f1f | ||
|
|
5c078d13c0 | ||
|
|
2825ac282a | ||
|
|
cd31a79f41 | ||
|
|
52eb9f46d2 | ||
|
|
08c430dcfb | ||
|
|
b486aeeac5 | ||
|
|
13da6102b3 | ||
|
|
1e1c4e017e | ||
|
|
ceaaab5a86 | ||
|
|
2b7d03d4f9 | ||
|
|
c8810557c1 | ||
|
|
2a4bfe49f5 | ||
|
|
b71879e90e | ||
|
|
63ae262752 | ||
|
|
2c3a4fef26 | ||
|
|
e696e32b95 | ||
|
|
3f9df49e3e | ||
|
|
1ac3d3c5a6 | ||
|
|
b3d1aade71 | ||
|
|
c300c723f7 | ||
|
|
d0ec069dcb | ||
|
|
6811406b24 | ||
|
|
0879994aff | ||
|
|
b2b5ccf69c | ||
|
|
6f5d4fbf30 | ||
|
|
8ad1ea499d | ||
|
|
4df17abea7 | ||
|
|
972c8a97dc | ||
|
|
135d17fbd3 | ||
|
|
095b2a8926 | ||
|
|
297677c2c6 | ||
|
|
2a00f23459 | ||
|
|
68ad916a91 | ||
|
|
0f3883152b | ||
|
|
60baee4020 | ||
|
|
5a7ff4780f | ||
|
|
dd2845401c | ||
|
|
a47c947065 | ||
|
|
203ac51681 | ||
|
|
a7986993bf | ||
|
|
ce59d40570 | ||
|
|
19f33e22d7 | ||
|
|
7d818e18be | ||
|
|
52af0aceea | ||
|
|
cd1cf6f615 | ||
|
|
2f925aeb22 | ||
|
|
f99b42da5e | ||
|
|
40beef061f | ||
|
|
f0a434c167 | ||
|
|
92b66dbf17 | ||
|
|
f336261d6b | ||
|
|
c6324b83ea | ||
|
|
54873992ab | ||
|
|
38d568a807 | ||
|
|
dcef1212c6 | ||
|
|
2030b70ff8 | ||
|
|
41a188eaec | ||
|
|
f3ddaf455c | ||
|
|
aa979e4bc7 | ||
|
|
baddc56f01 | ||
|
|
9836e43d10 | ||
|
|
c8cecd9f46 | ||
|
|
38c59a44de | ||
|
|
fddc86c2b1 | ||
|
|
0dddc9b5cb | ||
|
|
8b14bd4e87 | ||
|
|
5c77969e83 | ||
|
|
e049596252 | ||
|
|
afa1327242 | ||
|
|
da3b7f5a41 | ||
|
|
100f61ee24 | ||
|
|
dcd2794ff5 | ||
|
|
8d1d3fe12b | ||
|
|
87ae0f058a | ||
|
|
7e1be6625f | ||
|
|
0c6fcd1678 | ||
|
|
37ac46273d | ||
|
|
1e21416ccb | ||
|
|
e2353959c1 | ||
|
|
61d3d5b068 | ||
|
|
572e5ed9c6 | ||
|
|
4dbfa2febb | ||
|
|
0293d2027d | ||
|
|
c28b94890c | ||
|
|
63a910d2fc | ||
|
|
fbef5c6a6a | ||
|
|
674b6e9f43 | ||
|
|
7897b6ad55 | ||
|
|
6769b54e1c | ||
|
|
734ab4429f | ||
|
|
5db6a64f3f | ||
|
|
dcc458ea94 | ||
|
|
7354e7e3c9 | ||
|
|
07ef765f7e | ||
|
|
c58ad8b593 | ||
|
|
31091c1053 | ||
|
|
c95f9f5e6a | ||
|
|
93c3eb064d | ||
|
|
b13162c5bf | ||
|
|
ee1725684a | ||
|
|
a07b234659 |
No files matched your search
@@ -20,3 +20,5 @@
|
||||
# Whole-tree reformat with clang-format-19
|
||||
5267cde60e7642852d18f20ae8568643bb5293d5
|
||||
|
||||
# Minor reformat with clang-format-19
|
||||
9fdd96af61c969cb5732471223f00eda64b7a069
|
||||
@@ -13,6 +13,7 @@ env:
|
||||
BUILD_TYPE: Release
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
FEX_PORTABLE: 1
|
||||
|
||||
jobs:
|
||||
build_plus_test:
|
||||
@@ -33,7 +34,6 @@ jobs:
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
@@ -136,6 +136,9 @@ jobs:
|
||||
- name: FEXLinuxTests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
env:
|
||||
# These tests require non-portable install due to thunks.
|
||||
FEX_PORTABLE: 0
|
||||
run: cmake --build . --config $BUILD_TYPE --target fex_linux_tests_all
|
||||
|
||||
- name: FEXLinuxTests Results move
|
||||
|
||||
@@ -20,6 +20,7 @@ env:
|
||||
BUILD_TYPE: Release
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
FEX_PORTABLE: 1
|
||||
|
||||
jobs:
|
||||
glibc_fault_test:
|
||||
@@ -40,7 +41,6 @@ jobs:
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
|
||||
@@ -13,6 +13,7 @@ env:
|
||||
BUILD_TYPE: Release
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
FEX_PORTABLE: 1
|
||||
|
||||
jobs:
|
||||
hostrunner_tests:
|
||||
@@ -33,7 +34,6 @@ jobs:
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
|
||||
@@ -33,7 +33,6 @@ jobs:
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
|
||||
@@ -48,7 +48,6 @@ jobs:
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
@@ -78,7 +77,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_TESTS=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=$MINGW_TRIPLE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
|
||||
@@ -60,7 +60,6 @@ jobs:
|
||||
START_REV: ${{ github.event.pull_request.base.sha }}
|
||||
END_REV: ${{ github.event.pull_request.head.sha }}
|
||||
CHANGED_FILES: ${{ steps.changed-files.outputs.all_changed_files }}
|
||||
# Using --diff_from_common_commit option available in clang-format-19
|
||||
run: |
|
||||
python ./External/code-format-helper/code-format-helper.py \
|
||||
--repo "FEX-emu/FEX" \
|
||||
|
||||
@@ -13,6 +13,7 @@ env:
|
||||
BUILD_TYPE: Release
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
FEX_PORTABLE: 1
|
||||
|
||||
jobs:
|
||||
vixl_simulator:
|
||||
@@ -34,7 +35,6 @@ jobs:
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
|
||||
@@ -46,12 +46,12 @@ jobs:
|
||||
- name: Configure CMake arm64ec
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build_arm64ec
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=arm64ec-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=/usr -DBUILD_TESTS=False -DCMAKE_INSTALL_PREFIX=/usr
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=arm64ec-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=/usr -DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=/usr
|
||||
|
||||
- name: Configure CMake wow64
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build_wow64
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=aarch64-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=/usr -DBUILD_TESTS=False -DCMAKE_INSTALL_PREFIX=/usr
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -DCMAKE_TOOLCHAIN_FILE=$GITHUB_WORKSPACE/Data/CMake/toolchain_mingw.cmake -DMINGW_TRIPLE=aarch64-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=/usr -DBUILD_TESTING=False -DCMAKE_INSTALL_PREFIX=/usr
|
||||
|
||||
- name: Build arm64ec
|
||||
working-directory: ${{runner.workspace}}/build_arm64ec
|
||||
|
||||
@@ -46,3 +46,6 @@
|
||||
[submodule "External/tracy"]
|
||||
path = External/tracy
|
||||
url = https://github.com/wolfpld/tracy
|
||||
[submodule "External/range-v3"]
|
||||
path = External/range-v3
|
||||
url = https://github.com/ericniebler/range-v3.git
|
||||
+18
-54
@@ -4,7 +4,6 @@ project(FEX C CXX ASM)
|
||||
INCLUDE (CheckIncludeFiles)
|
||||
CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
|
||||
|
||||
option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
|
||||
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
|
||||
option(BUILD_THUNKS "Build thunks" FALSE)
|
||||
option(BUILD_FEXCONFIG "Build FEXConfig" TRUE)
|
||||
@@ -304,7 +303,8 @@ set (CMAKE_LINKER_FLAGS_RELEASE "${CMAKE_LINKER_FLAGS_RELEASE} -fomit-frame-poin
|
||||
|
||||
include_directories(External/robin-map/include/)
|
||||
|
||||
if (BUILD_TESTS OR ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
|
||||
include(CTest)
|
||||
if (BUILD_TESTING OR ENABLE_VIXL_DISASSEMBLER OR ENABLE_VIXL_SIMULATOR)
|
||||
add_subdirectory(External/vixl/)
|
||||
include_directories(SYSTEM External/vixl/src/)
|
||||
endif()
|
||||
@@ -335,7 +335,7 @@ endif()
|
||||
add_definitions(-Wno-trigraphs)
|
||||
add_definitions(-DGLOBAL_DATA_DIRECTORY="${DATA_DIRECTORY}/")
|
||||
|
||||
if (BUILD_TESTS)
|
||||
if (BUILD_TESTING)
|
||||
find_package(Catch2 3 QUIET)
|
||||
if (NOT Catch2_FOUND)
|
||||
add_subdirectory(External/Catch2/)
|
||||
@@ -345,6 +345,9 @@ if (BUILD_TESTS)
|
||||
endif()
|
||||
|
||||
include(Catch)
|
||||
else ()
|
||||
# Override any previously generated test list to avoid running stale test binaries
|
||||
file(GENERATE OUTPUT CTestTestfile.cmake CONTENT "# No tests since BUILD_TESTING is disabled")
|
||||
endif()
|
||||
|
||||
find_package(fmt QUIET)
|
||||
@@ -354,6 +357,12 @@ if (NOT fmt_FOUND)
|
||||
add_subdirectory(External/fmt/)
|
||||
endif()
|
||||
|
||||
find_package(range-v3 QUIET)
|
||||
if (NOT range-v3_FOUND)
|
||||
add_subdirectory(External/range-v3/)
|
||||
target_compile_definitions(range-v3 INTERFACE RANGES_DISABLE_DEPRECATED_WARNINGS)
|
||||
endif()
|
||||
|
||||
add_subdirectory(External/tiny-json/)
|
||||
include_directories(External/tiny-json/)
|
||||
|
||||
@@ -449,13 +458,8 @@ endif()
|
||||
|
||||
add_compile_options(-Wall)
|
||||
|
||||
include(CTest)
|
||||
if (BUILD_TESTS)
|
||||
if (BUILD_TESTING)
|
||||
message(STATUS "Unit tests are enabled")
|
||||
if (NOT BUILD_TESTING)
|
||||
# CMake checks this variable before generating CTestTestfile.cmake
|
||||
message(SEND_ERROR "Unit tests require BUILD_TESTING to be enabled")
|
||||
endif()
|
||||
|
||||
set (TEST_JOB_COUNT "" CACHE STRING "Override number of parallel jobs to use while running tests")
|
||||
if (TEST_JOB_COUNT)
|
||||
@@ -486,10 +490,11 @@ file(GLOB CONFIG_SOURCES CONFIGURE_DEPENDS ${CMAKE_CURRENT_SOURCE_DIR}/Data/*.js
|
||||
# Any application configuration json file gets installed
|
||||
foreach(CONFIG_SRC ${CONFIG_SOURCES})
|
||||
install(FILES ${CONFIG_SRC}
|
||||
DESTINATION ${DATA_DIRECTORY}/)
|
||||
DESTINATION ${DATA_DIRECTORY}/
|
||||
COMPONENT Runtime)
|
||||
endforeach()
|
||||
|
||||
if (BUILD_TESTS)
|
||||
if (BUILD_TESTING)
|
||||
add_subdirectory(unittests/)
|
||||
endif()
|
||||
|
||||
@@ -550,6 +555,7 @@ if (BUILD_THUNKS)
|
||||
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest
|
||||
)"
|
||||
DEPENDS guest-libs
|
||||
COMPONENT Runtime
|
||||
)
|
||||
|
||||
install(
|
||||
@@ -559,6 +565,7 @@ if (BUILD_THUNKS)
|
||||
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest_32
|
||||
)"
|
||||
DEPENDS guest-libs-32
|
||||
COMPONENT Runtime
|
||||
)
|
||||
|
||||
add_custom_target(uninstall_guest-libs
|
||||
@@ -600,46 +607,3 @@ if (OVERRIDE_VERSION STREQUAL "detect")
|
||||
else()
|
||||
set(GIT_DESCRIBE_STRING "FEX-${OVERRIDE_VERSION}")
|
||||
endif()
|
||||
|
||||
# Parse the version here
|
||||
# Change something like `FEX-2106.1-76-<hash>` in to a list
|
||||
string(REPLACE "-" ";" DESCRIBE_LIST ${GIT_DESCRIBE_STRING})
|
||||
|
||||
# Extract the `2106.1` element
|
||||
list(GET DESCRIBE_LIST 1 DESCRIBE_LIST)
|
||||
|
||||
# Change `2106.1` in to a list
|
||||
string(REPLACE "." ";" DESCRIBE_LIST ${DESCRIBE_LIST})
|
||||
|
||||
# Calculate list size
|
||||
list(LENGTH DESCRIBE_LIST LIST_SIZE)
|
||||
|
||||
# Pull out the major version
|
||||
list(GET DESCRIBE_LIST 0 FEX_VERSION_MAJOR)
|
||||
|
||||
# Minor version only exists if there is a .1 at the end
|
||||
# eg: 2106 versus 2106.1
|
||||
if (LIST_SIZE GREATER 1)
|
||||
list(GET DESCRIBE_LIST 1 FEX_VERSION_MINOR)
|
||||
endif()
|
||||
|
||||
# Package creation
|
||||
set (CPACK_GENERATOR "DEB")
|
||||
set (CPACK_PACKAGE_NAME fex-emu)
|
||||
set (CPACK_PACKAGE_FILE_NAME "${CPACK_PACKAGE_NAME}-${GIT_DESCRIBE_STRING}_${CMAKE_SYSTEM_PROCESSOR}")
|
||||
set (CPACK_PACKAGE_CONTACT "FEX-Emu Maintainers <team@fex-emu.com>")
|
||||
set (CPACK_PACKAGE_VERSION_MAJOR "${FEX_VERSION_MAJOR}")
|
||||
set (CPACK_PACKAGE_VERSION_MINOR "${FEX_VERSION_MINOR}")
|
||||
set (CPACK_PACKAGE_VERSION_PATCH "${FEX_VERSION_PATCH}")
|
||||
set (CPACK_PACKAGE_DESCRIPTION_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/Description.txt")
|
||||
|
||||
# Debian defines
|
||||
set (CPACK_DEBIAN_PACKAGE_DEPENDS "libc6, libstdc++6, libepoxy0, libsdl2-2.0-0, libegl1, libx11-6, squashfuse")
|
||||
set (CPACK_DEBIAN_PACKAGE_CONTROL_EXTRA
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/postinst;${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/prerm;${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/CPack/triggers")
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
|
||||
# binfmt_misc conflicts with qemu-user-static
|
||||
# We also only install binfmt_misc on aarch64 hosts
|
||||
set (CPACK_DEBIAN_PACKAGE_CONFLICTS "${CPACK_DEBIAN_PACKAGE_CONFLICTS}, qemu-user-static")
|
||||
endif()
|
||||
include (CPack)
|
||||
@@ -36,24 +36,33 @@ public:
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
}
|
||||
|
||||
void adr(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded adr(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
if (IsADRRange(Imm)) [[likely]] {
|
||||
constexpr uint32_t Op = 0b0001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
void adr(ARMEmitter::Register rd, ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded adr(ARMEmitter::Register rd, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::ADR});
|
||||
constexpr uint32_t Op = 0b0001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void adr(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded adr(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
adr(rd, &Label->Backward);
|
||||
return adr(rd, &Label->Backward);
|
||||
} else {
|
||||
adr(rd, &Label->Forward);
|
||||
return adr(rd, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -62,32 +71,42 @@ public:
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
}
|
||||
|
||||
void adrp(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded adrp(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
int64_t Imm = reinterpret_cast<int64_t>(Label->Location) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
|
||||
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b1001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
if (IsADRPRange(Imm) && IsADRPAligned(Imm)) [[likely]] {
|
||||
constexpr uint32_t Op = 0b1001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
void adrp(ARMEmitter::Register rd, ForwardLabel* Label) {
|
||||
|
||||
[[nodiscard]] BranchEncodeSucceeded adrp(ARMEmitter::Register rd, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::ADRP});
|
||||
constexpr uint32_t Op = 0b1001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void adrp(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded adrp(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
adrp(rd, &Label->Backward);
|
||||
return adrp(rd, &Label->Backward);
|
||||
} else {
|
||||
adrp(rd, &Label->Forward);
|
||||
return adrp(rd, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
void LongAddressGen(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded LongAddressGen(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
int64_t Imm = reinterpret_cast<int64_t>(Label->Location) - (GetCursorAddress<int64_t>());
|
||||
if (IsADRRange(Imm)) {
|
||||
// If the range is in ADR range then we can just use ADR.
|
||||
adr(rd, Label);
|
||||
return adr(rd, Label);
|
||||
} else if (IsADRPRange(Imm)) {
|
||||
int64_t ADRPImm = (reinterpret_cast<int64_t>(Label->Location) & ~0xFFFLL) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
|
||||
|
||||
@@ -102,23 +121,28 @@ public:
|
||||
// Now even an add
|
||||
add(ARMEmitter::Size::i64Bit, rd, rd, AlignedOffset);
|
||||
}
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unscaled offset too large");
|
||||
FEX_UNREACHABLE;
|
||||
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
void LongAddressGen(ARMEmitter::Register rd, ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded LongAddressGen(ARMEmitter::Register rd, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::LONG_ADDRESS_GEN});
|
||||
// Emit a register index and a nop. These will be backpatched.
|
||||
dc32(rd.Idx());
|
||||
nop();
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void LongAddressGen(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded LongAddressGen(ARMEmitter::Register rd, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
LongAddressGen(rd, &Label->Backward);
|
||||
return LongAddressGen(rd, &Label->Backward);
|
||||
} else {
|
||||
LongAddressGen(rd, &Label->Forward);
|
||||
return LongAddressGen(rd, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -174,7 +198,7 @@ public:
|
||||
// Logical immediate
|
||||
void and_(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint64_t Imm) {
|
||||
uint32_t n, immr, imms;
|
||||
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
|
||||
const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
|
||||
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
|
||||
and_(s, rd, rn, n, immr, imms);
|
||||
}
|
||||
@@ -185,7 +209,7 @@ public:
|
||||
|
||||
void ands(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint64_t Imm) {
|
||||
uint32_t n, immr, imms;
|
||||
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
|
||||
const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
|
||||
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
|
||||
ands(s, rd, rn, n, immr, imms);
|
||||
}
|
||||
@@ -196,14 +220,14 @@ public:
|
||||
|
||||
void orr(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint64_t Imm) {
|
||||
uint32_t n, immr, imms;
|
||||
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
|
||||
const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
|
||||
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
|
||||
orr(s, rd, rn, n, immr, imms);
|
||||
}
|
||||
|
||||
void eor(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint64_t Imm) {
|
||||
uint32_t n, immr, imms;
|
||||
[[maybe_unused]] const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
|
||||
const auto IsImm = IsImmLogical(Imm, RegSizeInBits(s), &n, &imms, &immr);
|
||||
LOGMAN_THROW_A_FMT(IsImm, "Couldn't encode immediate to logical op");
|
||||
eor(s, rd, rn, n, immr, imms);
|
||||
}
|
||||
@@ -333,7 +357,7 @@ public:
|
||||
bfi(s, rd, Reg::zr, lsb, width);
|
||||
}
|
||||
void bfxil(ARMEmitter::Size s, Register rd, Register rn, uint32_t lsb, uint32_t width) {
|
||||
[[maybe_unused]] const auto reg_size_bits = RegSizeInBits(s);
|
||||
const auto reg_size_bits = RegSizeInBits(s);
|
||||
const auto lsb_p_width = lsb + width;
|
||||
|
||||
LOGMAN_THROW_A_FMT(width >= 1, "bfxil needs width >= 1");
|
||||
@@ -862,12 +886,6 @@ public:
|
||||
}
|
||||
|
||||
private:
|
||||
static constexpr Condition InvertCondition(Condition cond) {
|
||||
// These behave as always, so it makes no sense to allow inverting these.
|
||||
LOGMAN_THROW_A_FMT(cond != Condition::CC_AL && cond != Condition::CC_NV, "Cannot invert CC_AL or CC_NV");
|
||||
return static_cast<Condition>(FEXCore::ToUnderlying(cond) ^ 1);
|
||||
}
|
||||
|
||||
void and_(ARMEmitter::Size s, ARMEmitter::Register rd, ARMEmitter::Register rn, uint32_t n, uint32_t immr, uint32_t imms) {
|
||||
constexpr uint32_t Op = 0b001'0010'00 << 22;
|
||||
DataProcessing_Logical_Imm(Op, s, rd, rn, n, immr, imms);
|
||||
@@ -977,7 +995,7 @@ private:
|
||||
}
|
||||
|
||||
void xbfiz_helper(bool is_signed, ARMEmitter::Size s, Register rd, Register rn, uint32_t lsb, uint32_t width) {
|
||||
[[maybe_unused]] const auto lsb_p_width = lsb + width;
|
||||
const auto lsb_p_width = lsb + width;
|
||||
const auto reg_size_bits = RegSizeInBits(s);
|
||||
|
||||
LOGMAN_THROW_A_FMT(lsb_p_width <= reg_size_bits, "lsb + width ({}) must be <= {}. lsb={}, width={}", lsb_p_width, reg_size_bits, lsb, width);
|
||||
|
||||
@@ -2244,8 +2244,7 @@ public:
|
||||
|
||||
template<IsQOrDRegister T>
|
||||
void movi(SubRegSize size, T rd, uint64_t Imm, uint16_t Shift = 0) {
|
||||
LOGMAN_THROW_A_FMT(size == SubRegSize::i8Bit || size == SubRegSize::i16Bit || size == SubRegSize::i32Bit ||
|
||||
size == SubRegSize::i64Bit,
|
||||
LOGMAN_THROW_A_FMT(size == SubRegSize::i8Bit || size == SubRegSize::i16Bit || size == SubRegSize::i32Bit || size == SubRegSize::i64Bit,
|
||||
"Unsupported movi size");
|
||||
|
||||
uint32_t cmode;
|
||||
|
||||
@@ -20,23 +20,31 @@ public:
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, Imm);
|
||||
}
|
||||
void b(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded b(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, Imm >> 2);
|
||||
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) [[likely]] {
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
void b(ARMEmitter::Condition Cond, ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded b(ARMEmitter::Condition Cond, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void b(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded b(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
b(Cond, &Label->Backward);
|
||||
return b(Cond, &Label->Backward);
|
||||
} else {
|
||||
b(Cond, &Label->Forward);
|
||||
return b(Cond, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -45,24 +53,32 @@ public:
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, Imm);
|
||||
}
|
||||
void bc(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded bc(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, Imm >> 2);
|
||||
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) [[likely]] {
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
|
||||
void bc(ARMEmitter::Condition Cond, ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded bc(ARMEmitter::Condition Cond, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void bc(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded bc(ARMEmitter::Condition Cond, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
bc(Cond, &Label->Backward);
|
||||
return bc(Cond, &Label->Backward);
|
||||
} else {
|
||||
bc(Cond, &Label->Forward);
|
||||
return bc(Cond, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -98,25 +114,32 @@ public:
|
||||
|
||||
UnconditionalBranch(Op, Imm);
|
||||
}
|
||||
void b(const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded b(const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b0001'01 << 26;
|
||||
if (Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0)) [[likely]] {
|
||||
constexpr uint32_t Op = 0b0001'01 << 26;
|
||||
UnconditionalBranch(Op, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
UnconditionalBranch(Op, Imm >> 2);
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
void b(ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded b(ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::B});
|
||||
constexpr uint32_t Op = 0b0001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void b(BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded b(BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
b(&Label->Backward);
|
||||
return b(&Label->Backward);
|
||||
} else {
|
||||
b(&Label->Forward);
|
||||
return b(&Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -126,25 +149,33 @@ public:
|
||||
UnconditionalBranch(Op, Imm);
|
||||
}
|
||||
|
||||
void bl(const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded bl(const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
constexpr uint32_t Op = 0b1001'01 << 26;
|
||||
if (Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0)) [[likely]] {
|
||||
constexpr uint32_t Op = 0b1001'01 << 26;
|
||||
UnconditionalBranch(Op, Imm >> 2);
|
||||
|
||||
UnconditionalBranch(Op, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
void bl(ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded bl(ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::B});
|
||||
constexpr uint32_t Op = 0b1001'01 << 26;
|
||||
|
||||
UnconditionalBranch(Op, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void bl(BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded bl(BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
bl(&Label->Backward);
|
||||
return bl(&Label->Backward);
|
||||
} else {
|
||||
bl(&Label->Forward);
|
||||
return bl(&Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -155,28 +186,35 @@ public:
|
||||
CompareAndBranch(Op, s, rt, Imm);
|
||||
}
|
||||
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded cbz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) [[likely]] {
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
CompareAndBranch(Op, s, rt, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
CompareAndBranch(Op, s, rt, Imm >> 2);
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded cbz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void cbz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded cbz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
cbz(s, rt, &Label->Backward);
|
||||
return cbz(s, rt, &Label->Backward);
|
||||
} else {
|
||||
cbz(s, rt, &Label->Forward);
|
||||
return cbz(s, rt, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -186,28 +224,35 @@ public:
|
||||
CompareAndBranch(Op, s, rt, Imm);
|
||||
}
|
||||
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) [[likely]] {
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
CompareAndBranch(Op, s, rt, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
CompareAndBranch(Op, s, rt, Imm >> 2);
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::BC});
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
|
||||
CompareAndBranch(Op, s, rt, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
cbnz(s, rt, &Label->Backward);
|
||||
return cbnz(s, rt, &Label->Backward);
|
||||
} else {
|
||||
cbnz(s, rt, &Label->Forward);
|
||||
return cbnz(s, rt, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -217,28 +262,35 @@ public:
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm);
|
||||
}
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded tbz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
if (Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0)) [[likely]] {
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
TestAndBranch(Op, rt, Bit, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm >> 2);
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded tbz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::TEST_BRANCH});
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void tbz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded tbz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
tbz(rt, Bit, &Label->Backward);
|
||||
return tbz(rt, Bit, &Label->Backward);
|
||||
} else {
|
||||
tbz(rt, Bit, &Label->Forward);
|
||||
return tbz(rt, Bit, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -247,27 +299,35 @@ public:
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm);
|
||||
}
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded tbnz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
if (Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0)) [[likely]] {
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
TestAndBranch(Op, rt, Bit, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
TestAndBranch(Op, rt, Bit, Imm >> 2);
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
}
|
||||
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded tbnz(ARMEmitter::Register rt, uint32_t Bit, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::TEST_BRANCH});
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
|
||||
TestAndBranch(Op, rt, Bit, 0);
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
void tbnz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] BranchEncodeSucceeded tbnz(ARMEmitter::Register rt, uint32_t Bit, BiDirectionalLabel* Label) {
|
||||
if (Label->Backward.Location) {
|
||||
tbnz(rt, Bit, &Label->Backward);
|
||||
return tbnz(rt, Bit, &Label->Backward);
|
||||
} else {
|
||||
tbnz(rt, Bit, &Label->Forward);
|
||||
return tbnz(rt, Bit, &Label->Forward);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -12,6 +12,7 @@
|
||||
#include <CodeEmitter/Registers.h>
|
||||
|
||||
#include <array>
|
||||
#include <bit>
|
||||
#include <cstdint>
|
||||
#include <utility>
|
||||
#include <type_traits>
|
||||
@@ -585,6 +586,11 @@ concept IsXOrWRegister = std::is_same_v<T, XRegister> || std::is_same_v<T, WRegi
|
||||
template<typename T>
|
||||
concept IsQOrDRegister = std::is_same_v<T, QRegister> || std::is_same_v<T, DRegister>;
|
||||
|
||||
enum class BranchEncodeSucceeded {
|
||||
Success,
|
||||
Failure,
|
||||
};
|
||||
|
||||
// Whether or not a given set of vector registers are sequential
|
||||
// in increasing order as far as the register file is concerned (modulo its size)
|
||||
//
|
||||
@@ -637,19 +643,25 @@ public:
|
||||
|
||||
// Bind a backward label to an address.
|
||||
// Address that is bound is the current emitter location.
|
||||
void Bind(BackwardLabel* Label) {
|
||||
[[nodiscard]] bool Bind(BackwardLabel* Label) {
|
||||
LOGMAN_THROW_A_FMT(Label->Location == nullptr, "Trying to bind a label twice");
|
||||
Label->Location = GetCursorAddress<uint8_t*>();
|
||||
|
||||
// Always binds because it is only storing a location.
|
||||
return true;
|
||||
}
|
||||
|
||||
void Bind(const ForwardLabel::Reference* Label) {
|
||||
[[nodiscard]] bool Bind(const ForwardLabel::Reference* Label) {
|
||||
uint8_t* CurrentAddress = GetCursorAddress<uint8_t*>();
|
||||
// Patch up the instructions
|
||||
switch (Label->Type) {
|
||||
case ForwardLabel::InstType::ADR: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
|
||||
if (!IsADRRange(Imm)) [[unlikely]] {
|
||||
// Can't bind.
|
||||
return false;
|
||||
}
|
||||
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
|
||||
uint32_t Inst = *Instruction & ~InstMask;
|
||||
@@ -661,7 +673,12 @@ public:
|
||||
case ForwardLabel::InstType::ADRP: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
|
||||
|
||||
if (!(IsADRPRange(Imm) && IsADRPAligned(Imm))) [[unlikely]] {
|
||||
// Can't bind.
|
||||
return false;
|
||||
}
|
||||
|
||||
Imm >>= 12;
|
||||
uint32_t InstMask = 0b11 << 29 | 0b1111'1111'1111'1111'111 << 5;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & 0x3F'FFFF;
|
||||
@@ -671,11 +688,13 @@ public:
|
||||
*Instruction = Inst;
|
||||
break;
|
||||
}
|
||||
|
||||
case ForwardLabel::InstType::B: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
if (!(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0))) [[unlikely]] {
|
||||
// Can't bind.
|
||||
return false;
|
||||
}
|
||||
Imm >>= 2;
|
||||
uint32_t InstMask = 0x3FF'FFFF;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
|
||||
@@ -685,11 +704,13 @@ public:
|
||||
|
||||
break;
|
||||
}
|
||||
|
||||
case ForwardLabel::InstType::TEST_BRANCH: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
if (!(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0))) [[unlikely]] {
|
||||
// Can't bind.
|
||||
return false;
|
||||
}
|
||||
Imm >>= 2;
|
||||
uint32_t InstMask = 0x3FFF;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
|
||||
@@ -703,7 +724,10 @@ public:
|
||||
case ForwardLabel::InstType::RELATIVE_LOAD: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
LOGMAN_THROW_A_FMT(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
if (!(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0))) [[unlikely]] {
|
||||
// Can't bind.
|
||||
return false;
|
||||
}
|
||||
Imm >>= 2;
|
||||
uint32_t InstMask = 0x7'FFFF;
|
||||
uint32_t Offset = static_cast<uint32_t>(Imm) & InstMask;
|
||||
@@ -752,27 +776,41 @@ public:
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unexpected inst type in label fixup");
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
// Bind a forward label to a location.
|
||||
// This walks all the instructions in the label's vector.
|
||||
// Then backpatching all instructions that have used the label.
|
||||
void Bind(ForwardLabel* Label) {
|
||||
[[nodiscard]] bool Bind(ForwardLabel* Label) {
|
||||
bool Bound = true;
|
||||
if (Label->FirstInst.Location) {
|
||||
Bind(&Label->FirstInst);
|
||||
Bound &= Bind(&Label->FirstInst);
|
||||
}
|
||||
for (auto& Inst : Label->Insts) {
|
||||
Bind(&Inst);
|
||||
Bound &= Bind(&Inst);
|
||||
}
|
||||
|
||||
return Bound;
|
||||
}
|
||||
|
||||
// Bind a bidirectional location to a location.
|
||||
// Binds both forwards and backwards depending on how the label was used.
|
||||
void Bind(BiDirectionalLabel* Label) {
|
||||
[[nodiscard]] bool Bind(BiDirectionalLabel* Label) {
|
||||
bool Bound = true;
|
||||
if (!Label->Backward.Location) {
|
||||
Bind(&Label->Backward);
|
||||
Bound &= Bind(&Label->Backward);
|
||||
}
|
||||
Bind(&Label->Forward);
|
||||
Bound &= Bind(&Label->Forward);
|
||||
|
||||
return Bound;
|
||||
}
|
||||
|
||||
static constexpr Condition InvertCondition(Condition cond) {
|
||||
// These behave as always, so it makes no sense to allow inverting these.
|
||||
LOGMAN_THROW_A_FMT(cond != Condition::CC_AL && cond != Condition::CC_NV, "Cannot invert CC_AL or CC_NV");
|
||||
return static_cast<Condition>(FEXCore::ToUnderlying(cond) ^ 1);
|
||||
}
|
||||
|
||||
#include <CodeEmitter/VixlUtils.inl>
|
||||
|
||||
@@ -1541,7 +1541,7 @@ public:
|
||||
void sqincp(SubRegSize size, XRegister rdn, PRegister pm) {
|
||||
SVEIncDecPredicateCountScalar(0, 1, 0b10, 0b00, size, rdn, pm);
|
||||
}
|
||||
void sqincp(SubRegSize size, XRegister rdn, PRegister pm, [[maybe_unused]] WRegister wn) {
|
||||
void sqincp(SubRegSize size, XRegister rdn, PRegister pm, WRegister wn) {
|
||||
LOGMAN_THROW_A_FMT(rdn.Idx() == wn.Idx(), "rdn and wn must be the same");
|
||||
SVEIncDecPredicateCountScalar(0, 1, 0b00, 0b00, size, rdn, pm);
|
||||
}
|
||||
@@ -1554,7 +1554,7 @@ public:
|
||||
void sqdecp(SubRegSize size, XRegister rdn, PRegister pm) {
|
||||
SVEIncDecPredicateCountScalar(0, 1, 0b10, 0b10, size, rdn, pm);
|
||||
}
|
||||
void sqdecp(SubRegSize size, XRegister rdn, PRegister pm, [[maybe_unused]] WRegister wn) {
|
||||
void sqdecp(SubRegSize size, XRegister rdn, PRegister pm, WRegister wn) {
|
||||
LOGMAN_THROW_A_FMT(rdn.Idx() == wn.Idx(), "rdn and wn must be the same");
|
||||
SVEIncDecPredicateCountScalar(0, 1, 0b00, 0b10, size, rdn, pm);
|
||||
}
|
||||
@@ -3296,7 +3296,7 @@ private:
|
||||
const auto log2_size_bytes = FEXCore::ilog2(size_bytes);
|
||||
|
||||
// We can index up to 512-bit registers with dup
|
||||
[[maybe_unused]] const auto max_index = (64U >> log2_size_bytes) - 1;
|
||||
const auto max_index = (64U >> log2_size_bytes) - 1;
|
||||
LOGMAN_THROW_A_FMT(Index <= max_index, "dup index ({}) too large. Must be within [0, {}].", Index, max_index);
|
||||
|
||||
// imm2:tsz make up a 7 bit wide field, with each increasing element size
|
||||
@@ -3326,7 +3326,7 @@ private:
|
||||
|
||||
uint32_t shift = 0;
|
||||
if (!is_uint8_imm) {
|
||||
[[maybe_unused]] const bool is_uint16_imm = (imm >> 16) == 0;
|
||||
const bool is_uint16_imm = (imm >> 16) == 0;
|
||||
|
||||
LOGMAN_THROW_A_FMT(is_uint16_imm, "Immediate ({}) must be a 16-bit value within [256, 65280]", imm);
|
||||
LOGMAN_THROW_A_FMT((imm % 256) == 0, "Immediate ({}) must be a multiple of 256", imm);
|
||||
@@ -4152,7 +4152,7 @@ private:
|
||||
|
||||
const auto& op_data = mem_op.MetaType.ScalarVectorType;
|
||||
const bool is_scaled = op_data.scale != 0;
|
||||
[[maybe_unused]] const auto msize_value = FEXCore::ToUnderlying(msize);
|
||||
const auto msize_value = FEXCore::ToUnderlying(msize);
|
||||
|
||||
LOGMAN_THROW_A_FMT(op_data.scale == 0 || op_data.scale == msize_value, "scale may only be 0 or {}", msize_value);
|
||||
|
||||
@@ -4266,7 +4266,7 @@ private:
|
||||
const auto msize_value = FEXCore::ToUnderlying(msize);
|
||||
const auto msize_bytes = 1U << msize_value;
|
||||
|
||||
[[maybe_unused]] const auto imm_limit = (32U << msize_value) - msize_bytes;
|
||||
const auto imm_limit = (32U << msize_value) - msize_bytes;
|
||||
const auto imm = mem_op.MetaType.VectorImmType.Imm;
|
||||
const auto imm_to_encode = imm >> msize_value;
|
||||
|
||||
@@ -4332,8 +4332,8 @@ private:
|
||||
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
|
||||
LOGMAN_THROW_A_FMT((imm % num_regs) == 0, "Offset must be a multiple of {}", num_regs);
|
||||
|
||||
[[maybe_unused]] const auto min_offset = -8 * num_regs;
|
||||
[[maybe_unused]] const auto max_offset = 7 * num_regs;
|
||||
const auto min_offset = -8 * num_regs;
|
||||
const auto max_offset = 7 * num_regs;
|
||||
LOGMAN_THROW_A_FMT(imm >= min_offset && imm <= max_offset,
|
||||
"Invalid load/store offset ({}). Offset must be a multiple of {} and be within [{}, {}]", imm, num_regs, min_offset,
|
||||
max_offset);
|
||||
@@ -4440,8 +4440,8 @@ private:
|
||||
LOGMAN_THROW_A_FMT(pg <= PReg::p7, "Can only use p0-p7 as a governing predicate");
|
||||
|
||||
const auto esize = static_cast<int>(16 << ssz);
|
||||
[[maybe_unused]] const auto max_imm = (esize << 3) - esize;
|
||||
[[maybe_unused]] const auto min_imm = -(max_imm + esize);
|
||||
const auto max_imm = (esize << 3) - esize;
|
||||
const auto min_imm = -(max_imm + esize);
|
||||
|
||||
LOGMAN_THROW_A_FMT((imm % esize) == 0, "imm ({}) must be a multiple of {}", imm, esize);
|
||||
LOGMAN_THROW_A_FMT(imm >= min_imm && imm <= max_imm, "imm ({}) must be within [{}, {}]", imm, min_imm, max_imm);
|
||||
@@ -4485,7 +4485,7 @@ private:
|
||||
const auto msize_value = FEXCore::ToUnderlying(msize);
|
||||
|
||||
const auto data_size_bytes = 1U << msize_value;
|
||||
[[maybe_unused]] const auto max_imm = (64U << msize_value) - data_size_bytes;
|
||||
const auto max_imm = (64U << msize_value) - data_size_bytes;
|
||||
LOGMAN_THROW_A_FMT((imm % data_size_bytes) == 0 && imm <= max_imm, "imm must be a multiple of {} and be within [0, {}]",
|
||||
data_size_bytes, max_imm);
|
||||
|
||||
@@ -4861,7 +4861,7 @@ private:
|
||||
"64-bit variants may only use Zm between z0-z15");
|
||||
|
||||
const auto Underlying = FEXCore::ToUnderlying(size);
|
||||
[[maybe_unused]] const uint32_t IndexMax = (16 / (1U << Underlying)) - 1;
|
||||
const uint32_t IndexMax = (16 / (1U << Underlying)) - 1;
|
||||
LOGMAN_THROW_A_FMT(index <= IndexMax, "Index must be within 0-{}", IndexMax);
|
||||
|
||||
// Can be bit 20 or 19 depending on whether or not the element size is 64-bit.
|
||||
@@ -5117,12 +5117,13 @@ private:
|
||||
requires (std::is_same_v<T, float> || std::is_same_v<T, double>)
|
||||
using FloatToEquivalentUInt = std::conditional_t<std::is_same_v<T, float>, uint32_t, uint64_t>;
|
||||
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
// Determines if a floating-point value is capable of being converted
|
||||
// into an 8-bit immediate. See pseudocode definition of VFPExpandImm
|
||||
// in ARM A-profile reference manual for a general overview of how this was derived.
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, float> || std::is_same_v<T, double>)
|
||||
[[nodiscard, maybe_unused]]
|
||||
[[nodiscard]]
|
||||
static bool IsValidFPValueForImm8(T value) {
|
||||
const uint64_t bits = FEXCore::BitCast<FloatToEquivalentUInt<T>>(value);
|
||||
const uint64_t datasize_idx = FEXCore::ilog2(sizeof(T)) - 1;
|
||||
@@ -5162,10 +5163,13 @@ private:
|
||||
|
||||
return true;
|
||||
}
|
||||
#endif
|
||||
|
||||
protected:
|
||||
static uint32_t FP32ToImm8(float value) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
LOGMAN_THROW_A_FMT(IsValidFPValueForImm8(value), "Value ({}) cannot be encoded into an 8-bit immediate", value);
|
||||
#endif
|
||||
|
||||
const auto bits = FEXCore::BitCast<uint32_t>(value);
|
||||
const auto sign = (bits & 0x80000000) >> 24;
|
||||
@@ -5176,7 +5180,9 @@ protected:
|
||||
}
|
||||
|
||||
static uint32_t FP64ToImm8(double value) {
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
LOGMAN_THROW_A_FMT(IsValidFPValueForImm8(value), "Value ({}) cannot be encoded into an 8-bit immediate", value);
|
||||
#endif
|
||||
|
||||
const auto bits = FEXCore::BitCast<uint64_t>(value);
|
||||
const auto sign = (bits & 0x80000000'00000000) >> 56;
|
||||
@@ -5202,7 +5208,7 @@ private:
|
||||
uint32_t shift = 0;
|
||||
if (!is_int8_imm) {
|
||||
const int32_t imm16_limit = 32768;
|
||||
[[maybe_unused]] const bool is_int16_imm = -imm16_limit <= imm && imm < imm16_limit;
|
||||
const bool is_int16_imm = -imm16_limit <= imm && imm < imm16_limit;
|
||||
|
||||
LOGMAN_THROW_A_FMT(is_int16_imm, "Immediate ({}) must be a 16-bit value within [-32768, 32512]", imm);
|
||||
LOGMAN_THROW_A_FMT((imm % 256) == 0, "Immediate ({}) must be a multiple of 256", imm);
|
||||
|
||||
@@ -30,7 +30,7 @@ public:
|
||||
const uint32_t SizeImm = FEXCore::ToUnderlying(size);
|
||||
const uint32_t IndexShift = SizeImm + 1;
|
||||
const uint32_t ElementSize = 1U << SizeImm;
|
||||
[[maybe_unused]] const uint32_t MaxIndex = 128U / (ElementSize * 8);
|
||||
const uint32_t MaxIndex = 128U / (ElementSize * 8);
|
||||
|
||||
LOGMAN_THROW_A_FMT(Index < MaxIndex, "Index too large. Index={}, Max Index: {}", Index, MaxIndex);
|
||||
|
||||
@@ -1381,7 +1381,7 @@ private:
|
||||
void ASIMDScalarXIndexedElement(uint32_t U, ScalarRegSize size, uint32_t opcode, VRegister rm, VRegister rn, VRegister rd, uint32_t index) {
|
||||
LOGMAN_THROW_A_FMT(size != ScalarRegSize::i8Bit, "Scalar size must not be 8-bit");
|
||||
|
||||
[[maybe_unused]] const auto invalid_bound = 16U >> FEXCore::ToUnderlying(size);
|
||||
const auto invalid_bound = 16U >> FEXCore::ToUnderlying(size);
|
||||
LOGMAN_THROW_A_FMT(index < invalid_bound, "Index ({}) must be within [0-{}]", index, invalid_bound - 1);
|
||||
|
||||
uint32_t Instr = 0b0101'1111'0000'0000'0000'0000'0000'0000;
|
||||
|
||||
@@ -41,7 +41,6 @@ static bool IsImmLogical(uint64_t value, unsigned width, unsigned* n = nullptr,
|
||||
[[maybe_unused]] constexpr auto kDRegSize = 64;
|
||||
|
||||
constexpr auto kWRegSize = 32;
|
||||
constexpr auto kXRegSize = 64;
|
||||
|
||||
LOGMAN_THROW_A_FMT((width == kBRegSize) || (width == kHRegSize) || (width == kSRegSize) || (width == kDRegSize), "Unexpected imm size");
|
||||
|
||||
@@ -129,8 +128,8 @@ static bool IsImmLogical(uint64_t value, unsigned width, unsigned* n = nullptr,
|
||||
// Compute the repeat distance d, and set up a bitmask covering the basic
|
||||
// unit of repetition (i.e. a word with the bottom d bits set). Also, in all
|
||||
// of these cases the N bit of the output will be zero.
|
||||
clz_a = CountLeadingZeros(a, kXRegSize);
|
||||
int clz_c = CountLeadingZeros(c, kXRegSize);
|
||||
clz_a = std::countl_zero(a);
|
||||
int clz_c = std::countl_zero(c);
|
||||
d = clz_a - clz_c;
|
||||
mask = ((UINT64_C(1) << d) - 1);
|
||||
out_n = 0;
|
||||
@@ -151,7 +150,7 @@ static bool IsImmLogical(uint64_t value, unsigned width, unsigned* n = nullptr,
|
||||
// of set bits in our word, meaning that we have the trivial case of
|
||||
// d == 64 and only one 'repetition'. Set up all the same variables as in
|
||||
// the general case above, and set the N bit in the output.
|
||||
clz_a = CountLeadingZeros(a, kXRegSize);
|
||||
clz_a = std::countl_zero(a);
|
||||
d = 64;
|
||||
mask = ~UINT64_C(0);
|
||||
out_n = 1;
|
||||
@@ -159,7 +158,7 @@ static bool IsImmLogical(uint64_t value, unsigned width, unsigned* n = nullptr,
|
||||
}
|
||||
|
||||
// If the repeat period d is not a power of two, it can't be encoded.
|
||||
if (!IsPowerOf2(d)) {
|
||||
if (!std::has_single_bit(uint32_t(d))) {
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -179,7 +178,7 @@ static bool IsImmLogical(uint64_t value, unsigned width, unsigned* n = nullptr,
|
||||
static const uint64_t multipliers[] = {
|
||||
0x0000000000000001UL, 0x0000000100000001UL, 0x0001000100010001UL, 0x0101010101010101UL, 0x1111111111111111UL, 0x5555555555555555UL,
|
||||
};
|
||||
uint64_t multiplier = multipliers[CountLeadingZeros(d, kXRegSize) - 57];
|
||||
uint64_t multiplier = multipliers[std::countl_zero(uint64_t(d)) - 57];
|
||||
uint64_t candidate = (b - a) * multiplier;
|
||||
|
||||
if (value != candidate) {
|
||||
@@ -194,7 +193,7 @@ static bool IsImmLogical(uint64_t value, unsigned width, unsigned* n = nullptr,
|
||||
// Count the set bits in our basic stretch. The special case of clz(0) == -1
|
||||
// makes the answer come out right for stretches that reach the very top of
|
||||
// the word (e.g. numbers like 0xffffc00000000000).
|
||||
int clz_b = (b == 0) ? -1 : CountLeadingZeros(b, kXRegSize);
|
||||
int clz_b = (b == 0) ? -1 : std::countl_zero(b);
|
||||
int s = clz_a - clz_b;
|
||||
|
||||
// Decide how many bits to rotate right by, to put the low bit of that basic
|
||||
@@ -285,11 +284,6 @@ INT_1_TO_63_LIST(DECLARE_IS_UINT_N)
|
||||
|
||||
private:
|
||||
|
||||
template<typename V>
|
||||
static inline bool IsPowerOf2(V value) {
|
||||
return (value != 0) && ((value & (value - 1)) == 0);
|
||||
}
|
||||
|
||||
// Some compilers dislike negating unsigned integers,
|
||||
// so we provide an equivalent.
|
||||
template<typename T>
|
||||
@@ -302,50 +296,4 @@ static inline uint64_t LowestSetBit(uint64_t value) {
|
||||
return value & UnsignedNegate(value);
|
||||
}
|
||||
|
||||
template<typename V>
|
||||
static inline int CountLeadingZeros(V value, int width = (sizeof(V) * 8)) {
|
||||
#if COMPILER_HAS_BUILTIN_CLZ
|
||||
if (width == 32) {
|
||||
return (value == 0) ? 32 : __builtin_clz(static_cast<unsigned>(value));
|
||||
} else if (width == 64) {
|
||||
return (value == 0) ? 64 : __builtin_clzll(value);
|
||||
}
|
||||
#endif
|
||||
return CountLeadingZerosFallBack(value, width);
|
||||
}
|
||||
|
||||
static inline int CountLeadingZerosFallBack(uint64_t value, int width) {
|
||||
LOGMAN_THROW_A_FMT(IsPowerOf2(width) && (width <= 64), "Invalid width");
|
||||
if (value == 0) {
|
||||
return width;
|
||||
}
|
||||
int count = 0;
|
||||
value = value << (64 - width);
|
||||
if ((value & UINT64_C(0xffffffff00000000)) == 0) {
|
||||
count += 32;
|
||||
value = value << 32;
|
||||
}
|
||||
if ((value & UINT64_C(0xffff000000000000)) == 0) {
|
||||
count += 16;
|
||||
value = value << 16;
|
||||
}
|
||||
if ((value & UINT64_C(0xff00000000000000)) == 0) {
|
||||
count += 8;
|
||||
value = value << 8;
|
||||
}
|
||||
if ((value & UINT64_C(0xf000000000000000)) == 0) {
|
||||
count += 4;
|
||||
value = value << 4;
|
||||
}
|
||||
if ((value & UINT64_C(0xc000000000000000)) == 0) {
|
||||
count += 2;
|
||||
value = value << 2;
|
||||
}
|
||||
if ((value & UINT64_C(0x8000000000000000)) == 0) {
|
||||
count += 1;
|
||||
}
|
||||
count += (value == 0);
|
||||
return count;
|
||||
}
|
||||
|
||||
public:
|
||||
@@ -4,7 +4,8 @@ file(GLOB GEN_CONFIG_SOURCES CONFIGURE_DEPENDS *.json.in)
|
||||
# Any application configuration json file gets installed
|
||||
foreach(CONFIG_SRC ${CONFIG_SOURCES})
|
||||
install(FILES ${CONFIG_SRC}
|
||||
DESTINATION ${DATA_DIRECTORY}/AppConfig/)
|
||||
DESTINATION ${DATA_DIRECTORY}/AppConfig/
|
||||
COMPONENT Runtime)
|
||||
endforeach()
|
||||
|
||||
# Any configuration file json file that needs to be generated
|
||||
@@ -21,5 +22,6 @@ foreach(GEN_CONFIG_SRC ${GEN_CONFIG_SOURCES})
|
||||
# Then install the configured json
|
||||
install(
|
||||
FILES ${CMAKE_BINARY_DIR}/Data/AppConfig/${CONFIG_NAME}
|
||||
DESTINATION ${DATA_DIRECTORY}/AppConfig/)
|
||||
DESTINATION ${DATA_DIRECTORY}/AppConfig/
|
||||
COMPONENT Runtime)
|
||||
endforeach()
|
||||
@@ -1,3 +0,0 @@
|
||||
x86 and x86-64 Linux emulator
|
||||
|
||||
FEX allows you to run x86 applications on ARM64 Linux devices. It offers broad compatibility with both 32-bit and 64-bit binaries, and it can be used alongside Wine/Proton to play Windows games.
|
||||
@@ -1,18 +0,0 @@
|
||||
#!/bin/sh
|
||||
set -e
|
||||
update_binfmt() {
|
||||
# Check for update-binfmts
|
||||
command -v update-binfmts >/dev/null || return 0
|
||||
|
||||
# Setup binfmt_misc
|
||||
update-binfmts --import FEX-x86
|
||||
update-binfmts --import FEX-x86_64
|
||||
}
|
||||
|
||||
# Install FEXInterpreter hardlink
|
||||
# Needs to be done before setting up binfmt_misc
|
||||
ln -f /usr/bin/FEXLoader /usr/bin/FEXInterpreter
|
||||
|
||||
if [ $(uname -m) = 'aarch64' ]; then
|
||||
update_binfmt
|
||||
fi
|
||||
@@ -1,17 +0,0 @@
|
||||
#!/bin/sh
|
||||
set -e
|
||||
update_binfmt() {
|
||||
# Check for update-binfmts
|
||||
command -v update-binfmts >/dev/null || return 0
|
||||
|
||||
# Uninstall
|
||||
update-binfmts --unimport FEX-x86
|
||||
update-binfmts --unimport FEX-x86_64
|
||||
}
|
||||
|
||||
if [ $(uname -m) = 'aarch64' ]; then
|
||||
update_binfmt
|
||||
fi
|
||||
|
||||
# Remove FEXInterpreter hardlink
|
||||
unlink /usr/bin/FEXInterpreter
|
||||
@@ -1 +0,0 @@
|
||||
activate-noawait ldconfig
|
||||
+1
-1
@@ -14,7 +14,7 @@ RUN mkdir build
|
||||
|
||||
ARG CC=clang-13
|
||||
ARG CXX=clang++-13
|
||||
RUN cmake -DCMAKE_INSTALL_PREFIX=/usr -DCMAKE_BUILD_TYPE=Release -DUSE_LINKER=lld -DENABLE_LTO=True -DBUILD_TESTS=False -DENABLE_ASSERTIONS=False -G Ninja .
|
||||
RUN cmake -DCMAKE_INSTALL_PREFIX=/usr -DCMAKE_BUILD_TYPE=Release -DUSE_LINKER=lld -DENABLE_LTO=True -DBUILD_TESTING=False -DENABLE_ASSERTIONS=False -G Ninja .
|
||||
RUN ninja
|
||||
|
||||
WORKDIR /FEX/build
|
||||
|
||||
@@ -10,7 +10,8 @@ function(GenBinFmt Name)
|
||||
# Then install the configured binfmt
|
||||
install(
|
||||
FILES ${CMAKE_BINARY_DIR}/Data/binfmts/${FMT_NAME}
|
||||
DESTINATION ${CMAKE_INSTALL_PREFIX}/share/binfmts/)
|
||||
DESTINATION ${CMAKE_INSTALL_PREFIX}/share/binfmts/
|
||||
COMPONENT Runtime)
|
||||
endfunction()
|
||||
|
||||
if (NOT USE_LEGACY_BINFMTMISC)
|
||||
@@ -19,7 +20,8 @@ if (NOT USE_LEGACY_BINFMTMISC)
|
||||
|
||||
install(
|
||||
FILES ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86.conf ${CMAKE_BINARY_DIR}/Data/binfmts/FEX-x86_64.conf
|
||||
DESTINATION ${CMAKE_INSTALL_PREFIX}/lib/binfmt.d/)
|
||||
DESTINATION ${CMAKE_INSTALL_PREFIX}/lib/binfmt.d/
|
||||
COMPONENT Runtime)
|
||||
else()
|
||||
GenBinFmt(FEX-x86.in)
|
||||
GenBinFmt(FEX-x86_64.in)
|
||||
|
||||
@@ -1 +1 @@
|
||||
:FEX-x86:M:0:\x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter:POCF
|
||||
:FEX-x86:M:0:\x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEX:POCF
|
||||
@@ -1,5 +1,5 @@
|
||||
package fex
|
||||
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter
|
||||
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEX
|
||||
magic \x7fELF\x01\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x03\x00
|
||||
offset 0
|
||||
mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff
|
||||
|
||||
@@ -1 +1 @@
|
||||
:FEX-x86_64:M:0:\x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter:POCF
|
||||
:FEX-x86_64:M:0:\x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00:\xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff:@CMAKE_INSTALL_PREFIX@/bin/FEX:POCF
|
||||
@@ -1,5 +1,5 @@
|
||||
package fex
|
||||
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEXInterpreter
|
||||
interpreter @CMAKE_INSTALL_PREFIX@/bin/FEX
|
||||
magic \x7fELF\x02\x01\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x02\x00\x3e\x00
|
||||
offset 0
|
||||
mask \xff\xff\xff\xff\xff\xfe\xfe\x00\x00\x00\x00\xff\xff\xff\xff\xff\xfe\xff\xff\xff
|
||||
|
||||
@@ -45,7 +45,7 @@ pkgs.mkShell {
|
||||
fi
|
||||
'';
|
||||
|
||||
# E.g. cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTS=False
|
||||
# E.g. cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTING=False
|
||||
FEX_CMAKE_TOOLCHAIN_ARM64EC = "--toolchain ${cmakeToolchainFile} -DMINGW_TRIPLE=arm64ec-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows";
|
||||
FEX_CMAKE_TOOLCHAIN_WOW64 = "--toolchain ${cmakeToolchainFile} -DMINGW_TRIPLE=aarch64-w64-mingw32 -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-windows";
|
||||
FEX_MESON_CROSSFILE = "--cross-file ${mesonCrossFile}";
|
||||
|
||||
@@ -18,4 +18,4 @@ then
|
||||
fi
|
||||
|
||||
set -o xtrace
|
||||
cmake $FEX_CMAKE_TOOLCHAIN_WOW64 -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTS=False $@
|
||||
cmake $FEX_CMAKE_TOOLCHAIN_WOW64 -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTING=False $@
|
||||
@@ -18,4 +18,4 @@ then
|
||||
fi
|
||||
|
||||
set -o xtrace
|
||||
cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTS=False $@
|
||||
cmake $FEX_CMAKE_TOOLCHAIN_ARM64EC -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -DENABLE_LTO=False -DBUILD_TESTING=False $@
|
||||
@@ -14,4 +14,4 @@ fi
|
||||
rm -rf unittests/FEXLinuxTests
|
||||
|
||||
set -o xtrace
|
||||
cmake . $FEX_CMAKE_TOOLCHAINS -DBUILD_TESTS=ON -DBUILD_FEX_LINUX_TESTS=ON
|
||||
cmake . $FEX_CMAKE_TOOLCHAINS -DBUILD_TESTING=ON -DBUILD_FEX_LINUX_TESTS=ON
|
||||
@@ -214,7 +214,6 @@ class ClangFormatHelper(FormatHelper):
|
||||
self.clang_fmt_path,
|
||||
"--binary=clang-format-19",
|
||||
"--diff",
|
||||
"--diff_from_common_commit",
|
||||
]
|
||||
|
||||
if args.start_rev and args.end_rev:
|
||||
|
||||
+1
Submodule External/range-v3 added at ca1388fb9d.
Vendored
+1
-1
Submodule External/vixl updated: 84bc10c107...ed690c9eca.
@@ -78,6 +78,6 @@ install (DIRECTORY include/FEXCore ${CMAKE_BINARY_DIR}/include/FEXCore
|
||||
DESTINATION include
|
||||
COMPONENT Development)
|
||||
|
||||
if (BUILD_TESTS)
|
||||
if (BUILD_TESTING)
|
||||
add_subdirectory(unittests/)
|
||||
endif()
|
||||
@@ -118,41 +118,6 @@ def print_man_env_option(name, desc, default, no_json_key):
|
||||
output_man.write("\\fBdefault:\\fR {0}\n".format(default))
|
||||
output_man.write(".Pp\n\n")
|
||||
|
||||
def print_man_options(options):
|
||||
output_man.write(".Sh OPTIONS\n")
|
||||
output_man.write(".Bl -tag -width -indent\n")
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
short = None
|
||||
long = op_key.lower()
|
||||
|
||||
if ("ShortArg" in op_vals):
|
||||
short = op_vals["ShortArg"]
|
||||
|
||||
default = op_vals["Default"]
|
||||
value_type = op_vals["Type"]
|
||||
|
||||
# Textual default rather than enum based
|
||||
if ("TextDefault" in op_vals):
|
||||
default = op_vals["TextDefault"]
|
||||
|
||||
if (value_type == "str" or value_type == "strarray" or value_type == "strenum"):
|
||||
# Wrap the string argument in quotes
|
||||
default = "'" + default + "'"
|
||||
print_man_option(
|
||||
short,
|
||||
long,
|
||||
op_vals["Desc"],
|
||||
default
|
||||
)
|
||||
if (value_type == "strenum"):
|
||||
Enums = op_vals["Enums"]
|
||||
output_man.write("\\fBAvailable Options:\\fR\n")
|
||||
output_man.write(", ".join(f"{enum_op_val}" for [_, enum_op_val] in Enums.items()))
|
||||
output_man.write("\n.sp\n")
|
||||
|
||||
output_man.write(".El\n")
|
||||
|
||||
def print_man_environment(options):
|
||||
output_man.write(".Sh ENVIRONMENT\n")
|
||||
output_man.write(".Bl -tag -width -indent\n")
|
||||
@@ -194,7 +159,7 @@ def print_man_environment_tail():
|
||||
"By default FEX will look in {$HOME, $XDG_CONFIG_HOME}/.fex-emu/",
|
||||
"This will override the full path",
|
||||
"If FEX_PORTABLE is declared then relative paths are also supported",
|
||||
"For FEXInterpreter: Relative to the FEXInterpreter binary",
|
||||
"For FEX: Relative to the FEX binary",
|
||||
"For WINE: Relative to %LOCALAPPDATA%"
|
||||
],
|
||||
"''", True)
|
||||
@@ -208,7 +173,7 @@ def print_man_environment_tail():
|
||||
"One must be careful with this option as it will override any applications that load with execve as well"
|
||||
"If you need to support applications that execve then use FEX_APP_CONFIG_LOCATION instead"
|
||||
"If FEX_PORTABLE is declared then relative paths are also supported",
|
||||
"For FEXInterpreter: Relative to the FEXInterpreter binary",
|
||||
"For FEX: Relative to the FEX binary",
|
||||
"For WINE: Relative to %LOCALAPPDATA%"
|
||||
],
|
||||
"''", True)
|
||||
@@ -227,8 +192,8 @@ def print_man_environment_tail():
|
||||
"PORTABLE",
|
||||
[
|
||||
"Allows FEX to run without installation. Global locations for configuration and binfmt_misc are ignored.",
|
||||
"For FEXInterpreter on Linux:",
|
||||
"These files are instead read from <FEXInterpreterPath>/fex-emu/ by default.",
|
||||
"For FEX on Linux:",
|
||||
"These files are instead read from <FEXPath>/fex-emu/ by default.",
|
||||
"For Arm64ec/Wow64 WINE builds:",
|
||||
"These files are instead read from $LOCALAPPDATA/fex-emu/ by default.",
|
||||
"For further customization, see FEX_APP_CONFIG_LOCATION and FEX_APP_DATA_LOCATION."
|
||||
@@ -240,20 +205,12 @@ def print_man_header():
|
||||
.Dt FEX
|
||||
.Os Linux
|
||||
.Sh NAME
|
||||
.Nm FEXLoader
|
||||
.Nm FEXInterpreter
|
||||
.Nm FEX
|
||||
.Nm FEXBash
|
||||
.Nd Fast x86-64 and x86 emulation.
|
||||
.Sh SYNOPSIS
|
||||
.Nm
|
||||
.Op options
|
||||
.Op Ar --
|
||||
.Ar Application
|
||||
<args> ...
|
||||
.Pp
|
||||
.Nm FEXInterpreter
|
||||
.Ar Application
|
||||
<args> ...
|
||||
.Ar <args> ...
|
||||
.Pp
|
||||
.Nm FEXBash
|
||||
.Ar <args> ...
|
||||
@@ -361,82 +318,6 @@ def print_config_option(type, group_name, json_name, default_value, short, choic
|
||||
|
||||
output_argloader.write("\n");
|
||||
|
||||
def print_argloader_options(options):
|
||||
output_argloader.write("#ifdef BEFORE_PARSE\n")
|
||||
output_argloader.write("#undef BEFORE_PARSE\n")
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
default = op_vals["Default"]
|
||||
|
||||
if (op_vals["Type"] == "str" or op_vals["Type"] == "strarray" or op_vals["Type"] == "strenum"):
|
||||
# Wrap the string argument in quotes
|
||||
default = "\"" + default + "\""
|
||||
|
||||
# Textual default rather than enum based
|
||||
if ("TextDefault" in op_vals):
|
||||
default = "\"" + op_vals["TextDefault"] + "\""
|
||||
|
||||
short = None
|
||||
choices = None
|
||||
|
||||
if ("ShortArg" in op_vals):
|
||||
short = op_vals["ShortArg"]
|
||||
if ("Choices" in op_vals):
|
||||
choices = op_vals["Choices"]
|
||||
|
||||
print_config_option(
|
||||
op_vals["Type"],
|
||||
op_group,
|
||||
op_key,
|
||||
default,
|
||||
short,
|
||||
choices,
|
||||
op_vals["Desc"])
|
||||
|
||||
output_argloader.write("\n")
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
def print_parse_argloader_options(options):
|
||||
output_argloader.write("#ifdef AFTER_PARSE\n")
|
||||
output_argloader.write("#undef AFTER_PARSE\n")
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
output_argloader.write("if (Options.is_set_by_user(\"{0}\")) {{\n".format(op_key))
|
||||
|
||||
value_type = op_vals["Type"]
|
||||
NeedsString = False
|
||||
conversion_func = "fextl::fmt::format(\"{}\", "
|
||||
if ("ArgumentHandler" in op_vals):
|
||||
NeedsString = True
|
||||
conversion_func = "FEXCore::Config::Handler::{0}(".format(op_vals["ArgumentHandler"])
|
||||
if (value_type == "str"):
|
||||
NeedsString = True
|
||||
conversion_func = "std::move("
|
||||
if (value_type == "bool"):
|
||||
# boolean values need a decimal specifier. Otherwise fmt prints strings.
|
||||
conversion_func = "fextl::fmt::format(\"{:d}\", "
|
||||
|
||||
if (value_type == "strenum"):
|
||||
output_argloader.write("\tfextl::string UserValue = Options[\"{0}\"];\n".format(op_key))
|
||||
output_argloader.write("\tSet(FEXCore::Config::ConfigOption::CONFIG_{}, FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, UserValue));\n".format(op_key.upper(), op_key, op_key, op_key))
|
||||
elif (value_type == "strarray"):
|
||||
# these need a bit more help
|
||||
output_argloader.write("\tauto Array = Options.all(\"{0}\");\n".format(op_key))
|
||||
output_argloader.write("\tfor (auto iter = Array.begin(); iter != Array.end(); ++iter) {\n")
|
||||
output_argloader.write("\t\tAppendStrArrayValue(FEXCore::Config::ConfigOption::CONFIG_{0}, *iter);\n".format(op_key.upper()))
|
||||
output_argloader.write("\t}\n")
|
||||
else:
|
||||
if (NeedsString):
|
||||
output_argloader.write("\tfextl::string UserValue = Options[\"{0}\"];\n".format(op_key))
|
||||
else:
|
||||
output_argloader.write("\t{0} UserValue = Options.get(\"{1}\");\n".format(value_type, op_key))
|
||||
|
||||
output_argloader.write("\tSet(FEXCore::Config::ConfigOption::CONFIG_{0}, {1}UserValue));\n".format(op_key.upper(), conversion_func))
|
||||
output_argloader.write("}\n")
|
||||
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
|
||||
def print_parse_envloader_options(options):
|
||||
output_argloader.write("#ifdef ENVLOADER\n")
|
||||
output_argloader.write("#undef ENVLOADER\n")
|
||||
@@ -447,13 +328,13 @@ def print_parse_envloader_options(options):
|
||||
value_type = op_vals["Type"]
|
||||
if (value_type == "strenum"):
|
||||
output_argloader.write("else if (Key == \"FEX_{0}\") {{\n".format(op_key.upper()))
|
||||
output_argloader.write("Value = FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, Value_View);\n".format(op_key, op_key, op_key))
|
||||
output_argloader.write("\tValue = FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, Value_View);\n".format(op_key, op_key))
|
||||
output_argloader.write("}\n")
|
||||
|
||||
if ("ArgumentHandler" in op_vals):
|
||||
conversion_func = "FEXCore::Config::Handler::{0}".format(op_vals["ArgumentHandler"])
|
||||
output_argloader.write("else if (Key == \"FEX_{0}\") {{\n".format(op_key.upper()))
|
||||
output_argloader.write("Value = {0}(Value_View);\n".format(conversion_func))
|
||||
output_argloader.write("\tValue = {0}(Value_View);\n".format(conversion_func))
|
||||
output_argloader.write("}\n")
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
@@ -467,15 +348,15 @@ def print_parse_jsonloader_options(options):
|
||||
value_type = op_vals["Type"]
|
||||
if (value_type == "strenum"):
|
||||
output_argloader.write("else if (KeyName == \"{0}\") {{\n".format(op_key))
|
||||
output_argloader.write("\tSet(KeyOption, FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, Value_View));\n".format(op_key, op_key, op_key))
|
||||
output_argloader.write("\tSet(KeyOption, FEXCore::Config::EnumParser<FEXCore::Config::{}ConfigPair>(FEXCore::Config::{}_EnumPairs, Value_View));\n".format(op_key, op_key))
|
||||
output_argloader.write("}\n")
|
||||
elif (value_type == "strarray"):
|
||||
output_argloader.write("else if (KeyName == \"{0}\") {{\n".format(op_key))
|
||||
output_argloader.write("\tAppendStrArrayValue(KeyOption, ConfigString);\n")
|
||||
output_argloader.write("}\n")
|
||||
assert op_key is not None, "No options found in JSONLOADER"
|
||||
output_argloader.write("else {{\n".format(op_key))
|
||||
output_argloader.write("Set(KeyOption, ConfigString);\n")
|
||||
output_argloader.write("else {\n")
|
||||
output_argloader.write("\tSet(KeyOption, ConfigString);\n")
|
||||
output_argloader.write("}\n")
|
||||
|
||||
output_argloader.write("#endif\n")
|
||||
@@ -517,41 +398,6 @@ def print_parse_enum_options(options):
|
||||
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
def check_for_duplicate_options(options):
|
||||
short_map = []
|
||||
long_map = []
|
||||
|
||||
# Spin through all the items and see if we have a duplicate option
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
short = None
|
||||
long = op_key.lower()
|
||||
long_invert = None
|
||||
if ("ShortArg" in op_vals):
|
||||
short = op_vals["ShortArg"]
|
||||
if (op_vals["Type"] == "bool"):
|
||||
long_invert = "no-" + long
|
||||
|
||||
# Check for short key duplication
|
||||
if (short != None):
|
||||
if (short in short_map):
|
||||
raise Exception("Short config '{0}' for option '{1}' has duplicate entry!".format(short, op_key))
|
||||
else:
|
||||
short_map.append(short)
|
||||
|
||||
# Check for long key duplication
|
||||
if (long in long_map):
|
||||
raise Exception("Long config '{0}' has duplicate entry!".format(long))
|
||||
else:
|
||||
long_map.append(long)
|
||||
|
||||
# Check for long key duplication
|
||||
if (long_invert != None):
|
||||
if (long_invert in long_map):
|
||||
raise Exception("Long config '{0}' has duplicate entry!".format(long_invert))
|
||||
else:
|
||||
long_map.append(long_invert)
|
||||
|
||||
if (len(sys.argv) < 5):
|
||||
sys.exit()
|
||||
|
||||
@@ -568,8 +414,6 @@ json_object = json.loads(json_text)
|
||||
options = json_object["Options"]
|
||||
unnamed_options = json_object["UnnamedOptions"]
|
||||
|
||||
check_for_duplicate_options(options)
|
||||
|
||||
# Generate config include file
|
||||
output_file = open(output_filename, "w")
|
||||
print_header()
|
||||
@@ -581,7 +425,6 @@ output_file.close()
|
||||
# Generate man file
|
||||
output_man = open(output_man_page, "w")
|
||||
print_man_header()
|
||||
print_man_options(options)
|
||||
print_man_environment(options)
|
||||
print_man_tail()
|
||||
|
||||
@@ -589,8 +432,6 @@ output_man.close()
|
||||
|
||||
# Generate argument loader code
|
||||
output_argloader = open(output_argumentloader_filename, "w")
|
||||
print_argloader_options(options);
|
||||
print_parse_argloader_options(options);
|
||||
|
||||
# Generate environment loader code
|
||||
print_parse_envloader_options(options);
|
||||
|
||||
@@ -58,6 +58,7 @@ class OpDefinition:
|
||||
JITDispatch: bool
|
||||
JITDispatchOverride: str
|
||||
TiedSource: int
|
||||
Inline: list
|
||||
Arguments: list
|
||||
EmitValidation: list
|
||||
Desc: list
|
||||
@@ -278,6 +279,12 @@ def parse_ops(ops):
|
||||
if "TiedSource" in op_val:
|
||||
OpDef.TiedSource = op_val["TiedSource"]
|
||||
|
||||
# Pad Inline out to the argument count
|
||||
OpDef.Inline = [''] * len(OpDef.Arguments)
|
||||
if "Inline" in op_val:
|
||||
Value = op_val["Inline"]
|
||||
OpDef.Inline[0:len(Value)] = Value
|
||||
|
||||
# Do some fixups of the data here
|
||||
if len(OpDef.EmitValidation) != 0:
|
||||
for i in range(len(OpDef.EmitValidation)):
|
||||
@@ -397,16 +404,16 @@ def print_ir_sizes():
|
||||
// Make sure our array maps directly to the IROps enum
|
||||
static_assert(IRSizes[IROps::OP_LAST] == -1ULL);
|
||||
|
||||
[[maybe_unused, nodiscard]] static size_t GetSize(IROps Op) { return IRSizes[Op]; }
|
||||
[[nodiscard, gnu::const, gnu::visibility("default")]] std::string_view const& GetName(IROps Op);
|
||||
[[nodiscard, gnu::const, gnu::visibility("default")]] uint8_t GetArgs(IROps Op);
|
||||
[[nodiscard, gnu::const, gnu::visibility("default")]] uint8_t GetRAArgs(IROps Op);
|
||||
[[nodiscard, gnu::const, gnu::visibility("default")]] FEXCore::IR::RegisterClassType GetRegClass(IROps Op);
|
||||
[[nodiscard, gnu::const, gnu::visibility("default")]] bool HasSideEffects(IROps Op);
|
||||
[[nodiscard, gnu::const, gnu::visibility("default")]] bool ImplicitFlagClobber(IROps Op);
|
||||
[[nodiscard, gnu::const, gnu::visibility("default")]] bool GetHasDest(IROps Op);
|
||||
[[nodiscard, gnu::const, gnu::visibility("default")]] bool LoweredX87(IROps Op);
|
||||
[[nodiscard, gnu::const, gnu::visibility("default")]] int8_t TiedSource(IROps Op);
|
||||
[[nodiscard]] inline size_t GetSize(IROps Op) { return IRSizes[Op]; }
|
||||
[[nodiscard, gnu::const]] std::string_view const& GetName(IROps Op);
|
||||
[[nodiscard, gnu::const]] uint8_t GetArgs(IROps Op);
|
||||
[[nodiscard, gnu::const]] uint8_t GetRAArgs(IROps Op);
|
||||
[[nodiscard, gnu::const]] FEXCore::IR::RegisterClassType GetRegClass(IROps Op);
|
||||
[[nodiscard, gnu::const]] bool HasSideEffects(IROps Op);
|
||||
[[nodiscard, gnu::const]] bool ImplicitFlagClobber(IROps Op);
|
||||
[[nodiscard, gnu::const]] bool GetHasDest(IROps Op);
|
||||
[[nodiscard, gnu::const]] bool LoweredX87(IROps Op);
|
||||
[[nodiscard, gnu::const]] int8_t TiedSource(IROps Op);
|
||||
|
||||
#undef IROP_SIZES
|
||||
#endif
|
||||
@@ -588,13 +595,13 @@ def print_ir_arg_printer():
|
||||
output_file.write("#endif\n")
|
||||
|
||||
def print_validation(op):
|
||||
if op.EmitValidation != None:
|
||||
output_file.write("\t\t#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED\n")
|
||||
if len(op.EmitValidation) != 0:
|
||||
output_file.write("#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED\n")
|
||||
|
||||
for Validation in op.EmitValidation:
|
||||
Sanitized = Validation.replace("\"", "\\\"")
|
||||
output_file.write("\tLOGMAN_THROW_A_FMT({}, \"{}\");\n".format(Validation, Sanitized))
|
||||
output_file.write("\t\t#endif\n")
|
||||
output_file.write("\t\tLOGMAN_THROW_A_FMT({}, \"{}\");\n".format(Validation, Sanitized))
|
||||
output_file.write("#endif\n")
|
||||
|
||||
# Print out IR allocator helpers
|
||||
def print_ir_allocator_helpers():
|
||||
@@ -773,9 +780,29 @@ def print_ir_allocator_helpers():
|
||||
output_file.write(") {\n")
|
||||
output_file.write("\t\tauto ListDataBegin = DualListData.ListBegin();\n")
|
||||
|
||||
idx = 0
|
||||
for arg in op.Arguments:
|
||||
if arg.IsSSA:
|
||||
output_file.write("\t\t{}->AddUse();\n".format(arg.Name))
|
||||
# Inline an immediate if we can
|
||||
inline = op.Inline[idx]
|
||||
idx += 1
|
||||
|
||||
if inline != '':
|
||||
Sized = "Size" in [x.Name for x in op.Arguments]
|
||||
P = ["Size" if Sized else "OpSize::i64Bit", arg.Name]
|
||||
|
||||
# A few cases need extra info plumbed.
|
||||
if inline == "SubtractZero":
|
||||
P += ["Src2"]
|
||||
elif inline == "Mem":
|
||||
P += ["OffsetType", "OffsetScale"]
|
||||
elif inline == "Memtso":
|
||||
P += ["OffsetType", "OffsetScale", "true /* TSO */"]
|
||||
inline = "Mem"
|
||||
|
||||
output_file.write(f"\t\t{arg.Name} = Inline{inline}({', '.join(P)});\n")
|
||||
|
||||
output_file.write(f"\t\t{arg.Name}->AddUse();\n")
|
||||
|
||||
# Insert validation here. This is skipped for the
|
||||
# OrderedNodeWrapper version because validation can depend on
|
||||
|
||||
@@ -18,14 +18,12 @@ set (SRCS
|
||||
Common/JitSymbols.cpp
|
||||
Interface/Context/Context.cpp
|
||||
Interface/Core/LookupCache.cpp
|
||||
Interface/Core/CodeCache.cpp
|
||||
Interface/Core/Core.cpp
|
||||
Interface/Core/CPUBackend.cpp
|
||||
Interface/Core/Addressing.cpp
|
||||
Interface/Core/CPUID.cpp
|
||||
Interface/Core/Frontend.cpp
|
||||
Interface/Core/ObjectCache/JobHandling.cpp
|
||||
Interface/Core/ObjectCache/NamedRegionObjectHandler.cpp
|
||||
Interface/Core/ObjectCache/ObjectCacheService.cpp
|
||||
Interface/Core/OpcodeDispatcher/AVX_128.cpp
|
||||
Interface/Core/OpcodeDispatcher/Crypto.cpp
|
||||
Interface/Core/OpcodeDispatcher/Flags.cpp
|
||||
@@ -33,7 +31,6 @@ set (SRCS
|
||||
Interface/Core/OpcodeDispatcher/X87.cpp
|
||||
Interface/Core/OpcodeDispatcher/X87F64.cpp
|
||||
Interface/Core/OpcodeDispatcher.cpp
|
||||
Interface/Core/X86Tables.cpp
|
||||
Interface/Core/X86HelperGen.cpp
|
||||
Interface/Core/ArchHelpers/Arm64Emitter.cpp
|
||||
Interface/Core/Dispatcher/Dispatcher.cpp
|
||||
@@ -61,16 +58,15 @@ set (SRCS
|
||||
Interface/Core/X86Tables/VEXTables.cpp
|
||||
Interface/Core/X86Tables/X87Tables.cpp
|
||||
Interface/GDBJIT/GDBJIT.cpp
|
||||
Interface/IR/AOTIR.cpp
|
||||
Interface/IR/IRDumper.cpp
|
||||
Interface/IR/IREmitter.cpp
|
||||
Interface/IR/PassManager.cpp
|
||||
Interface/IR/Passes/ConstProp.cpp
|
||||
Interface/IR/Passes/IRDumperPass.cpp
|
||||
Interface/IR/Passes/IRValidation.cpp
|
||||
Interface/IR/Passes/RedundantFlagCalculationElimination.cpp
|
||||
Interface/IR/Passes/RegisterAllocationPass.cpp
|
||||
Interface/IR/Passes/x87StackOptimizationPass.cpp
|
||||
Utils/LongJump.cpp
|
||||
Utils/Telemetry.cpp
|
||||
Utils/Threads.cpp
|
||||
Utils/Profiler.cpp
|
||||
@@ -207,7 +203,7 @@ add_custom_target(CONFIG_INC
|
||||
DEPENDS "${OUTPUT_MAN_NAME_COMPRESS}")
|
||||
|
||||
# Install the compressed man page
|
||||
install(FILES ${OUTPUT_MAN_NAME_COMPRESS} DESTINATION ${MAN_DIR}/man1)
|
||||
install(FILES ${OUTPUT_MAN_NAME_COMPRESS} COMPONENT Runtime DESTINATION ${MAN_DIR}/man1)
|
||||
|
||||
# Add in diagnostic colours if the option is available.
|
||||
# Ninja code generator will kill colours if this isn't here
|
||||
|
||||
@@ -2,12 +2,10 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
|
||||
#include <chrono>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <cstdio>
|
||||
#include <memory>
|
||||
#include <string_view>
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
@@ -294,7 +294,11 @@ struct FEX_PACKED X80SoftFloat {
|
||||
FCMP(softfloat_state* state, const X80SoftFloat& lhs, const X80SoftFloat& rhs, bool* eq, bool* lt, bool* nan) {
|
||||
*eq = extF80_eq(state, lhs, rhs);
|
||||
*lt = extF80_lt(state, lhs, rhs);
|
||||
*nan = IsNan(lhs) || IsNan(rhs);
|
||||
|
||||
// Use IEEE 754 semantics: unordered if neither <, =, nor > is true
|
||||
// This is more reliable than custom NaN detection
|
||||
bool gt = !(*eq) && !(*lt) && extF80_le(state, rhs, lhs);
|
||||
*nan = !(*eq) && !(*lt) && !gt;
|
||||
}
|
||||
|
||||
FEXCORE_PRESERVE_ALL_ATTR static X80SoftFloat FSCALE(softfloat_state* state, const X80SoftFloat& lhs, const X80SoftFloat& rhs) {
|
||||
|
||||
@@ -7,69 +7,58 @@
|
||||
#include <optional>
|
||||
|
||||
namespace FEXCore::StrConv {
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, bool* Result) {
|
||||
inline bool Conv(std::string_view Value, bool* Result) {
|
||||
*Result = std::strtoull(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, uint8_t* Result) {
|
||||
inline bool Conv(std::string_view Value, uint8_t* Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, int8_t* Result) {
|
||||
inline bool Conv(std::string_view Value, int8_t* Result) {
|
||||
*Result = std::strtol(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, uint16_t* Result) {
|
||||
inline bool Conv(std::string_view Value, uint16_t* Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, int16_t* Result) {
|
||||
inline bool Conv(std::string_view Value, int16_t* Result) {
|
||||
*Result = std::strtol(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, uint32_t* Result) {
|
||||
inline bool Conv(std::string_view Value, uint32_t* Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, int32_t* Result) {
|
||||
inline bool Conv(std::string_view Value, int32_t* Result) {
|
||||
*Result = std::strtol(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, uint64_t* Result) {
|
||||
inline bool Conv(std::string_view Value, uint64_t* Result) {
|
||||
*Result = std::strtoull(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, int64_t* Result) {
|
||||
inline bool Conv(std::string_view Value, int64_t* Result) {
|
||||
*Result = std::strtoll(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
template<typename T, typename = std::enable_if<std::is_enum<T>::value, T>>
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, T* Result) {
|
||||
inline bool Conv(std::string_view Value, T* Result) {
|
||||
*Result = static_cast<T>(std::stoull(Value.data(), nullptr, 0));
|
||||
return true;
|
||||
}
|
||||
|
||||
[[maybe_unused]]
|
||||
static bool Conv(std::string_view Value, fextl::string* Result) {
|
||||
inline bool Conv(std::string_view Value, fextl::string* Result) {
|
||||
*Result = Value;
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -4,6 +4,8 @@
|
||||
#ifdef _M_X86_64
|
||||
#include <xmmintrin.h>
|
||||
#include <immintrin.h>
|
||||
#else
|
||||
#include <cstdint>
|
||||
#endif
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
@@ -4,7 +4,6 @@
|
||||
"Multiblock": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"ShortArg": "m",
|
||||
"Desc": [
|
||||
"Controls multiblock code compilation",
|
||||
"Can cause long JIT compilation times and stutter"
|
||||
@@ -13,22 +12,10 @@
|
||||
"MaxInst": {
|
||||
"Type": "int32",
|
||||
"Default": "5000",
|
||||
"ShortArg": "n",
|
||||
"Desc": [
|
||||
"Maximum number of instruction to store in a block"
|
||||
]
|
||||
},
|
||||
"CacheObjectCodeCompilation": {
|
||||
"Type": "uint32",
|
||||
"Default": "FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE",
|
||||
"TextDefault": "none",
|
||||
"Choices": [ "none", "read", "readwrite" ],
|
||||
"ArgumentHandler": "CacheObjectCodeHandler",
|
||||
"Desc": [
|
||||
"Cache JIT object code to drive.",
|
||||
"Allows JIT code to be shared between applications"
|
||||
]
|
||||
},
|
||||
"HostFeatures": {
|
||||
"Type": "strenum",
|
||||
"Default": "FEXCore::Config::HostFeatures::OFF",
|
||||
@@ -70,7 +57,9 @@
|
||||
"ENABLEPRESERVEALLABI": "enablepreserveallabi",
|
||||
"DISABLEPRESERVEALLABI": "disablepreserveallabi",
|
||||
"ENABLEWFXT": "enablewfxt",
|
||||
"DISABLEWFXT": "disablewfxt"
|
||||
"DISABLEWFXT": "disablewfxt",
|
||||
"ENABLE3DNOW": "enable3dnow",
|
||||
"DISABLE3DNOW": "disable3dnow"
|
||||
},
|
||||
"Desc": [
|
||||
"Allows controlling of the CPU features in the JIT.",
|
||||
@@ -92,7 +81,8 @@
|
||||
"\t{enable,disable}rpres: Will force enable or disable rpres even if the host doesn't support it",
|
||||
"\t{enable,disable}svebitperm: Will force enable or disable svebitperm even if the host doesn't support it",
|
||||
"\t{enable,disable}preserveallabi: Will force enable or disable preserve_all abi even if the host doesn't support it",
|
||||
"\t{enable,disable}wfxt: Will force enable or disable wfxt even if the host doesn't support it"
|
||||
"\t{enable,disable}wfxt: Will force enable or disable wfxt even if the host doesn't support it",
|
||||
"\t{enable,disable}3dnow: Will force enable or disable 3DNow even if the host doesn't support it"
|
||||
]
|
||||
},
|
||||
"SmallTSCScale": {
|
||||
@@ -107,7 +97,6 @@
|
||||
"RootFS": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"ShortArg": "R",
|
||||
"Desc": [
|
||||
"Which Root filesystem prefix to use",
|
||||
"This can be a filesystem path",
|
||||
@@ -122,7 +111,6 @@
|
||||
"ThunkHostLibs": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_FULL_LIBDIR@/fex-emu/HostThunks",
|
||||
"ShortArg": "t",
|
||||
"Desc": [
|
||||
"Folder to find the host-side thunking libraries."
|
||||
]
|
||||
@@ -130,7 +118,6 @@
|
||||
"ThunkGuestLibs": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/share/fex-emu/GuestThunks",
|
||||
"ShortArg": "j",
|
||||
"Desc": [
|
||||
"Folder to find the guest-side thunking libraries."
|
||||
]
|
||||
@@ -138,7 +125,6 @@
|
||||
"ThunkConfig": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"ShortArg": "k",
|
||||
"Desc": [
|
||||
"A json file specifying where to overlay the thunks.",
|
||||
"This can be a filesystem path",
|
||||
@@ -153,7 +139,6 @@
|
||||
"Env": {
|
||||
"Type": "strarray",
|
||||
"Default": "",
|
||||
"ShortArg": "E",
|
||||
"Desc": [
|
||||
"Adds an environment variable to the emulated environment."
|
||||
]
|
||||
@@ -161,7 +146,6 @@
|
||||
"HostEnv": {
|
||||
"Type": "strarray",
|
||||
"Default": "",
|
||||
"ShortArg": "H",
|
||||
"Desc": [
|
||||
"Adds an environment variable to the host environment.",
|
||||
"This can be useful for setting environment variables that thunks can pick up.",
|
||||
@@ -180,7 +164,6 @@
|
||||
"SingleStep": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"ShortArg": "S",
|
||||
"Desc": [
|
||||
"Single stepping configuration."
|
||||
]
|
||||
@@ -188,7 +171,6 @@
|
||||
"GdbServer": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"ShortArg": "G",
|
||||
"Desc": [
|
||||
"Enables the GDB server."
|
||||
]
|
||||
@@ -222,7 +204,6 @@
|
||||
"DumpGPRs": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"ShortArg": "g",
|
||||
"Desc": [
|
||||
"When the test harness ends, print the GPR state."
|
||||
]
|
||||
@@ -230,7 +211,6 @@
|
||||
"O0": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"ShortArg": "O0",
|
||||
"Desc": [
|
||||
"Disables optimizations passes for debugging."
|
||||
]
|
||||
@@ -320,7 +300,6 @@
|
||||
"SilentLog": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"ShortArg": "s",
|
||||
"Desc": [
|
||||
"Disables logging"
|
||||
]
|
||||
@@ -328,7 +307,6 @@
|
||||
"OutputLog": {
|
||||
"Type": "str",
|
||||
"Default": "server",
|
||||
"ShortArg": "o",
|
||||
"Desc": [
|
||||
"File to write FEX output to.",
|
||||
"[stdout, stderr, server, <Filename>]"
|
||||
@@ -349,6 +327,13 @@
|
||||
"Enables FEX's low-overhead sampling profile statistics.",
|
||||
"Requires a supported version of Mangohud to see the results"
|
||||
]
|
||||
},
|
||||
"TraceProfiler": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Enables FEX's trace profiler. Using gpuvis or tracy"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Hacks": {
|
||||
@@ -403,14 +388,6 @@
|
||||
"This is required to ensure a split-lock doesn't tear inside the process"
|
||||
]
|
||||
},
|
||||
"TSOAutoMigration": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"Automatically enables TSO when shared memory is used.",
|
||||
"Should work without issues in most cases."
|
||||
]
|
||||
},
|
||||
"VolatileMetadata": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
@@ -473,32 +450,16 @@
|
||||
"Desc": [
|
||||
"Contrains the startup sleep to only apply to processes that match this name."
|
||||
]
|
||||
},
|
||||
"MonoHacks": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"Permits a hook-based SMC approach and smaller JIT blocks when mono is detected."
|
||||
]
|
||||
}
|
||||
},
|
||||
"Misc": {
|
||||
"AOTIRCapture": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Captures IR and generates an AOT IR cache.",
|
||||
"Captures both the loaded executable and libraries it loads."
|
||||
]
|
||||
},
|
||||
"AOTIRGenerate": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Scans file for executable code and generates an AOT IR cache.",
|
||||
"Does not run the executable."
|
||||
]
|
||||
},
|
||||
"AOTIRLoad": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Loads an AOT IR cache for the loaded executable."
|
||||
]
|
||||
},
|
||||
"ServerSocketPath": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
@@ -512,15 +473,32 @@
|
||||
"Desc": [
|
||||
"Disables inline syscalls in order to support seccomp handling"
|
||||
]
|
||||
},
|
||||
"ExtendedVolatileMetadata": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"Desc": [
|
||||
"Configuration provided volatile metadata. Only implemented for WoW64/arm64ec.",
|
||||
"Limited in its use but can be handy.",
|
||||
"Extends on top of what Microsoft has for volatile metadata, but also supported for WoW64.",
|
||||
"Colon delimited modules, then semi-colon delimited instructions, then comma delimited ranges",
|
||||
"Default disables TSO in the module, unless instructions overlap the range",
|
||||
"<module>;<offset begin>-<offset-end>,...;<instruction offset to force TSO>,...:<another>",
|
||||
"examples:",
|
||||
" * Disable TSO for a full module: Just provide the module name:",
|
||||
" `hl2_linux`",
|
||||
" * Disable TSO for a part of the module:",
|
||||
" `hl2_linux;<offset begin>-<offset-end>`",
|
||||
" * Disable TSO for a part of the module, but enable TSO for some instructions within the module",
|
||||
" `hl2_linux;<offset begin>-<offset-end>;<instruction offset>,<instruction offset>`",
|
||||
" * Disable TSO for multiple modules",
|
||||
" `hl2_linux:libsdl2.so`"
|
||||
]
|
||||
}
|
||||
}
|
||||
},
|
||||
"UnnamedOptions": {
|
||||
"Misc": {
|
||||
"IS_INTERPRETER": {
|
||||
"Type": "bool",
|
||||
"Default": "false"
|
||||
},
|
||||
"INTERPRETER_INSTALLED": {
|
||||
"Type": "bool",
|
||||
"Default": "false"
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
@@ -8,18 +9,12 @@
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
|
||||
#include <FEXCore/Core/Thunks.h>
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
|
||||
#include <string.h>
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
void InitializeStaticTables(OperatingMode Mode) {
|
||||
X86Tables::InitializeInfoTables(Mode);
|
||||
IR::InstallOpcodeHandlers(Mode);
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Context::Context> FEXCore::Context::Context::CreateNewContext(const FEXCore::HostFeatures& Features) {
|
||||
return fextl::make_unique<FEXCore::Context::ContextImpl>(Features);
|
||||
}
|
||||
|
||||
@@ -2,61 +2,49 @@
|
||||
#pragma once
|
||||
|
||||
#include "Common/JitSymbols.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/IR/AOTIR.h"
|
||||
#include <Interface/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/set.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
#include <stdint.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <mutex>
|
||||
#include <optional>
|
||||
#include <shared_mutex>
|
||||
|
||||
namespace FEXCore {
|
||||
class CodeLoader;
|
||||
class SignalDelegator;
|
||||
class ThunkHandler;
|
||||
|
||||
namespace CodeSerialize {
|
||||
class CodeObjectSerializeService;
|
||||
}
|
||||
namespace Core {
|
||||
struct DebugData;
|
||||
struct InternalThreadState;
|
||||
} // namespace Core
|
||||
|
||||
namespace CPU {
|
||||
class Arm64JITCore;
|
||||
class Dispatcher;
|
||||
} // namespace CPU
|
||||
|
||||
namespace HLE {
|
||||
struct SyscallArguments;
|
||||
class SyscallHandler;
|
||||
class SourcecodeResolver;
|
||||
struct SourcecodeMap;
|
||||
class SyscallHandler;
|
||||
} // namespace HLE
|
||||
} // namespace FEXCore
|
||||
|
||||
namespace FEXCore::IR {
|
||||
struct IRListCopy;
|
||||
class IRListView;
|
||||
namespace Validation {
|
||||
class IRValidation;
|
||||
}
|
||||
} // namespace FEXCore::IR
|
||||
|
||||
namespace FEXCore::Context {
|
||||
struct FEX_PACKED ExitFunctionLinkData {
|
||||
uint64_t HostCode;
|
||||
@@ -76,7 +64,23 @@ struct CustomIRResult {
|
||||
using BlockDelinkerFunc = void (*)(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record);
|
||||
constexpr uint32_t TSC_SCALE_MAXIMUM = 1'000'000'000; ///< 1Ghz
|
||||
|
||||
class ContextImpl final : public FEXCore::Context::Context, CPU::CodeBufferManager {
|
||||
class CodeCache : public AbstractCodeCache {
|
||||
public:
|
||||
CodeCache(ContextImpl&);
|
||||
~CodeCache();
|
||||
|
||||
ContextImpl& CTX;
|
||||
bool IsGeneratingCache = false;
|
||||
|
||||
void LoadData(Core::InternalThreadState&, std::byte* MappedCacheFile, const ExecutableFileSectionInfo&) override;
|
||||
bool SaveData(Core::InternalThreadState&, int TargetFD, const ExecutableFileSectionInfo&, uint64_t SerializedBaseAddress) override;
|
||||
|
||||
void InitiateCacheGeneration() override {
|
||||
IsGeneratingCache = true;
|
||||
}
|
||||
};
|
||||
|
||||
class ContextImpl final : public FEXCore::Context::Context, public CPU::CodeBufferManager {
|
||||
public:
|
||||
// Context base class implementation.
|
||||
bool InitCore() override;
|
||||
@@ -90,6 +94,7 @@ public:
|
||||
|
||||
bool IsAddressInCurrentBlock(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, uint64_t Size) override;
|
||||
bool IsCurrentBlockSingleInst(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
uint64_t GetGuestBlockEntry(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
|
||||
uint64_t RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState* Thread, uint64_t HostPC) override;
|
||||
uint32_t ReconstructCompactedEFLAGS(FEXCore::Core::InternalThreadState* Thread, bool WasInJIT, const uint64_t* HostGPRs, uint64_t PSTATE) override;
|
||||
@@ -145,24 +150,8 @@ public:
|
||||
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) override;
|
||||
FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) override;
|
||||
|
||||
FEXCore::IR::AOTIRCacheEntry* LoadAOTIRCacheEntry(const fextl::string& Name) override;
|
||||
void UnloadAOTIRCacheEntry(FEXCore::IR::AOTIRCacheEntry* Entry) override;
|
||||
|
||||
void SetAOTIRLoader(AOTIRLoaderCBFn CacheReader) override {
|
||||
IRCaptureCache.SetAOTIRLoader(std::move(CacheReader));
|
||||
}
|
||||
void SetAOTIRWriter(AOTIRWriterCBFn CacheWriter) override {
|
||||
IRCaptureCache.SetAOTIRWriter(std::move(CacheWriter));
|
||||
}
|
||||
void SetAOTIRRenamer(AOTIRRenamerCBFn CacheRenamer) override {
|
||||
IRCaptureCache.SetAOTIRRenamer(std::move(CacheRenamer));
|
||||
}
|
||||
|
||||
void FinalizeAOTIRCache() override {
|
||||
IRCaptureCache.FinalizeAOTIRCache();
|
||||
}
|
||||
void WriteFilesWithCode(AOTIRCodeFileWriterFn Writer) override {
|
||||
IRCaptureCache.WriteFilesWithCode(Writer);
|
||||
CodeCache& GetCodeCache() override {
|
||||
return CodeCache;
|
||||
}
|
||||
|
||||
void OnCodeBufferAllocated(CPU::CodeBuffer&) override;
|
||||
@@ -173,8 +162,6 @@ public:
|
||||
return CodeInvalidationMutex;
|
||||
}
|
||||
|
||||
void MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
|
||||
void ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) override;
|
||||
|
||||
bool IsAddressInCodeBuffer(FEXCore::Core::InternalThreadState* Thread, uintptr_t Address) const override;
|
||||
@@ -189,14 +176,13 @@ public:
|
||||
|
||||
void RemoveForceTSOInformation(uint64_t Address, uint64_t Size) override;
|
||||
|
||||
void MarkMonoDetected() override {
|
||||
MonoDetected = true;
|
||||
}
|
||||
|
||||
void MarkMonoBackpatcherBlock(uint64_t BlockEntry) override;
|
||||
|
||||
public:
|
||||
friend class FEXCore::HLE::SyscallHandler;
|
||||
#ifdef JIT_ARM64
|
||||
friend class FEXCore::CPU::Arm64JITCore;
|
||||
#endif
|
||||
|
||||
friend class FEXCore::IR::Validation::IRValidation;
|
||||
|
||||
struct {
|
||||
uint64_t VirtualMemSize {1ULL << 36};
|
||||
uint64_t TSCScale = 0;
|
||||
@@ -209,13 +195,9 @@ public:
|
||||
FEX_CONFIG_OPT(GdbServer, GDBSERVER);
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
|
||||
FEX_CONFIG_OPT(TSOAutoMigration, TSOAUTOMIGRATION);
|
||||
FEX_CONFIG_OPT(VectorTSOEnabled, VECTORTSOENABLED);
|
||||
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
|
||||
FEX_CONFIG_OPT(ABILocalFlags, ABILOCALFLAGS);
|
||||
FEX_CONFIG_OPT(AOTIRCapture, AOTIRCAPTURE);
|
||||
FEX_CONFIG_OPT(AOTIRGenerate, AOTIRGENERATE);
|
||||
FEX_CONFIG_OPT(AOTIRLoad, AOTIRLOAD);
|
||||
FEX_CONFIG_OPT(SMCChecks, SMCCHECKS);
|
||||
FEX_CONFIG_OPT(MaxInstPerBlock, MAXINST);
|
||||
FEX_CONFIG_OPT(RootFSPath, ROOTFS);
|
||||
@@ -224,12 +206,12 @@ public:
|
||||
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
|
||||
FEX_CONFIG_OPT(GDBSymbols, GDBSYMBOLS);
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
FEX_CONFIG_OPT(CacheObjectCodeCompilation, CACHEOBJECTCODECOMPILATION);
|
||||
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
|
||||
FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY);
|
||||
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
|
||||
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
|
||||
FEX_CONFIG_OPT(StrictInProcessSplitLocks, STRICTINPROCESSSPLITLOCKS);
|
||||
FEX_CONFIG_OPT(MonoHacks, MONOHACKS);
|
||||
} Config;
|
||||
|
||||
FEXCore::ForkableSharedMutex CodeInvalidationMutex;
|
||||
@@ -243,29 +225,23 @@ public:
|
||||
FEXCore::HLE::SourcecodeResolver* SourcecodeResolver {};
|
||||
FEXCore::ThunkHandler* ThunkHandler {};
|
||||
fextl::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
CodeCache CodeCache;
|
||||
|
||||
SignalDelegator* SignalDelegation {};
|
||||
X86GeneratedCode X86CodeGen;
|
||||
|
||||
ContextImpl(const FEXCore::HostFeatures& Features);
|
||||
~ContextImpl();
|
||||
|
||||
static bool ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP);
|
||||
|
||||
// Wrapper which takes CpuStateFrame instead of InternalThreadState and unique_locks CodeInvalidationMutex
|
||||
// Must be called from owning thread
|
||||
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP) {
|
||||
auto Thread = Frame->Thread;
|
||||
auto lk = GuardSignalDeferringSection(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP);
|
||||
|
||||
// NOTE: Other threads sharing the same CodeBuffer may reference
|
||||
// invalidated data ranges through their L1/L2 caches. This is
|
||||
// not currently a problem since FEX does not repurpose the
|
||||
// invalidated CodeBuffer memory range currently.
|
||||
ThreadRemoveCodeEntry(Thread, GuestRIP);
|
||||
}
|
||||
// This is used as a replacement for the SMC writes in the mono callsite backpatcher that avoids atomic operations
|
||||
// (safe as the invalidation mutex is locked) and manually invalidates the modified range. Allowing SMC to be detected
|
||||
// even if faulting is disabled.
|
||||
static void MonoBackpatcherWrite(FEXCore::Core::CpuStateFrame* Frame, uint8_t Size, uint64_t Address, uint64_t Value);
|
||||
|
||||
void RemoveCustomIREntrypoint(uintptr_t Entrypoint);
|
||||
void RemoveCustomIREntrypoint(FEXCore::Core::InternalThreadState* Thread, uintptr_t Entrypoint);
|
||||
|
||||
struct GenerateIRResult {
|
||||
std::optional<IR::IRListView> IRView;
|
||||
@@ -324,6 +300,10 @@ public:
|
||||
return ExitOnHLT;
|
||||
}
|
||||
|
||||
bool AreMonoHacksActive() const {
|
||||
return Config.MonoHacks && MonoDetected;
|
||||
}
|
||||
|
||||
protected:
|
||||
void UpdateAtomicTSOEmulationConfig() {
|
||||
if (SupportsHardwareTSO) {
|
||||
@@ -336,12 +316,9 @@ protected:
|
||||
VectorAtomicTSOEmulationEnabled = true;
|
||||
MemcpyAtomicTSOEmulationEnabled = true;
|
||||
} else {
|
||||
// Atomic TSO emulation only enabled if the config option is enabled.
|
||||
AtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled;
|
||||
// Atomic vector TSO emulation only enabled if TSO emulation is enabled and also vector TSO is enabled.
|
||||
VectorAtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled && Config.VectorTSOEnabled;
|
||||
// Atomic memcpy TSO emulation only enabled if TSO emulation is enabled and also memcpy TSO is enabled.
|
||||
MemcpyAtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled && Config.MemcpySetTSOEnabled;
|
||||
AtomicTSOEmulationEnabled = Config.TSOEnabled;
|
||||
VectorAtomicTSOEmulationEnabled = Config.TSOEnabled && Config.VectorTSOEnabled;
|
||||
MemcpyAtomicTSOEmulationEnabled = Config.TSOEnabled && Config.MemcpySetTSOEnabled;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -355,10 +332,6 @@ private:
|
||||
*/
|
||||
void InitializeCompiler(FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
IR::AOTIRCaptureCache IRCaptureCache;
|
||||
fextl::unique_ptr<FEXCore::CodeSerialize::CodeObjectSerializeService> CodeObjectCacheService;
|
||||
|
||||
bool IsMemoryShared = false;
|
||||
bool SupportsHardwareTSO = false;
|
||||
bool AtomicTSOEmulationEnabled = true;
|
||||
bool VectorAtomicTSOEmulationEnabled = false;
|
||||
@@ -371,11 +344,14 @@ private:
|
||||
std::atomic<bool> HasCustomIRHandlers {};
|
||||
struct CustomIRHandlerEntry final {
|
||||
CustomIREntrypointHandler Handler;
|
||||
void *Creator;
|
||||
void *Data;
|
||||
void* Creator;
|
||||
void* Data;
|
||||
};
|
||||
fextl::unordered_map<uint64_t, CustomIRHandlerEntry> CustomIRHandlers;
|
||||
IntervalList<uint64_t> ForceTSOValidRanges; // The ranges for which ForceTSOInstructions has populated data
|
||||
fextl::set<uint64_t> ForceTSOInstructions;
|
||||
|
||||
bool MonoDetected = false;
|
||||
std::atomic<uint64_t> MonoBackpatcherBlock;
|
||||
};
|
||||
} // namespace FEXCore::Context
|
||||
@@ -11,8 +11,7 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, b
|
||||
Ref Tmp = A.Base;
|
||||
|
||||
if (A.Offset) {
|
||||
Ref Offset = IREmit->Constant(A.Offset);
|
||||
Tmp = Tmp ? IREmit->_Add(GPRSize, Tmp, Offset) : Offset;
|
||||
Tmp = Tmp ? IREmit->Add(GPRSize, Tmp, A.Offset) : IREmit->Constant(A.Offset);
|
||||
}
|
||||
|
||||
if (A.Index) {
|
||||
@@ -25,7 +24,7 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, b
|
||||
Tmp = IREmit->_Lshl(GPRSize, A.Index, IREmit->Constant(Log2));
|
||||
}
|
||||
} else {
|
||||
Tmp = Tmp ? IREmit->_Add(GPRSize, Tmp, A.Index) : A.Index;
|
||||
Tmp = Tmp ? IREmit->Add(GPRSize, Tmp, A.Index) : A.Index;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -46,7 +45,7 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, b
|
||||
}
|
||||
|
||||
if (A.Segment && AddSegmentBase) {
|
||||
Tmp = Tmp ? IREmit->_Add(GPRSize, Tmp, A.Segment) : A.Segment;
|
||||
Tmp = Tmp ? IREmit->Add(GPRSize, Tmp, A.Segment) : A.Segment;
|
||||
}
|
||||
|
||||
return Tmp ?: IREmit->Constant(0);
|
||||
@@ -54,19 +53,15 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, b
|
||||
|
||||
AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO, bool Vector,
|
||||
IR::OpSize AccessSize) {
|
||||
auto SoftwareAddressCalculation = [IREmit, &A, GPRSize]() -> AddressMode {
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(IREmit, A, GPRSize, true),
|
||||
.Index = IREmit->Invalid(),
|
||||
};
|
||||
};
|
||||
|
||||
const auto Is32Bit = GPRSize == OpSize::i32Bit;
|
||||
const auto GPRSizeMatchesAddrSize = A.AddrSize == GPRSize;
|
||||
const auto OffsetIndexToLargeFor32Bit = Is32Bit && (A.Offset <= -16384 || A.Offset >= 16384);
|
||||
if (!GPRSizeMatchesAddrSize || OffsetIndexToLargeFor32Bit) {
|
||||
// If address size doesn't match GPR size then no optimizations can occur.
|
||||
return SoftwareAddressCalculation();
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(IREmit, A, GPRSize, true),
|
||||
.Index = IREmit->Invalid(),
|
||||
};
|
||||
}
|
||||
|
||||
// Loadstore rules:
|
||||
@@ -100,7 +95,7 @@ AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSi
|
||||
const bool OffsetIsSIMM9 = A.Offset && A.Offset >= -256 && A.Offset <= 255;
|
||||
const bool OffsetIsUnsignedScaled = A.Offset > 0 && (A.Offset & (AccessSizeAsImm - 1)) == 0 && (A.Offset / AccessSizeAsImm) <= 4095;
|
||||
|
||||
auto InlineImmOffsetLoadstore = [IREmit, &GPRSize](AddressMode A) -> AddressMode {
|
||||
if ((AtomicTSO && !Vector && HostSupportsTSOImm9 && OffsetIsSIMM9) || (!AtomicTSO && (OffsetIsSIMM9 || OffsetIsUnsignedScaled))) {
|
||||
// Peel off the offset
|
||||
AddressMode B = A;
|
||||
B.Offset = 0;
|
||||
@@ -111,32 +106,20 @@ AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSi
|
||||
.IndexType = MEM_OFFSET_SXTX,
|
||||
.IndexScale = 1,
|
||||
};
|
||||
};
|
||||
}
|
||||
|
||||
auto ScaledRegisterLoadstore = [IREmit, GPRSize](AddressMode A) -> AddressMode {
|
||||
if (AtomicTSO) {
|
||||
// TODO: LRCPC3 support for vector Imm9.
|
||||
} else if (!Is32Bit && A.Base && (A.Index || A.Segment) && !A.Offset && (A.IndexScale == 1 || A.IndexScale == AccessSizeAsImm)) {
|
||||
// ScaledRegisterLoadstore
|
||||
if (A.Index && A.Segment) {
|
||||
A.Base = IREmit->_Add(GPRSize, A.Base, A.Segment);
|
||||
A.Base = IREmit->Add(GPRSize, A.Base, A.Segment);
|
||||
} else if (A.Segment) {
|
||||
A.Index = A.Segment;
|
||||
A.IndexScale = 1;
|
||||
}
|
||||
return A;
|
||||
};
|
||||
|
||||
if (AtomicTSO) {
|
||||
if (!Vector) {
|
||||
if (HostSupportsTSOImm9 && OffsetIsSIMM9) {
|
||||
return InlineImmOffsetLoadstore(A);
|
||||
}
|
||||
} else {
|
||||
// TODO: LRCPC3 support for vector Imm9.
|
||||
}
|
||||
} else {
|
||||
if (OffsetIsSIMM9 || OffsetIsUnsignedScaled) {
|
||||
return InlineImmOffsetLoadstore(A);
|
||||
} else if (!Is32Bit && A.Base && (A.Index || A.Segment) && !A.Offset && (A.IndexScale == 1 || A.IndexScale == AccessSizeAsImm)) {
|
||||
return ScaledRegisterLoadstore(A);
|
||||
}
|
||||
return A;
|
||||
}
|
||||
|
||||
if (Vector || !AtomicTSO) {
|
||||
@@ -159,7 +142,10 @@ AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSi
|
||||
}
|
||||
|
||||
// Fallback on software address calculation
|
||||
return SoftwareAddressCalculation();
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(IREmit, A, GPRSize, true),
|
||||
.Index = IREmit->Invalid(),
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -712,7 +712,7 @@ void Arm64Emitter::SpillStaticRegs(ARMEmitter::Register TmpReg, bool FPRs, uint3
|
||||
// Now handle PF/AF
|
||||
if (PFAFSpillMask) {
|
||||
auto PFOffset = offsetof(FEXCore::Core::CpuStateFrame, State.pf_raw);
|
||||
[[maybe_unused]] auto AFOffset = offsetof(FEXCore::Core::CpuStateFrame, State.af_raw);
|
||||
auto AFOffset = offsetof(FEXCore::Core::CpuStateFrame, State.af_raw);
|
||||
LOGMAN_THROW_A_FMT(PFAFSpillMask == PFAFMask, "PF/AF not spilled together");
|
||||
LOGMAN_THROW_A_FMT(AFOffset == PFOffset + 4, "PF/AF are together");
|
||||
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
#pragma once
|
||||
|
||||
#include "FEXCore/Utils/EnumUtils.h"
|
||||
#include "Interface/Core/ObjectCache/Relocations.h"
|
||||
#include "Interface/Core/JIT/Relocations.h"
|
||||
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
#include <aarch64/disasm-aarch64.h>
|
||||
|
||||
@@ -317,7 +317,7 @@ namespace CPU {
|
||||
// Resize the code buffer and reallocate our code size
|
||||
CurrentCodeBuffer = CodeBuffers.StartLargerCodeBuffer();
|
||||
|
||||
RegisterForSignalHandler(PrevCodeBuffer);
|
||||
RegisterForSignalHandler(std::move(PrevCodeBuffer));
|
||||
return CurrentCodeBuffer.get();
|
||||
}
|
||||
|
||||
@@ -326,14 +326,13 @@ namespace CPU {
|
||||
// We have signal handlers that have generated code
|
||||
// This means that we can not safely clear the code at this point in time
|
||||
// Keep a reference to the old code buffer to delay deallocation
|
||||
SignalHandlerCodeBuffers.push_back(CodeBuffer);
|
||||
SignalHandlerCodeBuffers.push_back(std::move(CodeBuffer));
|
||||
} else {
|
||||
SignalHandlerCodeBuffers.clear();
|
||||
}
|
||||
}
|
||||
|
||||
fextl::shared_ptr<CodeBuffer> CPUBackend::CheckCodeBufferUpdate() {
|
||||
fextl::shared_ptr<CodeBuffer> OldCodeBuffer;
|
||||
auto NewCodeBuffer = CodeBuffers.GetLatest();
|
||||
if (CurrentCodeBuffer != NewCodeBuffer) {
|
||||
RegisterForSignalHandler(CurrentCodeBuffer);
|
||||
|
||||
@@ -17,6 +17,10 @@ $end_info$
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
union Relocation;
|
||||
}
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
namespace IR {
|
||||
@@ -157,18 +161,7 @@ namespace CPU {
|
||||
virtual CompiledCode CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, bool CheckTF) = 0;
|
||||
|
||||
/**
|
||||
* @brief Relocates a block of code from the JIT code object cache
|
||||
*
|
||||
* @param Entry - RIP of the entry
|
||||
* @param SerializationData - Serialization data referring to the object cache for `Entry`
|
||||
*
|
||||
* @return An executable function pointer relocated from the cache object
|
||||
*/
|
||||
[[nodiscard]]
|
||||
virtual void* RelocateJITObjectCode(uint64_t /* Entry */, const CodeSerialize::CodeObjectFileSection* /* SerializationData */) {
|
||||
return nullptr;
|
||||
}
|
||||
virtual fextl::vector<FEXCore::CPU::Relocation> TakeRelocations() = 0;
|
||||
|
||||
virtual void ClearCache() {}
|
||||
|
||||
|
||||
@@ -43,12 +43,15 @@ namespace ProductNames {
|
||||
static const char ARM_A715[] = "Cortex-A715";
|
||||
static const char ARM_A720[] = "Cortex-A720";
|
||||
static const char ARM_A725[] = "Cortex-A725";
|
||||
static const char ARM_C1Pro[] = "C1-Pro";
|
||||
static const char ARM_C1Premium[] = "C1-Premium";
|
||||
static const char ARM_X1[] = "Cortex-X1";
|
||||
static const char ARM_X1C[] = "Cortex-X1C";
|
||||
static const char ARM_X2[] = "Cortex-X2";
|
||||
static const char ARM_X3[] = "Cortex-X3";
|
||||
static const char ARM_X4[] = "Cortex-X4";
|
||||
static const char ARM_X925[] = "Cortex-X925";
|
||||
static const char ARM_C1Ultra[] = "C1-Ultra";
|
||||
static const char ARM_N1[] = "Neoverse N1";
|
||||
static const char ARM_N2[] = "Neoverse N2";
|
||||
static const char ARM_N3[] = "Neoverse N3";
|
||||
@@ -59,6 +62,7 @@ namespace ProductNames {
|
||||
static const char ARM_A65[] = "Cortex-A65";
|
||||
static const char ARM_A510[] = "Cortex-A510";
|
||||
static const char ARM_A520[] = "Cortex-A520";
|
||||
static const char ARM_C1Nano[] = "C1-Nano";
|
||||
|
||||
static const char ARM_Kryo200[] = "Kryo 2xx";
|
||||
static const char ARM_Kryo300[] = "Kryo 3xx";
|
||||
@@ -70,6 +74,7 @@ namespace ProductNames {
|
||||
|
||||
static const char ARM_Denver[] = "Nvidia Denver";
|
||||
static const char ARM_Carmel[] = "Nvidia Carmel";
|
||||
static const char ARM_Olympus[] = "Nvidia Olympus";
|
||||
|
||||
static const char ARM_Firestorm_M1[] = "Apple Firestorm (M1)";
|
||||
static const char ARM_Icestorm_M1[] = "Apple Icestorm (M1)";
|
||||
@@ -85,6 +90,9 @@ namespace ProductNames {
|
||||
static const char ARM_Blizzard_M2Max[] = "Apple Blizzard (M2 Max)";
|
||||
|
||||
static const char ARM_ORYON_1[] = "Oryon-1";
|
||||
static const char ARM_Ampere_1[] = "AmpereOne";
|
||||
static const char ARM_Ampere_1A[] = "AmpereOneA";
|
||||
static const char ARM_Ampere_1B[] = "AmpereOneB";
|
||||
#else
|
||||
#endif
|
||||
} // namespace ProductNames
|
||||
@@ -170,7 +178,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
// CPU priority order
|
||||
// This is mostly arbitrary but will sort by some sort of CPU priority by performance
|
||||
// Relative list so things they will commonly end up in big.little configurations sort of relate
|
||||
static constexpr std::array<CPUMIDR, 58> CPUMIDRs = {{
|
||||
static constexpr std::array<CPUMIDR, 66> CPUMIDRs = {{
|
||||
// Typically big CPU cores
|
||||
{0x51, 0x001, 1, ProductNames::ARM_ORYON_1}, // Qualcomm Oryon-1
|
||||
|
||||
@@ -181,38 +189,46 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
{0x61, 0x025, 1, ProductNames::ARM_Firestorm_M1Pro}, // Apple Firestorm (M1 Pro)
|
||||
{0x61, 0x023, 1, ProductNames::ARM_Firestorm_M1}, // Apple Firestorm (M1)
|
||||
|
||||
{0x41, 0xd85, 1, ProductNames::ARM_X925}, // X925
|
||||
{0x41, 0xd87, 1, ProductNames::ARM_A725}, // A725
|
||||
{0x41, 0xd84, 1, ProductNames::ARM_V3}, // V3
|
||||
{0x41, 0xd83, 1, ProductNames::ARM_V3AE}, // V3AE
|
||||
{0x41, 0xd8e, 1, ProductNames::ARM_N3}, // N3
|
||||
{0x41, 0xd82, 1, ProductNames::ARM_X4}, // X4
|
||||
{0x41, 0xd81, 1, ProductNames::ARM_A720}, // A720
|
||||
{0x41, 0xd4e, 1, ProductNames::ARM_X3}, // X3
|
||||
{0x41, 0xd4d, 1, ProductNames::ARM_A715}, // A715
|
||||
{0x41, 0xd4f, 1, ProductNames::ARM_V2}, // V2
|
||||
{0x41, 0xd4b, 1, ProductNames::ARM_A78C}, // A78C
|
||||
{0x41, 0xd4a, 1, ProductNames::ARM_E1}, // E1
|
||||
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
|
||||
{0x41, 0xd48, 1, ProductNames::ARM_X2}, // X2
|
||||
{0x41, 0xd47, 1, ProductNames::ARM_A710}, // A710
|
||||
{0x41, 0xd4C, 1, ProductNames::ARM_X1C}, // X1C
|
||||
{0x41, 0xd44, 1, ProductNames::ARM_X1}, // X1
|
||||
{0x41, 0xd42, 1, ProductNames::ARM_A78AE}, // A78AE
|
||||
{0x41, 0xd41, 1, ProductNames::ARM_A78}, // A78
|
||||
{0x41, 0xd40, 1, ProductNames::ARM_V1}, // V1
|
||||
{0x41, 0xd0e, 1, ProductNames::ARM_A76AE}, // A76AE
|
||||
{0x41, 0xd0d, 1, ProductNames::ARM_A77}, // A77
|
||||
{0x41, 0xd0c, 1, ProductNames::ARM_N1}, // N1
|
||||
{0x41, 0xd0b, 1, ProductNames::ARM_A76}, // A76
|
||||
{0x51, 0x804, 1, ProductNames::ARM_Kryo400}, // Kryo 4xx Gold (A76 based)
|
||||
{0x41, 0xd0a, 1, ProductNames::ARM_A75}, // A75
|
||||
{0x51, 0x802, 1, ProductNames::ARM_Kryo300}, // Kryo 3xx Gold (A75 based)
|
||||
{0x41, 0xd09, 1, ProductNames::ARM_A73}, // A73
|
||||
{0x51, 0x800, 1, ProductNames::ARM_Kryo200}, // Kryo 2xx Gold (A73 based)
|
||||
{0x41, 0xd08, 1, ProductNames::ARM_A72}, // A72
|
||||
{0x41, 0xd8c, 1, ProductNames::ARM_C1Ultra}, // C1-Ultra
|
||||
{0x41, 0xd90, 1, ProductNames::ARM_C1Premium}, // C1-Premium
|
||||
{0x41, 0xd8b, 1, ProductNames::ARM_C1Pro}, // C1-Pro
|
||||
{0x41, 0xd85, 1, ProductNames::ARM_X925}, // X925
|
||||
{0x41, 0xd87, 1, ProductNames::ARM_A725}, // A725
|
||||
{0x41, 0xd84, 1, ProductNames::ARM_V3}, // V3
|
||||
{0x41, 0xd83, 1, ProductNames::ARM_V3AE}, // V3AE
|
||||
{0x41, 0xd8e, 1, ProductNames::ARM_N3}, // N3
|
||||
{0x41, 0xd82, 1, ProductNames::ARM_X4}, // X4
|
||||
{0x41, 0xd81, 1, ProductNames::ARM_A720}, // A720
|
||||
{0x41, 0xd4e, 1, ProductNames::ARM_X3}, // X3
|
||||
{0x41, 0xd4d, 1, ProductNames::ARM_A715}, // A715
|
||||
{0x41, 0xd4f, 1, ProductNames::ARM_V2}, // V2
|
||||
{0x41, 0xd4b, 1, ProductNames::ARM_A78C}, // A78C
|
||||
{0x41, 0xd4a, 1, ProductNames::ARM_E1}, // E1
|
||||
{0x41, 0xd49, 1, ProductNames::ARM_N2}, // N2
|
||||
{0x41, 0xd48, 1, ProductNames::ARM_X2}, // X2
|
||||
{0x41, 0xd47, 1, ProductNames::ARM_A710}, // A710
|
||||
{0x41, 0xd4C, 1, ProductNames::ARM_X1C}, // X1C
|
||||
{0x41, 0xd44, 1, ProductNames::ARM_X1}, // X1
|
||||
{0x41, 0xd42, 1, ProductNames::ARM_A78AE}, // A78AE
|
||||
{0x41, 0xd41, 1, ProductNames::ARM_A78}, // A78
|
||||
{0x41, 0xd40, 1, ProductNames::ARM_V1}, // V1
|
||||
{0x41, 0xd0e, 1, ProductNames::ARM_A76AE}, // A76AE
|
||||
{0x41, 0xd0d, 1, ProductNames::ARM_A77}, // A77
|
||||
{0x41, 0xd0c, 1, ProductNames::ARM_N1}, // N1
|
||||
{0x41, 0xd0b, 1, ProductNames::ARM_A76}, // A76
|
||||
{0x51, 0x804, 1, ProductNames::ARM_Kryo400}, // Kryo 4xx Gold (A76 based)
|
||||
{0x41, 0xd0a, 1, ProductNames::ARM_A75}, // A75
|
||||
{0x51, 0x802, 1, ProductNames::ARM_Kryo300}, // Kryo 3xx Gold (A75 based)
|
||||
{0x41, 0xd09, 1, ProductNames::ARM_A73}, // A73
|
||||
{0x51, 0x800, 1, ProductNames::ARM_Kryo200}, // Kryo 2xx Gold (A73 based)
|
||||
{0x41, 0xd08, 1, ProductNames::ARM_A72}, // A72
|
||||
|
||||
{0x4e, 0x004, 1, ProductNames::ARM_Carmel}, // Carmel
|
||||
{0xc0, 0xac3, 1, ProductNames::ARM_Ampere_1}, // AmpereOne
|
||||
{0xc0, 0xac4, 1, ProductNames::ARM_Ampere_1A}, // AmpereOneA
|
||||
{0xc0, 0xac5, 1, ProductNames::ARM_Ampere_1B}, // AmpereOneB
|
||||
|
||||
{0x4e, 0x010, 1, ProductNames::ARM_Olympus}, // Olympus
|
||||
{0x4e, 0x004, 1, ProductNames::ARM_Carmel}, // Carmel
|
||||
|
||||
// Denver rated above A57 to match TX2 weirdness
|
||||
{0x4e, 0x003, 1, ProductNames::ARM_Denver}, // Denver
|
||||
@@ -227,6 +243,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
{0x61, 0x024, 0, ProductNames::ARM_Icestorm_M1Pro}, // Apple Icestorm (M1 Pro)
|
||||
{0x61, 0x022, 0, ProductNames::ARM_Icestorm_M1}, // Apple Icestorm (M1)
|
||||
|
||||
{0x41, 0xd8a, 1, ProductNames::ARM_C1Nano}, // C1-Nano
|
||||
{0x41, 0xd80, 0, ProductNames::ARM_A520}, // A520
|
||||
{0x41, 0xd46, 0, ProductNames::ARM_A510}, // A510
|
||||
{0x41, 0xd06, 0, ProductNames::ARM_A65}, // A65
|
||||
@@ -925,38 +942,38 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) con
|
||||
(0 << 30) | // Reserved
|
||||
(0 << 31); // Reserved
|
||||
|
||||
Res.edx = (1 << 0) | // FPU
|
||||
(1 << 1) | // Virtual mode extensions
|
||||
(1 << 2) | // Debugging extensions
|
||||
(1 << 3) | // Page size extensions
|
||||
(1 << 4) | // TSC
|
||||
(1 << 5) | // MSR support
|
||||
(1 << 6) | // PAE
|
||||
(1 << 7) | // Machine Check Exception
|
||||
(1 << 8) | // CMPXCHG8B
|
||||
(1 << 9) | // APIC
|
||||
(0 << 10) | // Reserved
|
||||
(1 << 11) | // SYSCALL/SYSRET
|
||||
(1 << 12) | // MTRR
|
||||
(1 << 13) | // Page global extension
|
||||
(1 << 14) | // Machine Check architecture
|
||||
(1 << 15) | // CMOV
|
||||
(1 << 16) | // Page attribute table
|
||||
(1 << 17) | // Page-size extensions
|
||||
(0 << 18) | // Reserved
|
||||
(0 << 19) | // Reserved
|
||||
(1 << 20) | // NX
|
||||
(0 << 21) | // Reserved
|
||||
(1 << 22) | // MMXExt
|
||||
(1 << 23) | // MMX
|
||||
(1 << 24) | // FXSAVE/FXRSTOR
|
||||
(1 << 25) | // FXSAVE/FXRSTOR Optimizations
|
||||
(0 << 26) | // 1 gigabit pages
|
||||
(SUPPORTS_RDTSCP << 27) | // RDTSCP
|
||||
(0 << 28) | // Reserved
|
||||
(1 << 29) | // Long Mode
|
||||
(1 << 30) | // 3DNow! Extensions
|
||||
(1 << 31); // 3DNow!
|
||||
Res.edx = (1 << 0) | // FPU
|
||||
(1 << 1) | // Virtual mode extensions
|
||||
(1 << 2) | // Debugging extensions
|
||||
(1 << 3) | // Page size extensions
|
||||
(1 << 4) | // TSC
|
||||
(1 << 5) | // MSR support
|
||||
(1 << 6) | // PAE
|
||||
(1 << 7) | // Machine Check Exception
|
||||
(1 << 8) | // CMPXCHG8B
|
||||
(1 << 9) | // APIC
|
||||
(0 << 10) | // Reserved
|
||||
(1 << 11) | // SYSCALL/SYSRET
|
||||
(1 << 12) | // MTRR
|
||||
(1 << 13) | // Page global extension
|
||||
(1 << 14) | // Machine Check architecture
|
||||
(1 << 15) | // CMOV
|
||||
(1 << 16) | // Page attribute table
|
||||
(1 << 17) | // Page-size extensions
|
||||
(0 << 18) | // Reserved
|
||||
(0 << 19) | // Reserved
|
||||
(1 << 20) | // NX
|
||||
(0 << 21) | // Reserved
|
||||
(1 << 22) | // MMXExt
|
||||
(1 << 23) | // MMX
|
||||
(1 << 24) | // FXSAVE/FXRSTOR
|
||||
(1 << 25) | // FXSAVE/FXRSTOR Optimizations
|
||||
(0 << 26) | // 1 gigabit pages
|
||||
(SUPPORTS_RDTSCP << 27) | // RDTSCP
|
||||
(0 << 28) | // Reserved
|
||||
(1 << 29) | // Long Mode
|
||||
(CTX->HostFeatures.Supports3DNow << 30) | // 3DNow! Extensions
|
||||
(CTX->HostFeatures.Supports3DNow << 31); // 3DNow!
|
||||
return Res;
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,27 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <Interface/Context/Context.h>
|
||||
|
||||
#include <FEXCore/HLE/SourcecodeResolver.h>
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
ExecutableFileInfo::~ExecutableFileInfo() = default;
|
||||
|
||||
} // namespace FEXCore
|
||||
|
||||
namespace FEXCore::Context {
|
||||
|
||||
CodeCache::CodeCache(ContextImpl& CTX_)
|
||||
: CTX(CTX_) {}
|
||||
CodeCache::~CodeCache() = default;
|
||||
|
||||
void CodeCache::LoadData(Core::InternalThreadState& Thread, std::byte* MappedCacheFile, const ExecutableFileSectionInfo& GuestRIPLookup) {
|
||||
// TODO
|
||||
}
|
||||
|
||||
bool CodeCache::SaveData(Core::InternalThreadState& Thread, int fd, const ExecutableFileSectionInfo& SourceBinary, uint64_t SerializedBaseAddress) {
|
||||
// TODO
|
||||
return true;
|
||||
}
|
||||
|
||||
} // namespace FEXCore::Context
|
||||
@@ -14,11 +14,11 @@ $end_info$
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include <Interface/GDBJIT/GDBJIT.h>
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
@@ -78,10 +78,7 @@ namespace FEXCore::Context {
|
||||
ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
|
||||
: HostFeatures {Features}
|
||||
, CPUID {this}
|
||||
, IRCaptureCache {this} {
|
||||
if (Config.CacheObjectCodeCompilation() != FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE) {
|
||||
CodeObjectCacheService = fextl::make_unique<FEXCore::CodeSerialize::CodeObjectSerializeService>(this);
|
||||
}
|
||||
, CodeCache {*this} {
|
||||
if (!Config.Is64BitMode()) {
|
||||
// When operating in 32-bit mode, the virtual memory we care about is only the lower 32-bits.
|
||||
Config.VirtualMemSize = 1ULL << 32;
|
||||
@@ -105,14 +102,6 @@ ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
}
|
||||
|
||||
ContextImpl::~ContextImpl() {
|
||||
{
|
||||
if (CodeObjectCacheService) {
|
||||
CodeObjectCacheService->Shutdown();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
struct GetFrameBlockInfoResult {
|
||||
const CPU::CPUBackend::JITCodeHeader* InlineHeader;
|
||||
const CPU::CPUBackend::JITCodeTail* InlineTail;
|
||||
@@ -139,6 +128,11 @@ bool ContextImpl::IsCurrentBlockSingleInst(FEXCore::Core::InternalThreadState* T
|
||||
return InlineTail && InlineTail->SingleInst;
|
||||
}
|
||||
|
||||
uint64_t ContextImpl::GetGuestBlockEntry(FEXCore::Core::InternalThreadState* Thread) {
|
||||
auto [_, InlineTail] = GetFrameBlockInfo(Thread->CurrentFrame);
|
||||
return InlineTail ? InlineTail->RIP : 0;
|
||||
}
|
||||
|
||||
uint64_t ContextImpl::RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState* Thread, uint64_t HostPC) {
|
||||
const auto Frame = Thread->CurrentFrame;
|
||||
const uint64_t BlockBegin = Frame->State.InlineJITBlockHeader;
|
||||
@@ -349,36 +343,9 @@ bool ContextImpl::InitCore() {
|
||||
Dispatcher = FEXCore::CPU::Dispatcher::Create(this);
|
||||
|
||||
// Set up the SignalDelegator config since core is initialized.
|
||||
FEXCore::SignalDelegator::SignalDelegatorConfig SignalConfig {
|
||||
.DispatcherBegin = Dispatcher->Start,
|
||||
.DispatcherEnd = Dispatcher->End,
|
||||
SignalDelegation->SetConfig(Dispatcher->MakeSignalDelegatorConfig());
|
||||
|
||||
.AbsoluteLoopTopAddress = Dispatcher->AbsoluteLoopTopAddress,
|
||||
.AbsoluteLoopTopAddressFillSRA = Dispatcher->AbsoluteLoopTopAddressFillSRA,
|
||||
.SignalHandlerReturnAddress = Dispatcher->SignalHandlerReturnAddress,
|
||||
.SignalHandlerReturnAddressRT = Dispatcher->SignalHandlerReturnAddressRT,
|
||||
|
||||
.PauseReturnInstruction = Dispatcher->PauseReturnInstruction,
|
||||
.ThreadPauseHandlerAddressSpillSRA = Dispatcher->ThreadPauseHandlerAddressSpillSRA,
|
||||
.ThreadPauseHandlerAddress = Dispatcher->ThreadPauseHandlerAddress,
|
||||
|
||||
// Stop handlers.
|
||||
.ThreadStopHandlerAddressSpillSRA = Dispatcher->ThreadStopHandlerAddressSpillSRA,
|
||||
.ThreadStopHandlerAddress = Dispatcher->ThreadStopHandlerAddress,
|
||||
|
||||
// SRA information.
|
||||
.SRAGPRCount = Dispatcher->GetSRAGPRCount(),
|
||||
.SRAFPRCount = Dispatcher->GetSRAFPRCount(),
|
||||
};
|
||||
|
||||
Dispatcher->GetSRAGPRMapping(SignalConfig.SRAGPRMapping);
|
||||
Dispatcher->GetSRAFPRMapping(SignalConfig.SRAFPRMapping);
|
||||
|
||||
// Give this configuration to the SignalDelegator.
|
||||
SignalDelegation->SetConfig(SignalConfig);
|
||||
|
||||
#ifndef _WIN32
|
||||
#elif !defined(_M_ARM_64EC)
|
||||
#if defined(_WIN32) && !defined(_M_ARM_64EC)
|
||||
// WOW64 always needs the interrupt fault check to be enabled.
|
||||
Config.NeedsPendingInterruptFaultCheck = true;
|
||||
#endif
|
||||
@@ -398,12 +365,6 @@ void ContextImpl::HandleCallback(FEXCore::Core::InternalThreadState* Thread, uin
|
||||
void ContextImpl::ExecuteThread(FEXCore::Core::InternalThreadState* Thread) {
|
||||
Dispatcher->ExecuteDispatch(Thread->CurrentFrame);
|
||||
|
||||
if (CodeObjectCacheService) {
|
||||
// Ensure the Code Object Serialization service has fully serialized this thread's data before clearing the cache
|
||||
// Use the thread's object cache ref counter for this
|
||||
CodeSerialize::CodeObjectSerializeService::WaitForEmptyJobQueue(&Thread->ObjectCacheRefCounter);
|
||||
}
|
||||
|
||||
// If it is the parent thread that died then just leave
|
||||
// TODO: This doesn't make sense when the parent thread doesn't outlive its children
|
||||
}
|
||||
@@ -441,22 +402,6 @@ ContextImpl::CreateThread(uint64_t InitialRIP, uint64_t StackPointer, const FEXC
|
||||
Thread->CurrentFrame->State.gregs[X86State::REG_RSP] = StackPointer;
|
||||
Thread->CurrentFrame->State.rip = InitialRIP;
|
||||
|
||||
// Set up default code segment.
|
||||
// Default code segment indexes match the numbers that the Linux kernel uses.
|
||||
Thread->CurrentFrame->State.cs_idx = 6 << 3;
|
||||
auto &GDT = Thread->CurrentFrame->State.gdt[Thread->CurrentFrame->State.cs_idx >> 3];
|
||||
Thread->CurrentFrame->State.SetGDTBase(&GDT, 0);
|
||||
Thread->CurrentFrame->State.SetGDTLimit(&GDT, 0xF'FFFFU);
|
||||
|
||||
if (Config.Is64BitMode) {
|
||||
GDT.L = 1; // L = Long Mode = 64-bit
|
||||
GDT.D = 0; // D = Default Operand SIze = Reserved
|
||||
}
|
||||
else {
|
||||
GDT.L = 0; // L = Long Mode = 32-bit
|
||||
GDT.D = 1; // D = Default Operand Size = 32-bit
|
||||
}
|
||||
|
||||
// Copy over the new thread state to the new object
|
||||
if (NewThreadState) {
|
||||
memcpy(&Thread->CurrentFrame->State, NewThreadState, sizeof(FEXCore::Core::CPUState));
|
||||
@@ -520,12 +465,6 @@ void ContextImpl::OnCodeBufferAllocated(CPU::CodeBuffer& Buffer) {
|
||||
void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer) {
|
||||
FEXCORE_PROFILE_INSTANT("ClearCodeCache");
|
||||
|
||||
if (CodeObjectCacheService) {
|
||||
// Ensure the Code Object Serialization service has fully serialized this thread's data before clearing the cache
|
||||
// Use the thread's object cache ref counter for this
|
||||
CodeSerialize::CodeObjectSerializeService::WaitForEmptyJobQueue(&Thread->ObjectCacheRefCounter);
|
||||
}
|
||||
|
||||
if (NewCodeBuffer) {
|
||||
// Allocate new CodeBuffer + L3 LookupCache and clear L1+L2 caches
|
||||
Thread->CPUBackend->ClearCache();
|
||||
@@ -579,7 +518,8 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
auto BlockInfo = Thread->FrontendDecoder->GetDecodedBlockInfo();
|
||||
auto CodeBlocks = &BlockInfo->Blocks;
|
||||
|
||||
Thread->OpDispatcher->BeginFunction(GuestRIP, CodeBlocks, BlockInfo->TotalInstructionCount, BlockInfo->Is64BitMode);
|
||||
Thread->OpDispatcher->BeginFunction(GuestRIP, CodeBlocks, BlockInfo->TotalInstructionCount, BlockInfo->Is64BitMode,
|
||||
AreMonoHacksActive() && MonoBackpatcherBlock.load(std::memory_order_relaxed) == GuestRIP);
|
||||
|
||||
const auto GPRSize = Thread->OpDispatcher->GetGPROpSize();
|
||||
|
||||
@@ -607,7 +547,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
if (InstsInBlock == 0) {
|
||||
// Special case for an empty instruction block.
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, Block.Entry - GuestRIP));
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_InlineEntrypointOffset(GPRSize, Block.Entry - GuestRIP));
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < InstsInBlock; ++i) {
|
||||
@@ -637,11 +577,12 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
Thread->OpDispatcher->_GuestOpcode(InstAddress - GuestRIP);
|
||||
}
|
||||
|
||||
if (Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL) {
|
||||
auto ExistingCodePtr = reinterpret_cast<uint64_t*>(Block.Entry + BlockInstructionsLength);
|
||||
|
||||
auto CodeChanged = Thread->OpDispatcher->_ValidateCode(ExistingCodePtr[0], ExistingCodePtr[1],
|
||||
(uintptr_t)ExistingCodePtr - GuestRIP, DecodedInfo->InstSize);
|
||||
if (Config.SMCChecks == FEXCore::Config::CONFIG_SMC_FULL || Block.ForceFullSMCDetection) {
|
||||
auto ExistingCodePtr = reinterpret_cast<uint8_t*>(Block.Entry + BlockInstructionsLength);
|
||||
auto InstAddressReg = Thread->OpDispatcher->_EntrypointOffset(GPRSize, InstAddress - GuestRIP);
|
||||
std::array<uint8_t, 0x10> CodeOriginal;
|
||||
memcpy(CodeOriginal.data(), ExistingCodePtr, DecodedInfo->InstSize);
|
||||
auto CodeChanged = Thread->OpDispatcher->_ValidateCode(CodeOriginal, InstAddressReg, DecodedInfo->InstSize);
|
||||
|
||||
auto InvalidateCodeCond = Thread->OpDispatcher->CondJump(CodeChanged);
|
||||
|
||||
@@ -651,7 +592,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
Thread->OpDispatcher->SetCurrentCodeBlock(CodeWasChangedBlock);
|
||||
Thread->OpDispatcher->_ThreadRemoveCodeEntry();
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, InstAddress - GuestRIP));
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_InlineEntrypointOffset(GPRSize, InstAddress - GuestRIP));
|
||||
|
||||
auto NextOpBlock = Thread->OpDispatcher->CreateNewCodeBlockAfter(CurrentBlock);
|
||||
|
||||
@@ -659,15 +600,21 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
Thread->OpDispatcher->SetCurrentCodeBlock(NextOpBlock);
|
||||
}
|
||||
|
||||
if (TableInfo && TableInfo->OpcodeDispatcher) {
|
||||
auto Fn = TableInfo->OpcodeDispatcher;
|
||||
if (TableInfo && TableInfo->OpcodeDispatcher.OpDispatch) {
|
||||
auto Fn = TableInfo->OpcodeDispatcher.OpDispatch;
|
||||
Thread->OpDispatcher->ResetHandledLock();
|
||||
Thread->OpDispatcher->ResetDecodeFailure();
|
||||
IR::ForceTSOMode ForceTSO =
|
||||
BlockInForceTSOValidRange ?
|
||||
(InstForceTSOIt != ForceTSOInstructions.end() && *InstForceTSOIt == InstAddress ? IR::ForceTSOMode::ForceEnabled :
|
||||
IR::ForceTSOMode::ForceDisabled) :
|
||||
IR::ForceTSOMode::NoOverride;
|
||||
IR::ForceTSOMode ForceTSO = IR::ForceTSOMode::NoOverride;
|
||||
if (BlockInForceTSOValidRange) {
|
||||
if (InstForceTSOIt != ForceTSOInstructions.end() && *InstForceTSOIt == InstAddress) {
|
||||
ForceTSO = IR::ForceTSOMode::ForceEnabled;
|
||||
} else {
|
||||
ForceTSO = IR::ForceTSOMode::ForceDisabled;
|
||||
}
|
||||
} else if (DecodedInfo->Flags & X86Tables::DecodeFlags::FLAG_FORCE_TSO) {
|
||||
ForceTSO = IR::ForceTSOMode::ForceEnabled;
|
||||
}
|
||||
|
||||
Thread->OpDispatcher->SetForceTSO(ForceTSO);
|
||||
std::invoke(Fn, Thread->OpDispatcher, DecodedInfo);
|
||||
if (Thread->OpDispatcher->HadDecodeFailure()) {
|
||||
@@ -717,7 +664,8 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
if (NeedsBlockEnd) {
|
||||
// We had some instructions. Early exit
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, Block.Entry + BlockInstructionsLength - GuestRIP));
|
||||
Thread->OpDispatcher->ExitFunction(
|
||||
Thread->OpDispatcher->_InlineEntrypointOffset(GPRSize, Block.Entry + BlockInstructionsLength - GuestRIP));
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -760,27 +708,10 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
}
|
||||
|
||||
ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst) {
|
||||
// JIT Code object cache lookup
|
||||
if (CodeObjectCacheService) {
|
||||
auto CodeCacheEntry = CodeObjectCacheService->FetchCodeObjectFromCache(GuestRIP);
|
||||
if (CodeCacheEntry) {
|
||||
auto CompiledCode = Thread->CPUBackend->RelocateJITObjectCode(GuestRIP, CodeCacheEntry);
|
||||
if (CompiledCode) {
|
||||
return {
|
||||
.CompiledCode = {},
|
||||
.DebugData = nullptr, // nullptr here ensures that code serialization doesn't occur on from cache read
|
||||
.StartAddr = 0, // Unused
|
||||
.Length = 0, // Unused
|
||||
.NeedsAddGuestCodeRanges = false,
|
||||
};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (SourcecodeResolver && Config.GDBSymbols()) {
|
||||
auto AOTIRCacheEntry = SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
|
||||
if (AOTIRCacheEntry.Entry && !AOTIRCacheEntry.Entry->ContainsCode) {
|
||||
AOTIRCacheEntry.Entry->SourcecodeMap = SourcecodeResolver->GenerateMap(AOTIRCacheEntry.Entry->Filename, AOTIRCacheEntry.Entry->FileId);
|
||||
auto MappedSection = SyscallHandler->LookupExecutableFileSection(*Thread, GuestRIP);
|
||||
if (MappedSection) {
|
||||
MappedSection->FileInfo.SourcecodeMap = SourcecodeResolver->GenerateMap(MappedSection->FileInfo.Filename, MappedSection->FileInfo.FileId);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -855,51 +786,44 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
if (Config.BlockJITNaming()) {
|
||||
auto FragmentBasePtr = CompiledCode.BlockBegin;
|
||||
|
||||
if (DebugData) {
|
||||
auto GuestRIPLookup = SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
|
||||
auto GuestRIPLookup = SyscallHandler->LookupExecutableFileSection(*Thread, GuestRIP);
|
||||
|
||||
if (DebugData->Subblocks.size()) {
|
||||
for (auto& Subblock : DebugData->Subblocks) {
|
||||
auto BlockBasePtr = FragmentBasePtr + Subblock.HostCodeOffset;
|
||||
if (GuestRIPLookup.Entry) {
|
||||
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, CompiledCode.Size, GuestRIPLookup.Entry->Filename,
|
||||
GuestRIP - GuestRIPLookup.VAFileStart);
|
||||
} else {
|
||||
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, GuestRIP, Subblock.HostCodeSize);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
if (GuestRIPLookup.Entry) {
|
||||
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, CompiledCode.Size, GuestRIPLookup.Entry->Filename,
|
||||
GuestRIP - GuestRIPLookup.VAFileStart);
|
||||
if (DebugData->Subblocks.size()) {
|
||||
for (auto& Subblock : DebugData->Subblocks) {
|
||||
auto BlockBasePtr = FragmentBasePtr + Subblock.HostCodeOffset;
|
||||
if (GuestRIPLookup) {
|
||||
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, CompiledCode.Size, GuestRIPLookup->FileInfo.Filename,
|
||||
GuestRIP - GuestRIPLookup->FileStartVA);
|
||||
} else {
|
||||
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, GuestRIP, CompiledCode.Size);
|
||||
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, GuestRIP, Subblock.HostCodeSize);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
if (GuestRIPLookup) {
|
||||
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, CompiledCode.Size, GuestRIPLookup->FileInfo.Filename,
|
||||
GuestRIP - GuestRIPLookup->FileStartVA);
|
||||
} else {
|
||||
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, GuestRIP, CompiledCode.Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Tell the object cache service to serialize the code if enabled
|
||||
if (CodeObjectCacheService && Config.CacheObjectCodeCompilation == FEXCore::Config::ConfigObjectCodeHandler::CONFIG_READWRITE && DebugData) {
|
||||
CodeObjectCacheService->AsyncAddSerializationJob(
|
||||
fextl::make_unique<CodeSerialize::AsyncJobHandler::SerializationJobData>(CodeSerialize::AsyncJobHandler::SerializationJobData {
|
||||
.GuestRIP = GuestRIP,
|
||||
.GuestCodeLength = Length,
|
||||
.GuestCodeHash = 0,
|
||||
.HostCodeBegin = CompiledCode.BlockBegin,
|
||||
.HostCodeLength = CompiledCode.Size,
|
||||
.HostCodeHash = 0,
|
||||
.ThreadJobRefCount = &Thread->ObjectCacheRefCounter,
|
||||
.Relocations = std::move(*DebugData->Relocations),
|
||||
}));
|
||||
if (Config.LibraryJITNaming() || Config.GDBSymbols()) {
|
||||
auto MappedSection = SyscallHandler->LookupExecutableFileSection(*Thread, GuestRIP);
|
||||
if (MappedSection) {
|
||||
if (Config.LibraryJITNaming()) {
|
||||
Symbols.RegisterNamedRegion(Thread->SymbolBuffer.get(), CodePtr, DebugData->HostCodeSize, MappedSection->FileInfo.Filename);
|
||||
}
|
||||
|
||||
if (Config.GDBSymbols()) {
|
||||
GDBJITRegister(MappedSection->FileInfo, MappedSection->FileStartVA, GuestRIP, (uintptr_t)CodePtr, *DebugData);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Clear any relocations that might have been generated
|
||||
Thread->CPUBackend->ClearRelocations();
|
||||
|
||||
if (IRCaptureCache.PostCompileCode(Thread, CompiledCode.BlockBegin, GuestRIP, StartAddr, Length, {}, DebugData.get(), false)) {
|
||||
// Early exit
|
||||
return (uintptr_t)CodePtr;
|
||||
if (!CodeCache.IsGeneratingCache) {
|
||||
Thread->CPUBackend->ClearRelocations();
|
||||
}
|
||||
|
||||
if (NeedsAddGuestCodeRanges) {
|
||||
@@ -978,23 +902,6 @@ void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* T
|
||||
InvalidateGuestThreadCodeRange(Thread, Accumulator, Start, Length);
|
||||
}
|
||||
|
||||
void ContextImpl::MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) {
|
||||
if (!Thread) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (!IsMemoryShared) {
|
||||
IsMemoryShared = true;
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
|
||||
if (Config.TSOAutoMigration) {
|
||||
// Only the lookup cache is cleared here, so that old code can keep running until next compilation.
|
||||
// This will leak previously compiled blocks until the CodeBuffer is cleared for some other reason.
|
||||
Thread->LookupCache->ClearCache();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
bool ContextImpl::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) {
|
||||
LogMan::Throw::AFmt(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to "
|
||||
"be unique_locked here");
|
||||
@@ -1002,6 +909,10 @@ bool ContextImpl::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thre
|
||||
return Thread->LookupCache->Erase(Thread->CurrentFrame, GuestRIP);
|
||||
}
|
||||
|
||||
void ContextImpl::ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP) {
|
||||
static_cast<ContextImpl*>(Frame->Thread->CTX)->SyscallHandler->InvalidateGuestCodeRange(Frame->Thread, GuestRIP, 1);
|
||||
}
|
||||
|
||||
std::optional<CustomIRResult>
|
||||
ContextImpl::AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void* Creator, void* Data) {
|
||||
LOGMAN_THROW_A_FMT(Config.Is64BitMode || !(Entrypoint >> 32), "64-bit Entrypoint in 32-bit mode {:x}", Entrypoint);
|
||||
@@ -1075,25 +986,36 @@ void ContextImpl::RemoveForceTSOInformation(uint64_t Address, uint64_t Size) {
|
||||
ForceTSOInstructions.erase(ForceTSOInstructions.lower_bound(Address), ForceTSOInstructions.upper_bound(Address + Size));
|
||||
}
|
||||
|
||||
void ContextImpl::RemoveCustomIREntrypoint(uintptr_t Entrypoint) {
|
||||
void ContextImpl::MarkMonoBackpatcherBlock(uint64_t BlockEntry) {
|
||||
MonoBackpatcherBlock.store(BlockEntry, std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
void ContextImpl::RemoveCustomIREntrypoint(FEXCore::Core::InternalThreadState* Thread, uintptr_t Entrypoint) {
|
||||
LOGMAN_THROW_A_FMT(Config.Is64BitMode || !(Entrypoint >> 32), "64-bit Entrypoint in 32-bit mode {:x}", Entrypoint);
|
||||
|
||||
std::scoped_lock lk(CustomIRMutex);
|
||||
|
||||
InvalidatedEntryAccumulator Accumulator;
|
||||
InvalidateGuestCodeRange(nullptr, Accumulator, Entrypoint, 1);
|
||||
CustomIRHandlers.erase(Entrypoint);
|
||||
|
||||
HasCustomIRHandlers = !CustomIRHandlers.empty();
|
||||
SyscallHandler->InvalidateGuestCodeRange(Thread, Entrypoint, 1);
|
||||
}
|
||||
|
||||
IR::AOTIRCacheEntry* ContextImpl::LoadAOTIRCacheEntry(const fextl::string& filename) {
|
||||
auto rv = IRCaptureCache.LoadAOTIRCacheEntry(filename);
|
||||
return rv;
|
||||
}
|
||||
void ContextImpl::MonoBackpatcherWrite(FEXCore::Core::CpuStateFrame* Frame, uint8_t Size, uint64_t Address, uint64_t Value) {
|
||||
auto Thread = Frame->Thread;
|
||||
auto CTX = static_cast<ContextImpl*>(Thread->CTX);
|
||||
{
|
||||
auto lk = GuardSignalDeferringSection(CTX->CodeInvalidationMutex, Thread);
|
||||
|
||||
void ContextImpl::UnloadAOTIRCacheEntry(IR::AOTIRCacheEntry* Entry) {
|
||||
IRCaptureCache.UnloadAOTIRCacheEntry(Entry);
|
||||
if (Size == 8) {
|
||||
*reinterpret_cast<uint64_t*>(Address) = Value;
|
||||
} else if (Size == 4) {
|
||||
*reinterpret_cast<uint32_t*>(Address) = Value;
|
||||
} else {
|
||||
ERROR_AND_DIE_FMT("Unexpected write size for backpatcher: {}", Size);
|
||||
}
|
||||
}
|
||||
|
||||
CTX->SyscallHandler->InvalidateGuestCodeRange(Thread, Address, Size);
|
||||
}
|
||||
|
||||
void ContextImpl::ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) {
|
||||
|
||||
@@ -1,7 +1,8 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#include "Common/SoftFloat.h"
|
||||
#include "Common/VectorRegType.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
@@ -16,14 +17,18 @@
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <condition_variable>
|
||||
#ifdef VIXL_SIMULATOR
|
||||
#include <aarch64/simulator-aarch64.h>
|
||||
#endif
|
||||
|
||||
#include <array>
|
||||
#include <bit>
|
||||
#include <csignal>
|
||||
#include <cstring>
|
||||
#include <signal.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
@@ -86,7 +91,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
FillStaticRegs();
|
||||
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
|
||||
cbnz(ARMEmitter::Size::i32Bit, ENTRY_FILL_SRA_SINGLE_INST_REG, &CompileSingleStep);
|
||||
(void)cbnz(ARMEmitter::Size::i32Bit, ENTRY_FILL_SRA_SINGLE_INST_REG, &CompileSingleStep);
|
||||
|
||||
ARMEmitter::BiDirectionalLabel LoopTop {};
|
||||
|
||||
@@ -134,31 +139,39 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
// We want to ensure that we are 16 byte aligned at the top of this loop
|
||||
Align16B();
|
||||
ARMEmitter::BiDirectionalLabel FullLookup {};
|
||||
ARMEmitter::BiDirectionalLabel CallBlock {};
|
||||
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
AbsoluteLoopTopAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
// Load in our RIP
|
||||
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
// Clobbers TMP1/2
|
||||
// Check the EC code bitmap incase we need to exit the JIT to call into native code.
|
||||
ARMEmitter::ForwardLabel l_NotECCode;
|
||||
ldr(TMP1, ARMEmitter::XReg::x18, TEB_PEB_OFFSET);
|
||||
ldr(TMP1, TMP1, PEB_EC_CODE_BITMAP_OFFSET);
|
||||
|
||||
lsr(ARMEmitter::Size::i64Bit, TMP2, RipReg, 15);
|
||||
and_(ARMEmitter::Size::i64Bit, TMP2, TMP2, 0x1fffffffffff8);
|
||||
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
lsr(ARMEmitter::Size::i64Bit, TMP2, RipReg, 12);
|
||||
lsrv(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP2);
|
||||
tbz(TMP1, 0, &l_NotECCode);
|
||||
|
||||
str(REG_CALLRET_SP, STATE_PTR(CpuStateFrame, State.callret_sp));
|
||||
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, StaticRegisters[X86State::REG_RSP], 0);
|
||||
mov(EC_CALL_CHECKER_PC_REG, RipReg);
|
||||
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionEC));
|
||||
br(TMP2);
|
||||
|
||||
(void)!Bind(&l_NotECCode);
|
||||
#endif
|
||||
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
cbnz(ARMEmitter::Size::i32Bit, TMP1, &CompileSingleStep);
|
||||
|
||||
// L1 Cache
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
|
||||
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP4, ARMEmitter::ShiftType::LSL, 4);
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP4, TMP1, TMP1, 0);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, RipReg);
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP1, &FullLookup);
|
||||
|
||||
br(TMP4);
|
||||
|
||||
// L1C check failed, do a full lookup
|
||||
Bind(&FullLookup);
|
||||
(void)cbnz(ARMEmitter::Size::i32Bit, TMP1, &CompileSingleStep);
|
||||
|
||||
// This is the block cache lookup routine
|
||||
// It matches what is going on it LookupCache.h::FindBlock
|
||||
@@ -183,7 +196,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 3);
|
||||
|
||||
// If page pointer is zero then we have no block
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &NoBlock);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &NoBlock);
|
||||
|
||||
// Steal the page offset
|
||||
and_(ARMEmitter::Size::i64Bit, TMP2, TMP4, 0x0FFF);
|
||||
@@ -198,10 +211,11 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
// If the guest address doesn't match, Compile the block.
|
||||
sub(TMP2, TMP2, RipReg);
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP2, &NoBlock);
|
||||
(void)cbnz(ARMEmitter::Size::i64Bit, TMP2, &NoBlock);
|
||||
|
||||
|
||||
// Check the host address to see if it matches, else compile the block.
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP4, &NoBlock);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP4, &NoBlock);
|
||||
|
||||
// If we've made it here then we have a real compiled block
|
||||
{
|
||||
@@ -287,39 +301,9 @@ void Dispatcher::EmitDispatcher() {
|
||||
br(TMP1);
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
// Clobbers TMP1/2
|
||||
auto EmitECExitCheck = [&]() {
|
||||
// Check the EC code bitmap incase we need to exit the JIT to call into native code.
|
||||
ARMEmitter::ForwardLabel l_NotECCode;
|
||||
ldr(TMP1, ARMEmitter::XReg::x18, TEB_PEB_OFFSET);
|
||||
ldr(TMP1, TMP1, PEB_EC_CODE_BITMAP_OFFSET);
|
||||
|
||||
lsr(ARMEmitter::Size::i64Bit, TMP2, RipReg, 15);
|
||||
and_(ARMEmitter::Size::i64Bit, TMP2, TMP2, 0x1fffffffffff8);
|
||||
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
lsr(ARMEmitter::Size::i64Bit, TMP2, RipReg, 12);
|
||||
lsrv(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP2);
|
||||
tbz(TMP1, 0, &l_NotECCode);
|
||||
|
||||
str(REG_CALLRET_SP, STATE_PTR(CpuStateFrame, State.callret_sp));
|
||||
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, StaticRegisters[X86State::REG_RSP], 0);
|
||||
mov(EC_CALL_CHECKER_PC_REG, RipReg);
|
||||
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionEC));
|
||||
br(TMP2);
|
||||
|
||||
Bind(&l_NotECCode);
|
||||
};
|
||||
#endif
|
||||
|
||||
// Need to create the block
|
||||
{
|
||||
Bind(&NoBlock);
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
EmitECExitCheck();
|
||||
#endif
|
||||
(void)Bind(&NoBlock);
|
||||
|
||||
EmitSignalGuardedRegion([&]() {
|
||||
SpillStaticRegs(TMP1);
|
||||
@@ -353,11 +337,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
}
|
||||
|
||||
{
|
||||
Bind(&CompileSingleStep);
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
EmitECExitCheck();
|
||||
#endif
|
||||
(void)Bind(&CompileSingleStep);
|
||||
|
||||
EmitSignalGuardedRegion([&]() {
|
||||
SpillStaticRegs(TMP1);
|
||||
@@ -519,7 +499,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::zr, ARMEmitter::XReg::zr, REG_CALLRET_SP, -0x10);
|
||||
|
||||
// Now go back to the regular dispatcher loop
|
||||
b(&LoopTop);
|
||||
(void)b(&LoopTop);
|
||||
}
|
||||
|
||||
auto EmitLongALUOpHandler = [&](auto R, auto Offset) {
|
||||
@@ -569,8 +549,8 @@ void Dispatcher::EmitDispatcher() {
|
||||
FABI_F80_I16_I32_PTR,
|
||||
FABI_F32_I16_F80_PTR,
|
||||
FABI_F64_I16_F80_PTR,
|
||||
FABI_F64_I16_F64_PTR,
|
||||
FABI_F64_I16_F64_F64_PTR,
|
||||
FABI_F64_F64_PTR,
|
||||
FABI_F64_F64_F64_PTR,
|
||||
FABI_I16_I16_F80_PTR,
|
||||
FABI_I32_I16_F80_PTR,
|
||||
FABI_I64_I16_F80_PTR,
|
||||
@@ -578,7 +558,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
FABI_F80_I16_F80_PTR,
|
||||
FABI_F80_I16_F80_F80_PTR,
|
||||
FABI_F80x2_I16_F80_PTR,
|
||||
FABI_F64x2_I16_F64_PTR,
|
||||
FABI_F64x2_F64_PTR,
|
||||
FABI_I32_I64_I64_V128_V128_I16,
|
||||
FABI_I32_V128_V128_I16,
|
||||
}};
|
||||
@@ -588,14 +568,15 @@ void Dispatcher::EmitDispatcher() {
|
||||
}
|
||||
}
|
||||
|
||||
Bind(&l_CTX);
|
||||
(void)Bind(&l_CTX);
|
||||
dc64(reinterpret_cast<uintptr_t>(CTX));
|
||||
Bind(&l_Sleep);
|
||||
(void)Bind(&l_Sleep);
|
||||
dc64(reinterpret_cast<uint64_t>(SleepThread));
|
||||
Bind(&l_CompileBlock);
|
||||
(void)Bind(&l_CompileBlock);
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMFCompileBlock(&FEXCore::Context::ContextImpl::CompileBlock);
|
||||
dc64(PMFCompileBlock.GetConvertedPointer());
|
||||
Bind(&l_CompileSingleStep);
|
||||
(void)Bind(&l_CompileSingleStep);
|
||||
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMFCompileSingleStep(&FEXCore::Context::ContextImpl::CompileSingleStep);
|
||||
dc64(PMFCompileSingleStep.GetConvertedPointer());
|
||||
|
||||
@@ -777,7 +758,7 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
if (!TMP_ABIARGS) {
|
||||
fmov(VABI1.D(), VTMP1.D());
|
||||
mov(VABI1.Q(), VTMP1.Q());
|
||||
}
|
||||
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
@@ -810,7 +791,7 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
|
||||
|
||||
FillF64Result();
|
||||
} break;
|
||||
case FABI_F64_I16_F64_PTR: {
|
||||
case FABI_F64_F64_PTR: {
|
||||
// Linux Reg/Win32 Reg:
|
||||
// tmp4 (x4/x13): FallbackHandler
|
||||
// x30: return
|
||||
@@ -820,18 +801,17 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
|
||||
if (!TMP_ABIARGS) {
|
||||
fmov(VABI1.D(), VTMP1.D());
|
||||
}
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
mov(ARMEmitter::XReg::x0, STATE);
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, double, uint64_t>(FallbackPointerReg);
|
||||
GenerateIndirectRuntimeCall<double, double, uint64_t>(FallbackPointerReg);
|
||||
} else {
|
||||
blr(FallbackPointerReg);
|
||||
}
|
||||
|
||||
FillF64Result();
|
||||
} break;
|
||||
case FABI_F64_I16_F64_F64_PTR: {
|
||||
case FABI_F64_F64_F64_PTR: {
|
||||
// Linux Reg/Win32 Reg:
|
||||
// tmp4 (x4/x13): FallbackHandler
|
||||
// x30: return
|
||||
@@ -844,10 +824,9 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
|
||||
fmov(VABI2.D(), VTMP2.D());
|
||||
}
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
mov(ARMEmitter::XReg::x0, STATE);
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, double, double, uint64_t>(FallbackPointerReg);
|
||||
GenerateIndirectRuntimeCall<double, double, double, uint64_t>(FallbackPointerReg);
|
||||
} else {
|
||||
blr(FallbackPointerReg);
|
||||
}
|
||||
@@ -1007,7 +986,7 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
|
||||
|
||||
FillF80x2Result();
|
||||
} break;
|
||||
case FABI_F64x2_I16_F64_PTR: {
|
||||
case FABI_F64x2_F64_PTR: {
|
||||
// Linux Reg/Win32 Reg:
|
||||
// tmp4 (x4/x13): FallbackHandler
|
||||
// x30: return
|
||||
@@ -1016,14 +995,13 @@ uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
|
||||
|
||||
SpillForABICall(CTX->HostFeatures.SupportsPreserveAllABI, TMP3, true);
|
||||
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
mov(ARMEmitter::XReg::x0, STATE);
|
||||
if (!TMP_ABIARGS) {
|
||||
fmov(VABI1.D(), VTMP1.D());
|
||||
}
|
||||
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
// GenerateIndirectRuntimeCall<FEXCore::VectorScalarF64Pair, uint16_t, FEXCore::VectorRegType, uint64_t>(FallbackPointerReg);
|
||||
// GenerateIndirectRuntimeCall<FEXCore::VectorScalarF64Pair, FEXCore::VectorRegType, uint64_t>(FallbackPointerReg);
|
||||
} else {
|
||||
blr(FallbackPointerReg);
|
||||
}
|
||||
@@ -1125,6 +1103,53 @@ void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState* Thread)
|
||||
}
|
||||
}
|
||||
|
||||
SignalDelegatorConfig Dispatcher::MakeSignalDelegatorConfig() const {
|
||||
// PF/AF are the final two SRA registers. We only want GPRs
|
||||
const auto GPRCount = uint16_t(StaticRegisters.size() - 2);
|
||||
const auto FPRCount = uint16_t(StaticFPRegisters.size());
|
||||
|
||||
const auto GetSRAGPRMapping = [GPRCount, this] {
|
||||
SignalDelegatorConfig::SRAIndexMapping Mapping {};
|
||||
for (size_t i = 0; i < GPRCount; ++i) {
|
||||
Mapping[i] = StaticRegisters[i].Idx();
|
||||
}
|
||||
return Mapping;
|
||||
};
|
||||
|
||||
const auto GetSRAFPRMapping = [FPRCount, this] {
|
||||
SignalDelegatorConfig::SRAIndexMapping Mapping {};
|
||||
for (size_t i = 0; i < FPRCount; ++i) {
|
||||
Mapping[i] = StaticFPRegisters[i].Idx();
|
||||
}
|
||||
return Mapping;
|
||||
};
|
||||
|
||||
return FEXCore::SignalDelegatorConfig {
|
||||
.DispatcherBegin = Start,
|
||||
.DispatcherEnd = End,
|
||||
|
||||
.AbsoluteLoopTopAddress = AbsoluteLoopTopAddress,
|
||||
.AbsoluteLoopTopAddressFillSRA = AbsoluteLoopTopAddressFillSRA,
|
||||
.SignalHandlerReturnAddress = SignalHandlerReturnAddress,
|
||||
.SignalHandlerReturnAddressRT = SignalHandlerReturnAddressRT,
|
||||
|
||||
.PauseReturnInstruction = PauseReturnInstruction,
|
||||
.ThreadPauseHandlerAddressSpillSRA = ThreadPauseHandlerAddressSpillSRA,
|
||||
.ThreadPauseHandlerAddress = ThreadPauseHandlerAddress,
|
||||
|
||||
// Stop handlers.
|
||||
.ThreadStopHandlerAddressSpillSRA = ThreadStopHandlerAddressSpillSRA,
|
||||
.ThreadStopHandlerAddress = ThreadStopHandlerAddress,
|
||||
|
||||
// SRA information.
|
||||
.SRAGPRCount = GPRCount,
|
||||
.SRAFPRCount = FPRCount,
|
||||
|
||||
.SRAGPRMapping = GetSRAGPRMapping(),
|
||||
.SRAFPRMapping = GetSRAFPRMapping(),
|
||||
};
|
||||
}
|
||||
|
||||
fextl::unique_ptr<Dispatcher> Dispatcher::Create(FEXCore::Context::ContextImpl* CTX) {
|
||||
return fextl::make_unique<Dispatcher>(CTX);
|
||||
}
|
||||
|
||||
@@ -2,25 +2,18 @@
|
||||
#pragma once
|
||||
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
#include <aarch64/simulator-aarch64.h>
|
||||
#endif
|
||||
|
||||
#include <array>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <signal.h>
|
||||
#include <stddef.h>
|
||||
#include <stack>
|
||||
#include <tuple>
|
||||
|
||||
namespace FEXCore {
|
||||
struct GuestSigAction;
|
||||
}
|
||||
struct SignalDelegatorConfig;
|
||||
} // namespace FEXCore
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct CpuStateFrame;
|
||||
@@ -42,6 +35,32 @@ public:
|
||||
Dispatcher(FEXCore::Context::ContextImpl* ctx);
|
||||
~Dispatcher();
|
||||
|
||||
void InitThreadPointers(FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
void ExecuteDispatch(FEXCore::Core::CpuStateFrame* Frame);
|
||||
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP);
|
||||
#else
|
||||
void ExecuteDispatch(FEXCore::Core::CpuStateFrame* Frame) {
|
||||
DispatchPtr(Frame, false);
|
||||
}
|
||||
|
||||
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP) {
|
||||
CallbackPtr(Frame, RIP);
|
||||
}
|
||||
#endif
|
||||
|
||||
SignalDelegatorConfig MakeSignalDelegatorConfig() const;
|
||||
|
||||
protected:
|
||||
FEXCore::Context::ContextImpl* CTX;
|
||||
|
||||
using AsmDispatch = void (*)(FEXCore::Core::CpuStateFrame* Frame, bool SingleInst);
|
||||
using JITCallback = void (*)(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP);
|
||||
|
||||
AsmDispatch DispatchPtr;
|
||||
JITCallback CallbackPtr;
|
||||
private:
|
||||
/**
|
||||
* @name Dispatch Helper functions
|
||||
* @{ */
|
||||
@@ -59,62 +78,14 @@ public:
|
||||
uint64_t GuestSignal_SIGILL {};
|
||||
uint64_t GuestSignal_SIGTRAP {};
|
||||
uint64_t GuestSignal_SIGSEGV {};
|
||||
uint64_t IntCallbackReturnAddress {};
|
||||
|
||||
uint64_t PauseReturnInstruction {};
|
||||
std::array<uint64_t, FallbackABI::FABI_UNKNOWN> ABIPointers {};
|
||||
|
||||
/** @} */
|
||||
|
||||
uint64_t Start {};
|
||||
uint64_t End {};
|
||||
|
||||
void InitThreadPointers(FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
void ExecuteDispatch(FEXCore::Core::CpuStateFrame* Frame);
|
||||
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP);
|
||||
#else
|
||||
void ExecuteDispatch(FEXCore::Core::CpuStateFrame* Frame) {
|
||||
DispatchPtr(Frame, false);
|
||||
}
|
||||
|
||||
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP) {
|
||||
CallbackPtr(Frame, RIP);
|
||||
}
|
||||
#endif
|
||||
|
||||
uint16_t GetSRAGPRCount() const {
|
||||
// PF/AF are the final two SRA registers.
|
||||
// Only return the SRA for GPRs.
|
||||
return StaticRegisters.size() - 2;
|
||||
}
|
||||
|
||||
uint16_t GetSRAFPRCount() const {
|
||||
return StaticFPRegisters.size();
|
||||
}
|
||||
|
||||
void GetSRAGPRMapping(uint8_t Mapping[16]) const {
|
||||
for (size_t i = 0; i < StaticRegisters.size() - 2; ++i) {
|
||||
Mapping[i] = StaticRegisters[i].Idx();
|
||||
}
|
||||
}
|
||||
|
||||
void GetSRAFPRMapping(uint8_t Mapping[16]) const {
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); ++i) {
|
||||
Mapping[i] = StaticFPRegisters[i].Idx();
|
||||
}
|
||||
}
|
||||
|
||||
protected:
|
||||
FEXCore::Context::ContextImpl* CTX;
|
||||
|
||||
using AsmDispatch = void (*)(FEXCore::Core::CpuStateFrame* Frame, bool SingleInst);
|
||||
using JITCallback = void (*)(FEXCore::Core::CpuStateFrame* Frame, uint64_t RIP);
|
||||
|
||||
AsmDispatch DispatchPtr;
|
||||
JITCallback CallbackPtr;
|
||||
private:
|
||||
// Long division helpers
|
||||
uint64_t LUDIVHandlerAddress {};
|
||||
uint64_t LDIVHandlerAddress {};
|
||||
|
||||
@@ -71,7 +71,23 @@ Decoder::Decoder(FEXCore::Core::InternalThreadState* Thread)
|
||||
: Thread {Thread}
|
||||
, CTX {static_cast<FEXCore::Context::ContextImpl*>(Thread->CTX)}
|
||||
, OSABI {CTX->SyscallHandler ? CTX->SyscallHandler->GetOSABI() : FEXCore::HLE::SyscallOSABI::OS_UNKNOWN}
|
||||
, PoolObject {CTX->FrontendAllocator, sizeof(FEXCore::X86Tables::DecodedInst) * DefaultDecodedBufferSize} {}
|
||||
, PoolObject {CTX->FrontendAllocator, sizeof(FEXCore::X86Tables::DecodedInst) * DefaultDecodedBufferSize} {
|
||||
|
||||
FEX_CONFIG_OPT(ReducedPrecision, X87REDUCEDPRECISION);
|
||||
if (ReducedPrecision) {
|
||||
X87Table = &FEXCore::X86Tables::X87F64Ops;
|
||||
} else {
|
||||
X87Table = &FEXCore::X86Tables::X87F80Ops;
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsAVX && CTX->HostFeatures.SupportsSVE256) {
|
||||
VEXTable = &FEXCore::X86Tables::VEXTableOps;
|
||||
VEXTableGroup = &FEXCore::X86Tables::VEXTableGroupOps;
|
||||
} else if (CTX->HostFeatures.SupportsAVX) {
|
||||
VEXTable = &FEXCore::X86Tables::VEXTableOps_AVX128;
|
||||
VEXTableGroup = &FEXCore::X86Tables::VEXTableGroupOps_AVX128;
|
||||
}
|
||||
}
|
||||
|
||||
bool Decoder::CheckRangeExecutable(uint64_t Address, uint64_t Size) {
|
||||
// Treat FEX-internal X86 callbacks as always executable
|
||||
@@ -83,6 +99,7 @@ bool Decoder::CheckRangeExecutable(uint64_t Address, uint64_t Size) {
|
||||
auto RangeInfo = CTX->SyscallHandler->QueryGuestExecutableRange(Thread, Address);
|
||||
ExecutableRangeBase = RangeInfo.Base;
|
||||
ExecutableRangeEnd = RangeInfo.Base + RangeInfo.Size;
|
||||
ExecutableRangeWritable = RangeInfo.Writable;
|
||||
|
||||
if (RangeInfo.Size == 0) {
|
||||
return false;
|
||||
@@ -242,13 +259,13 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
|
||||
|
||||
if (HasSIB) {
|
||||
FEXCore::X86Tables::SIBDecoded SIB;
|
||||
if (DecodeInst->DecodedSIB) {
|
||||
if (DecodeInst->Flags & DecodeFlags::FLAG_DECODED_SIB) {
|
||||
SIB.Hex = DecodeInst->SIB;
|
||||
} else {
|
||||
// Haven't yet grabbed SIB, pull it now
|
||||
DecodeInst->SIB = ReadByte();
|
||||
SIB.Hex = DecodeInst->SIB;
|
||||
DecodeInst->DecodedSIB = true;
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_SIB;
|
||||
}
|
||||
|
||||
// If the SIB base is 0b101, aka BP or R13 then we have a 32bit displacement
|
||||
@@ -314,6 +331,13 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
|
||||
}
|
||||
|
||||
bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op, DecodedHeader Options) {
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_ARCH_DISPATCHER) [[unlikely]] {
|
||||
// Dispatcher Op.
|
||||
// TODO: Move this in to `NormalOpHeader`, Dispatch tables have a bug currently where some subtables don't inherit flags correctly.
|
||||
// Can be seen by running FEX asm tests if this is removed.
|
||||
return NormalOp(&Info->OpcodeDispatcher.Indirect[BlockInfo.Is64BitMode ? 1 : 0], Op);
|
||||
}
|
||||
|
||||
DecodeInst->OP = Op;
|
||||
DecodeInst->TableInfo = Info;
|
||||
|
||||
@@ -377,9 +401,9 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
|
||||
// If we require ModRM and haven't decoded it yet, do it now
|
||||
// Some instructions have to read modrm upfront, others do it later
|
||||
if (HasMODRM && !DecodeInst->DecodedModRM) {
|
||||
if (HasMODRM && !(DecodeInst->Flags & DecodeFlags::FLAG_DECODED_MODRM)) {
|
||||
DecodeInst->ModRM = ReadByte();
|
||||
DecodeInst->DecodedModRM = true;
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
|
||||
}
|
||||
|
||||
// New instruction size decoding
|
||||
@@ -412,9 +436,8 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
// If the default operating mode is 32bit and we have the operand size flag then the operating size drops to 16bit
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_16BIT);
|
||||
DestSize = 2;
|
||||
} else if ((HasXMMDst || HasMMDst || BlockInfo.Is64BitMode) &&
|
||||
(HasWideningDisplacement || DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BIT ||
|
||||
DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BITDEF)) {
|
||||
} else if ((HasXMMDst || HasMMDst || BlockInfo.Is64BitMode) && (HasWideningDisplacement || DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BIT ||
|
||||
DstSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BITDEF)) {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeDstSize(DecodeFlags::SIZE_64BIT);
|
||||
DestSize = 8;
|
||||
} else {
|
||||
@@ -441,9 +464,8 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
// See table 1-2. Operand-Size Overrides for this decoding
|
||||
// If the default operating mode is 32bit and we have the operand size flag then the operating size drops to 16bit
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_16BIT);
|
||||
} else if ((HasXMMSrc || HasMMSrc || BlockInfo.Is64BitMode) &&
|
||||
(HasWideningDisplacement || SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BIT ||
|
||||
SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BITDEF)) {
|
||||
} else if ((HasXMMSrc || HasMMSrc || BlockInfo.Is64BitMode) && (HasWideningDisplacement || SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BIT ||
|
||||
SrcSizeFlag == FEXCore::X86Tables::InstFlags::SIZE_64BITDEF)) {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_64BIT);
|
||||
} else {
|
||||
DecodeInst->Flags |= DecodeFlags::GenSizeSrcSize(DecodeFlags::SIZE_32BIT);
|
||||
@@ -612,11 +634,20 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
Literal = static_cast<int32_t>(Literal);
|
||||
}
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Size = DestSize;
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.SignExtend = true;
|
||||
}
|
||||
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::Literal;
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal;
|
||||
++CurrentSrc;
|
||||
|
||||
if (Bytes == 8) [[unlikely]] {
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Size = 4;
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::Literal;
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal >> 32;
|
||||
}
|
||||
|
||||
Bytes = 0;
|
||||
DecodeInst->Src[CurrentSrc].Type = DecodedOperand::OpType::Literal;
|
||||
DecodeInst->Src[CurrentSrc].Data.Literal.Value = Literal;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Bytes == 0, "Inst at 0x{:x}: 0x{:04x} '{}' Had an instruction of size {} with {} remaining", DecodeInst->PC,
|
||||
@@ -626,7 +657,7 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
}
|
||||
|
||||
bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op) {
|
||||
DecodeInst->OP = Op;
|
||||
DecodeInst->OPRaw = DecodeInst->OP = Op;
|
||||
DecodeInst->TableInfo = Info;
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
|
||||
@@ -642,10 +673,13 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
// A normal instruction is the most likely.
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_INST) [[likely]] {
|
||||
return NormalOp(Info, Op);
|
||||
} else if (Info->Type == FEXCore::X86Tables::TYPE_ARCH_DISPATCHER) [[unlikely]] {
|
||||
// Dispatcher Op.
|
||||
return NormalOp(&Info->OpcodeDispatcher.Indirect[BlockInfo.Is64BitMode ? 1 : 0], Op);
|
||||
} else if (Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_11) {
|
||||
uint8_t ModRMByte = ReadByte();
|
||||
DecodeInst->ModRM = ModRMByte;
|
||||
DecodeInst->DecodedModRM = true;
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
|
||||
|
||||
FEXCore::X86Tables::ModRMDecoded ModRM;
|
||||
ModRM.Hex = DecodeInst->ModRM;
|
||||
@@ -662,24 +696,24 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
constexpr uint16_t PF_F2 = 3;
|
||||
|
||||
uint16_t PrefixType = PF_NONE;
|
||||
if (DecodeInst->LastEscapePrefix == 0xF3) {
|
||||
if (LastEscapePrefix == 0xF3) {
|
||||
PrefixType = PF_F3;
|
||||
} else if (DecodeInst->LastEscapePrefix == 0xF2) {
|
||||
} else if (LastEscapePrefix == 0xF2) {
|
||||
PrefixType = PF_F2;
|
||||
} else if (DecodeInst->LastEscapePrefix == 0x66) {
|
||||
} else if (LastEscapePrefix == 0x66) {
|
||||
PrefixType = PF_66;
|
||||
}
|
||||
|
||||
// We have ModRM
|
||||
uint8_t ModRMByte = ReadByte();
|
||||
DecodeInst->ModRM = ModRMByte;
|
||||
DecodeInst->DecodedModRM = true;
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
|
||||
|
||||
FEXCore::X86Tables::ModRMDecoded ModRM;
|
||||
ModRM.Hex = DecodeInst->ModRM;
|
||||
|
||||
uint16_t LocalOp = OPD(Info->Type, PrefixType, ModRM.reg);
|
||||
FEXCore::X86Tables::X86InstInfo* LocalInfo = &SecondInstGroupOps[LocalOp];
|
||||
const FEXCore::X86Tables::X86InstInfo* LocalInfo = &SecondInstGroupOps[LocalOp];
|
||||
#undef OPD
|
||||
if (LocalInfo->Type == FEXCore::X86Tables::TYPE_SECOND_GROUP_MODRM && ModRM.mod == 0b11) {
|
||||
// Everything in this group is privileged instructions aside from XGETBV
|
||||
@@ -700,11 +734,16 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
// We have ModRM
|
||||
uint8_t ModRMByte = ReadByte();
|
||||
DecodeInst->ModRM = ModRMByte;
|
||||
DecodeInst->DecodedModRM = true;
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
|
||||
|
||||
uint16_t X87Op = ((Op - 0xD8) << 8) | ModRMByte;
|
||||
return NormalOp(&X87Ops[X87Op], X87Op);
|
||||
return NormalOp(&(*X87Table)[X87Op], X87Op);
|
||||
} else if (Info->Type == FEXCore::X86Tables::TYPE_VEX_TABLE_PREFIX) {
|
||||
if (!VEXTable) {
|
||||
// AVX not enabled.
|
||||
return false;
|
||||
}
|
||||
|
||||
uint16_t map_select = 1;
|
||||
uint16_t pp = 0;
|
||||
const uint8_t Byte1 = ReadByte();
|
||||
@@ -742,7 +781,6 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_OPTION_AVX_W;
|
||||
}
|
||||
if (!(map_select >= 1 && map_select <= 3)) {
|
||||
LogMan::Msg::EFmt("We don't understand a map_select of: {}", map_select);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -752,13 +790,13 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
Op = OPD(map_select, pp, VEXOp);
|
||||
#undef OPD
|
||||
|
||||
FEXCore::X86Tables::X86InstInfo* LocalInfo = &VEXTableOps[Op];
|
||||
const FEXCore::X86Tables::X86InstInfo* LocalInfo = &(*VEXTable)[Op];
|
||||
|
||||
if (LocalInfo->Type >= FEXCore::X86Tables::TYPE_VEX_GROUP_12 && LocalInfo->Type <= FEXCore::X86Tables::TYPE_VEX_GROUP_17) {
|
||||
// We have ModRM
|
||||
uint8_t ModRMByte = ReadByte();
|
||||
DecodeInst->ModRM = ModRMByte;
|
||||
DecodeInst->DecodedModRM = true;
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
|
||||
|
||||
FEXCore::X86Tables::ModRMDecoded ModRM;
|
||||
ModRM.Hex = DecodeInst->ModRM;
|
||||
@@ -766,7 +804,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
#define OPD(group, pp, opcode) (((group - TYPE_VEX_GROUP_12) << 4) | (pp << 3) | (opcode))
|
||||
Op = OPD(LocalInfo->Type, pp, ModRM.reg);
|
||||
#undef OPD
|
||||
return NormalOp(&VEXTableGroupOps[Op], Op, options);
|
||||
return NormalOp(&(*VEXTableGroup)[Op], Op, options);
|
||||
} else {
|
||||
return NormalOp(LocalInfo, Op, options);
|
||||
}
|
||||
@@ -782,6 +820,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
|
||||
bool Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
InstructionSize = 0;
|
||||
LastEscapePrefix = 0;
|
||||
Instruction.fill(0);
|
||||
|
||||
DecodeInst = &DecodedBuffer[DecodedSize];
|
||||
@@ -803,7 +842,7 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
// Decode ModRM
|
||||
uint8_t ModRMByte = ReadByte();
|
||||
DecodeInst->ModRM = ModRMByte;
|
||||
DecodeInst->DecodedModRM = true;
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_DECODED_MODRM;
|
||||
|
||||
FEXCore::X86Tables::ModRMDecoded ModRM;
|
||||
ModRM.Hex = DecodeInst->ModRM;
|
||||
@@ -842,7 +881,7 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
uint16_t LocalOp = (Prefix << 8) | ReadByte();
|
||||
|
||||
bool NoOverlay66 = (FEXCore::X86Tables::H0F38TableOps[LocalOp].Flags & InstFlags::FLAGS_NO_OVERLAY66) != 0;
|
||||
if (DecodeInst->LastEscapePrefix == 0x66 && NoOverlay66) { // Operand Size
|
||||
if (LastEscapePrefix == 0x66 && NoOverlay66) { // Operand Size
|
||||
// Remove prefix so it doesn't effect calculations.
|
||||
// This is only an escape prefix rather than modifier now
|
||||
DecodeInst->Flags &= ~DecodeFlags::FLAG_OPERAND_SIZE;
|
||||
@@ -858,7 +897,7 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
constexpr uint16_t PF_3A_REX = (1 << 1);
|
||||
|
||||
uint16_t Prefix = PF_3A_NONE;
|
||||
if (DecodeInst->LastEscapePrefix == 0x66) { // Operand Size
|
||||
if (LastEscapePrefix == 0x66) { // Operand Size
|
||||
Prefix = PF_3A_66;
|
||||
}
|
||||
|
||||
@@ -884,17 +923,17 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
|
||||
if (NoOverlay) { // This section of the table ignores prefix extention
|
||||
return NormalOpHeader(&FEXCore::X86Tables::SecondBaseOps[EscapeOp], EscapeOp);
|
||||
} else if (DecodeInst->LastEscapePrefix == 0xF3) { // REP
|
||||
} else if (LastEscapePrefix == 0xF3) { // REP
|
||||
// Remove prefix so it doesn't effect calculations.
|
||||
// This is only an escape prefix rather tan modifier now
|
||||
DecodeInst->Flags &= ~DecodeFlags::FLAG_REP_PREFIX;
|
||||
return NormalOpHeader(&FEXCore::X86Tables::RepModOps[EscapeOp], EscapeOp);
|
||||
} else if (DecodeInst->LastEscapePrefix == 0xF2) { // REPNE
|
||||
} else if (LastEscapePrefix == 0xF2) { // REPNE
|
||||
// Remove prefix so it doesn't effect calculations.
|
||||
// This is only an escape prefix rather tan modifier now
|
||||
DecodeInst->Flags &= ~DecodeFlags::FLAG_REPNE_PREFIX;
|
||||
return NormalOpHeader(&FEXCore::X86Tables::RepNEModOps[EscapeOp], EscapeOp);
|
||||
} else if (DecodeInst->LastEscapePrefix == 0x66 && !NoOverlay66) { // Operand Size
|
||||
} else if (LastEscapePrefix == 0x66 && !NoOverlay66) { // Operand Size
|
||||
// Remove prefix so it doesn't effect calculations.
|
||||
// This is only an escape prefix rather tan modifier now
|
||||
DecodeInst->Flags &= ~DecodeFlags::FLAG_OPERAND_SIZE;
|
||||
@@ -910,7 +949,7 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
}
|
||||
case 0x66: // Operand Size prefix
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_OPERAND_SIZE;
|
||||
DecodeInst->LastEscapePrefix = Op;
|
||||
LastEscapePrefix = Op;
|
||||
DecodeFlags::PushOpAddr(&DecodeInst->Flags, DecodeFlags::FLAG_OPERAND_SIZE_LAST);
|
||||
break;
|
||||
case 0x67: // Address Size override prefix
|
||||
@@ -941,11 +980,11 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
break;
|
||||
case 0xF2: // REPNE prefix
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REPNE_PREFIX;
|
||||
DecodeInst->LastEscapePrefix = Op;
|
||||
LastEscapePrefix = Op;
|
||||
break;
|
||||
case 0xF3: // REP prefix
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REP_PREFIX;
|
||||
DecodeInst->LastEscapePrefix = Op;
|
||||
LastEscapePrefix = Op;
|
||||
break;
|
||||
case 0x64: // FS prefix
|
||||
DecodeInst->Flags = (DecodeInst->Flags & ~FEXCore::X86Tables::DecodeFlags::FLAG_SEGMENTS) | DecodeFlags::FLAG_FS_PREFIX;
|
||||
@@ -955,7 +994,10 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
break;
|
||||
default:
|
||||
[[likely]] { // Default base table
|
||||
auto Info = &FEXCore::X86Tables::BaseOps[Op];
|
||||
const X86InstInfo* Info = &FEXCore::X86Tables::BaseOps[Op];
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_ARCH_DISPATCHER) {
|
||||
Info = &Info->OpcodeDispatcher.Indirect[BlockInfo.Is64BitMode ? 1 : 0];
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_REX_PREFIX) {
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_PREFIX;
|
||||
@@ -1007,11 +1049,27 @@ Decoder::DecodedBlockStatus Decoder::DecodeInstruction(uint64_t PC) {
|
||||
DecodeInst->TableInfo = nullptr;
|
||||
DecodeInst->InstSize = 0;
|
||||
return ErrorDuringDecoding ? DecodedBlockStatus::INVALID_INST : DecodedBlockStatus::NOEXEC_INST;
|
||||
} else if (!DecodeInst->TableInfo || !DecodeInst->TableInfo->OpcodeDispatcher) {
|
||||
} else if (!DecodeInst->TableInfo || (DecodeInst->TableInfo->Type == TYPE_INST && !DecodeInst->TableInfo->OpcodeDispatcher.OpDispatch)) {
|
||||
// If there wasn't an error during decoding but we have no dispatcher for the instruction then claim invalid instruction.
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
}
|
||||
|
||||
if (CTX->AreMonoHacksActive()) {
|
||||
// Unity uses a standard SPSC ringbuffer with cached read/write pointers and thread waiting flags at the following
|
||||
// offsets, which are consistent between 32-bit and 64-bit Unity versions from 2015 onwards.
|
||||
auto IsKnownAtomicDisplacement = [](uint64_t Displacement) {
|
||||
return Displacement == 0x80 || Displacement == 0x84 || Displacement == 0xC0 || Displacement == 0xC4;
|
||||
};
|
||||
|
||||
if (DecodeInst->OP == 0x8b && DecodeInst->Src[0].IsGPRIndirect() &&
|
||||
IsKnownAtomicDisplacement(DecodeInst->Src[0].Data.GPRIndirect.Displacement)) {
|
||||
DecodeInst->Flags |= X86Tables::DecodeFlags::FLAG_FORCE_TSO;
|
||||
}
|
||||
if (DecodeInst->OP == 0x89 && DecodeInst->Dest.IsGPRIndirect() && IsKnownAtomicDisplacement(DecodeInst->Dest.Data.GPRIndirect.Displacement)) {
|
||||
DecodeInst->Flags |= X86Tables::DecodeFlags::FLAG_FORCE_TSO;
|
||||
}
|
||||
}
|
||||
|
||||
return DecodedBlockStatus::SUCCESS;
|
||||
}
|
||||
|
||||
@@ -1027,6 +1085,13 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
const auto InstEnd = DecodeInst->PC + DecodeInst->InstSize;
|
||||
|
||||
if (DecodeInst->TableInfo->Flags & FEXCore::X86Tables::InstFlags::FLAGS_CALL) {
|
||||
if (ExecutableRangeWritable && CTX->AreMonoHacksActive()) {
|
||||
// Mono generated code often contains noreturn calls with garbage following them, and calls are always backpatched
|
||||
// after CIL compilation leading to n recompiles for a multiblock with n calls. Choose to minimize stutters over
|
||||
// raw performance and disable tracking past calls for mono generated code.
|
||||
return;
|
||||
}
|
||||
|
||||
AddBranchTarget(InstEnd);
|
||||
BlockInfo.EntryPoints.emplace(InstEnd);
|
||||
return;
|
||||
@@ -1058,10 +1123,16 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
TargetRIP &= 0xFFFFFFFFU;
|
||||
}
|
||||
|
||||
if (Conditional) {
|
||||
// If we are conditional then a target can be the instruction past the conditional instruction
|
||||
AddBranchTarget(InstEnd);
|
||||
}
|
||||
|
||||
// If the target RIP is x86 code within the symbol ranges then we are golden
|
||||
// Forbid cross-page branches to both avoid massive (range-wise) code blocks in highly fragmented code and trying to decode unmapped branch targets
|
||||
bool ValidMultiblockMember =
|
||||
TargetRIP >= SymbolMinAddress && TargetRIP < std::min(FEXCore::AlignUp(InstEnd, FEXCore::Utils::FEX_PAGE_SIZE), SymbolMaxAddress);
|
||||
// Forbid distant branches to have the cost code better match the guest code layout, avoiding massive (range-wise) code
|
||||
// blocks in highly fragmented guest code. Such branches are often not-taken branches to garbage in obfuscated code.
|
||||
constexpr uint64_t MAX_FORWARD_BRANCH_DIST = FEXCore::Utils::FEX_PAGE_SIZE * 4;
|
||||
bool ValidMultiblockMember = TargetRIP >= SymbolMinAddress && TargetRIP < std::min(InstEnd + MAX_FORWARD_BRANCH_DIST, SymbolMaxAddress);
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ValidMultiblockMember = ValidMultiblockMember && !RtlIsEcCode(TargetRIP);
|
||||
@@ -1072,9 +1143,6 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
if (Conditional) {
|
||||
MaxCondBranchForward = std::max(MaxCondBranchForward, TargetRIP);
|
||||
MaxCondBranchBackwards = std::min(MaxCondBranchBackwards, TargetRIP);
|
||||
|
||||
// If we are conditional then a target can be the instruction past the conditional instruction
|
||||
AddBranchTarget(InstEnd);
|
||||
}
|
||||
|
||||
AddBranchTarget(TargetRIP);
|
||||
@@ -1085,6 +1153,60 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
}
|
||||
}
|
||||
|
||||
bool Decoder::IsBranchMonoTailcall(uint64_t NumInstructions) const {
|
||||
// While the mono call backpatching block can easily be detected due it being the only one to contain SMC-faulting
|
||||
// atomics, that can't be said for the tailcall jump backpatcher which has changed several times across versions and
|
||||
// can be partially inlined. To work around this, instead detect the tailcall site itself and force full non-signal-based
|
||||
// SMC detection for that single block.
|
||||
if (!ExecutableRangeWritable) {
|
||||
// We only care about jitted code
|
||||
return false;
|
||||
}
|
||||
|
||||
// See mini-{amd64,x86}.c in the mono codebase, specifically where METHOD_JUMP patches are emitted.
|
||||
if (GetGPROpSize() == IR::OpSize::i32Bit) {
|
||||
// Matches:
|
||||
// LEAVE
|
||||
// <none> / NOP / MOV EAX, EAX / LEA EBP, [EBP+0]
|
||||
// JMP imm32
|
||||
if (DecodeInst->OP != 0xE9 || NumInstructions < 2) {
|
||||
return false;
|
||||
}
|
||||
|
||||
auto PrevInst = std::prev(DecodeInst);
|
||||
if (PrevInst->OP == 0xC9) {
|
||||
return true;
|
||||
}
|
||||
|
||||
if (NumInstructions < 3 || std::prev(PrevInst)->OP != 0xC9) {
|
||||
return false;
|
||||
}
|
||||
|
||||
return PrevInst->OP == 0x90 || (PrevInst->OP == 0x8B && PrevInst->ModRM == 0xC0) ||
|
||||
(PrevInst->OP == 0x8D && PrevInst->ModRM == 0x6D && PrevInst->Src[1].IsLiteral() && PrevInst->Src[1].Literal() == 0);
|
||||
} else {
|
||||
FEXCore::X86Tables::ModRMDecoded ModRM;
|
||||
ModRM.Hex = DecodeInst->ModRM;
|
||||
if (DecodeInst->OPRaw == 0xFF && ModRM.reg == 4 && DecodeInst->Src[0].IsGPR()) {
|
||||
if (DecodeInst->Src[0].Data.GPR.GPR == FEXCore::X86State::REG_RAX) {
|
||||
// Found in versions of mono from 2024 onwards - matches:
|
||||
// REX.W JMP rax
|
||||
return (DecodeInst->Flags & (DecodeFlags::FLAG_REX_PREFIX | DecodeFlags::FLAG_REX_WIDENING | DecodeFlags::FLAG_REX_XGPR_B |
|
||||
DecodeFlags::FLAG_REX_XGPR_X | DecodeFlags::FLAG_REX_XGPR_R)) ==
|
||||
(DecodeFlags::FLAG_REX_PREFIX | DecodeFlags::FLAG_REX_WIDENING);
|
||||
} else if (NumInstructions > 1 && DecodeInst->Src[0].Data.GPR.GPR == FEXCore::X86State::REG_R11) {
|
||||
// Found in older versions of mono - match:
|
||||
// MOV r11, imm64
|
||||
// JMP r11
|
||||
auto PrevInst = std::prev(DecodeInst);
|
||||
return PrevInst->OP == 0xBB && PrevInst->Dest.IsGPR() && PrevInst->Dest.Data.GPR.GPR == FEXCore::X86State::REG_R11;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
bool Decoder::InstCanContinue() const {
|
||||
if (DecodeInst->PC + DecodeInst->InstSize == NextBlockStartAddress) {
|
||||
return false;
|
||||
@@ -1187,7 +1309,7 @@ const uint8_t* Decoder::AdjustAddrForSpecialRegion(const uint8_t* _InstStream, u
|
||||
return _InstStream - EntryPoint + RIP;
|
||||
}
|
||||
|
||||
void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState *Thread, const uint8_t* _InstStream, uint64_t PC, uint64_t MaxInst) {
|
||||
void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, const uint8_t* _InstStream, uint64_t PC, uint64_t MaxInst) {
|
||||
FEXCORE_PROFILE_SCOPED("DecodeInstructions");
|
||||
BlockInfo.TotalInstructionCount = 0;
|
||||
BlockInfo.Blocks.clear();
|
||||
@@ -1199,8 +1321,8 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState *Thre
|
||||
DecodedBuffer = PoolObject.ReownOrClaimBuffer();
|
||||
|
||||
// Decode operating mode from thread's CS segment.
|
||||
const auto CSSegment = Thread->CurrentFrame->State.gdt[Thread->CurrentFrame->State.cs_idx >> 3];
|
||||
BlockInfo.Is64BitMode = CSSegment.L == 1;
|
||||
const auto CSSegment = Core::CPUState::GetSegmentFromIndex(Thread->CurrentFrame->State, Thread->CurrentFrame->State.cs_idx);
|
||||
BlockInfo.Is64BitMode = CSSegment->L == 1;
|
||||
LOGMAN_THROW_A_FMT(BlockInfo.Is64BitMode == CTX->Config.Is64BitMode, "Expected operating mode to not change at runtime!");
|
||||
|
||||
// XXX: Load symbol data
|
||||
@@ -1345,6 +1467,7 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState *Thre
|
||||
// If the branch target is within our multiblock range then we can keep going on
|
||||
// We don't want to short circuit this since we want to calculate our ranges still
|
||||
// NOTE: This will invalidate BlockIt, this is fine as we immediately break from the loop and EraseBlock cannot be true
|
||||
BlockIt->ForceFullSMCDetection = CTX->AreMonoHacksActive() && IsBranchMonoTailcall(BlockIt->NumInstructions);
|
||||
BranchTargetInMultiblockRange();
|
||||
}
|
||||
|
||||
|
||||
@@ -4,18 +4,21 @@
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
#include <FEXCore/Utils/ThreadPoolAllocator.h>
|
||||
#include <FEXCore/fextl/set.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <stddef.h>
|
||||
#include <optional>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
namespace FEXCore::HLE {
|
||||
enum class SyscallOSABI;
|
||||
}
|
||||
|
||||
namespace FEXCore::Frontend {
|
||||
class Decoder final {
|
||||
@@ -34,6 +37,7 @@ public:
|
||||
FEXCore::X86Tables::DecodedInst* DecodedInstructions;
|
||||
DecodedBlockStatus BlockStatus;
|
||||
bool IsEntryPoint {};
|
||||
bool ForceFullSMCDetection {};
|
||||
};
|
||||
|
||||
struct DecodedBlockInformation final {
|
||||
@@ -45,7 +49,7 @@ public:
|
||||
};
|
||||
|
||||
Decoder(FEXCore::Core::InternalThreadState* Thread);
|
||||
void DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState *Thread, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst);
|
||||
void DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst);
|
||||
|
||||
const DecodedBlockInformation* GetDecodedBlockInfo() const {
|
||||
return &BlockInfo;
|
||||
@@ -86,6 +90,7 @@ private:
|
||||
DecodedBlockStatus DecodeInstruction(uint64_t PC);
|
||||
|
||||
void BranchTargetInMultiblockRange();
|
||||
bool IsBranchMonoTailcall(uint64_t NumInstructions) const;
|
||||
bool InstCanContinue() const;
|
||||
|
||||
void AddBranchTarget(uint64_t Target);
|
||||
@@ -109,6 +114,7 @@ private:
|
||||
|
||||
uint64_t ExecutableRangeBase {};
|
||||
uint64_t ExecutableRangeEnd {};
|
||||
bool ExecutableRangeWritable {};
|
||||
bool HitNonExecutableRange {};
|
||||
|
||||
const uint8_t* InstStream {};
|
||||
@@ -119,6 +125,7 @@ private:
|
||||
static constexpr size_t MAX_INST_SIZE = 15;
|
||||
uint8_t InstructionSize {};
|
||||
std::array<uint8_t, MAX_INST_SIZE> Instruction;
|
||||
uint8_t LastEscapePrefix {};
|
||||
FEXCore::X86Tables::DecodedInst* DecodeInst;
|
||||
|
||||
// This is for multiblock data tracking
|
||||
@@ -147,6 +154,11 @@ private:
|
||||
&FEXCore::Frontend::Decoder::DecodeModRM_16,
|
||||
};
|
||||
|
||||
const std::array<X86Tables::X86InstInfo, X86Tables::MAX_X87_TABLE_SIZE>* X87Table;
|
||||
|
||||
const std::array<X86Tables::X86InstInfo, X86Tables::MAX_VEX_TABLE_SIZE>* VEXTable {};
|
||||
const std::array<X86Tables::X86InstInfo, X86Tables::MAX_VEX_GROUP_TABLE_SIZE>* VEXTableGroup {};
|
||||
|
||||
const uint8_t* AdjustAddrForSpecialRegion(const uint8_t* _InstStream, uint64_t EntryPoint, uint64_t RIP);
|
||||
};
|
||||
} // namespace FEXCore::Frontend
|
||||
@@ -340,7 +340,7 @@ struct OpHandlers<IR::OP_F80SCALE> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SIN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return sin(src);
|
||||
}
|
||||
@@ -348,7 +348,7 @@ struct OpHandlers<IR::OP_F64SIN> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64COS> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return cos(src);
|
||||
}
|
||||
@@ -356,7 +356,7 @@ struct OpHandlers<IR::OP_F64COS> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SINCOS> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorScalarF64Pair handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorScalarF64Pair handle(double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
double sin, cos;
|
||||
#ifdef _WIN32
|
||||
@@ -371,7 +371,7 @@ struct OpHandlers<IR::OP_F64SINCOS> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64TAN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return tan(src);
|
||||
}
|
||||
@@ -379,7 +379,7 @@ struct OpHandlers<IR::OP_F64TAN> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64F2XM1> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return exp2(src) - 1.0;
|
||||
}
|
||||
@@ -387,7 +387,7 @@ struct OpHandlers<IR::OP_F64F2XM1> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64ATAN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return atan2(src1, src2);
|
||||
}
|
||||
@@ -395,7 +395,7 @@ struct OpHandlers<IR::OP_F64ATAN> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FPREM> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return fmod(src1, src2);
|
||||
}
|
||||
@@ -403,7 +403,7 @@ struct OpHandlers<IR::OP_F64FPREM> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FPREM1> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return remainder(src1, src2);
|
||||
}
|
||||
@@ -411,7 +411,7 @@ struct OpHandlers<IR::OP_F64FPREM1> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FYL2X> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return src2 * log2(src1);
|
||||
}
|
||||
@@ -419,7 +419,7 @@ struct OpHandlers<IR::OP_F64FYL2X> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SCALE> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(uint16_t FCW, double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
if (src1 == 0.0) { // src1 might be +/- zero
|
||||
return src1; // this will return negative or positive zero if when appropriate
|
||||
@@ -445,7 +445,6 @@ struct OpHandlers<IR::OP_F80BCDSTORE> {
|
||||
uint64_t Tmp = Src1.ToI64(&State.State);
|
||||
X80SoftFloat Rv;
|
||||
uint8_t* BCD = reinterpret_cast<uint8_t*>(&Rv);
|
||||
memset(BCD, 0, 10);
|
||||
|
||||
for (size_t i = 0; i < 9; ++i) {
|
||||
if (Tmp == 0) {
|
||||
|
||||
@@ -82,24 +82,22 @@ void InterpreterOps::FillFallbackIndexPointers(Core::FallbackABIInfo* Info, uint
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80SCALE>::handle)};
|
||||
|
||||
// Double Precision Unary
|
||||
Info[Core::OPINDEX_F64SIN] = {ABIHandlers[FABI_F64_I16_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64SIN>::handle)};
|
||||
Info[Core::OPINDEX_F64COS] = {ABIHandlers[FABI_F64_I16_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64COS>::handle)};
|
||||
Info[Core::OPINDEX_F64SINCOS] = {ABIHandlers[FABI_F64x2_I16_F64_PTR],
|
||||
Info[Core::OPINDEX_F64SIN] = {ABIHandlers[FABI_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64SIN>::handle)};
|
||||
Info[Core::OPINDEX_F64COS] = {ABIHandlers[FABI_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64COS>::handle)};
|
||||
Info[Core::OPINDEX_F64SINCOS] = {ABIHandlers[FABI_F64x2_F64_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64SINCOS>::handle)};
|
||||
Info[Core::OPINDEX_F64TAN] = {ABIHandlers[FABI_F64_I16_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64TAN>::handle)};
|
||||
Info[Core::OPINDEX_F64F2XM1] = {ABIHandlers[FABI_F64_I16_F64_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64F2XM1>::handle)};
|
||||
Info[Core::OPINDEX_F64TAN] = {ABIHandlers[FABI_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64TAN>::handle)};
|
||||
Info[Core::OPINDEX_F64F2XM1] = {ABIHandlers[FABI_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64F2XM1>::handle)};
|
||||
|
||||
// Double Precision Binary
|
||||
Info[Core::OPINDEX_F64ATAN] = {ABIHandlers[FABI_F64_I16_F64_F64_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64ATAN>::handle)};
|
||||
Info[Core::OPINDEX_F64FPREM] = {ABIHandlers[FABI_F64_I16_F64_F64_PTR],
|
||||
Info[Core::OPINDEX_F64ATAN] = {ABIHandlers[FABI_F64_F64_F64_PTR], reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64ATAN>::handle)};
|
||||
Info[Core::OPINDEX_F64FPREM] = {ABIHandlers[FABI_F64_F64_F64_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM>::handle)};
|
||||
Info[Core::OPINDEX_F64FPREM1] = {ABIHandlers[FABI_F64_I16_F64_F64_PTR],
|
||||
Info[Core::OPINDEX_F64FPREM1] = {ABIHandlers[FABI_F64_F64_F64_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM1>::handle)};
|
||||
Info[Core::OPINDEX_F64FYL2X] = {ABIHandlers[FABI_F64_I16_F64_F64_PTR],
|
||||
Info[Core::OPINDEX_F64FYL2X] = {ABIHandlers[FABI_F64_F64_F64_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64FYL2X>::handle)};
|
||||
Info[Core::OPINDEX_F64SCALE] = {ABIHandlers[FABI_F64_I16_F64_F64_PTR],
|
||||
Info[Core::OPINDEX_F64SCALE] = {ABIHandlers[FABI_F64_F64_F64_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64SCALE>::handle)};
|
||||
|
||||
// SSE4.2 string instructions
|
||||
@@ -220,21 +218,21 @@ bool InterpreterOps::GetFallbackHandler(const IR::IROp_Header* IROp, FallbackInf
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_UNARY_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = {FABI_F64_I16_F64_PTR, Core::OPINDEX_F64##OP}; \
|
||||
return true; \
|
||||
#define COMMON_UNARY_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = {FABI_F64_F64_PTR, Core::OPINDEX_F64##OP}; \
|
||||
return true; \
|
||||
}
|
||||
#define COMMON_UNARYPAIR_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = {FABI_F64x2_I16_F64_PTR, Core::OPINDEX_F64##OP}; \
|
||||
return true; \
|
||||
#define COMMON_UNARYPAIR_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = {FABI_F64x2_F64_PTR, Core::OPINDEX_F64##OP}; \
|
||||
return true; \
|
||||
}
|
||||
|
||||
#define COMMON_BINARY_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = {FABI_F64_I16_F64_F64_PTR, Core::OPINDEX_F64##OP}; \
|
||||
return true; \
|
||||
#define COMMON_BINARY_F64_OP(OP) \
|
||||
case IR::OP_F64##OP: { \
|
||||
*Info = {FABI_F64_F64_F64_PTR, Core::OPINDEX_F64##OP}; \
|
||||
return true; \
|
||||
}
|
||||
|
||||
// Unary
|
||||
|
||||
@@ -19,8 +19,8 @@ enum FallbackABI {
|
||||
FABI_F80_I16_I32_PTR,
|
||||
FABI_F32_I16_F80_PTR,
|
||||
FABI_F64_I16_F80_PTR,
|
||||
FABI_F64_I16_F64_PTR,
|
||||
FABI_F64_I16_F64_F64_PTR,
|
||||
FABI_F64_F64_PTR,
|
||||
FABI_F64_F64_F64_PTR,
|
||||
FABI_I16_I16_F80_PTR,
|
||||
FABI_I32_I16_F80_PTR,
|
||||
FABI_I64_I16_F80_PTR,
|
||||
@@ -28,7 +28,7 @@ enum FallbackABI {
|
||||
FABI_F80_I16_F80_PTR,
|
||||
FABI_F80_I16_F80_F80_PTR,
|
||||
FABI_F80x2_I16_F80_PTR,
|
||||
FABI_F64x2_I16_F64_PTR,
|
||||
FABI_F64x2_F64_PTR,
|
||||
FABI_I32_I64_I64_V128_V128_I16,
|
||||
FABI_I32_V128_V128_I16,
|
||||
FABI_UNKNOWN,
|
||||
|
||||
@@ -515,6 +515,12 @@ DEF_OP(AndWithFlags) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AndShift) {
|
||||
auto Op = IROp->C<IR::IROp_XorShift>();
|
||||
|
||||
and_(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1), GetReg(Op->Src2), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
|
||||
}
|
||||
|
||||
DEF_OP(XorShift) {
|
||||
auto Op = IROp->C<IR::IROp_XorShift>();
|
||||
|
||||
@@ -582,7 +588,7 @@ DEF_OP(ShiftFlags) {
|
||||
and_(ARMEmitter::Size::i32Bit, TMP1, Src2, OpSize == IR::OpSize::i64Bit ? 0x3f : 0x1f);
|
||||
|
||||
ARMEmitter::ForwardLabel Done;
|
||||
cbz(EmitSize, TMP1, &Done);
|
||||
(void)cbz(EmitSize, TMP1, &Done);
|
||||
{
|
||||
// PF/SF/ZF/OF
|
||||
if (OpSize >= IR::OpSize::i32Bit) {
|
||||
@@ -646,7 +652,7 @@ DEF_OP(ShiftFlags) {
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP2);
|
||||
}
|
||||
}
|
||||
Bind(&Done);
|
||||
(void)Bind(&Done);
|
||||
|
||||
// TODO: Make RA less dumb so this can't happen (e.g. with late-kill).
|
||||
if (PFOutput != PFTemp) {
|
||||
@@ -663,7 +669,7 @@ DEF_OP(RotateFlags) {
|
||||
|
||||
// If shift=0, flags are unaffected. Wrap the whole implementation in a cbz.
|
||||
ARMEmitter::ForwardLabel Done;
|
||||
cbz(EmitSize, Shift, &Done);
|
||||
(void)cbz(EmitSize, Shift, &Done);
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
const auto BitSize = IR::OpSizeToSize(Op->Size) * 8;
|
||||
@@ -695,7 +701,7 @@ DEF_OP(RotateFlags) {
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP3);
|
||||
}
|
||||
}
|
||||
Bind(&Done);
|
||||
(void)Bind(&Done);
|
||||
}
|
||||
|
||||
DEF_OP(Extr) {
|
||||
@@ -761,14 +767,14 @@ DEF_OP(PDep) {
|
||||
// Now, they're copied, so we can start setting Dest (even if it overlaps with
|
||||
// one of them). Handle early exit case
|
||||
mov(EmitSize, Dest, 0);
|
||||
cbz(EmitSize, OrigMask, &Done);
|
||||
(void)cbz(EmitSize, OrigMask, &Done);
|
||||
|
||||
// Setup for first iteration
|
||||
neg(EmitSize, T0, Mask);
|
||||
and_(EmitSize, T0, T0, Mask);
|
||||
|
||||
// Main loop
|
||||
Bind(&NextBit);
|
||||
(void)Bind(&NextBit);
|
||||
sbfx(EmitSize, T1, Input, 0, 1);
|
||||
eor(EmitSize, Mask, Mask, T0);
|
||||
and_(EmitSize, T0, T1, T0);
|
||||
@@ -776,10 +782,10 @@ DEF_OP(PDep) {
|
||||
orr(EmitSize, Dest, Dest, T0);
|
||||
lsr(EmitSize, Input, Input, 1);
|
||||
and_(EmitSize, T0, Mask, T1);
|
||||
cbnz(EmitSize, T0, &NextBit);
|
||||
(void)cbnz(EmitSize, T0, &NextBit);
|
||||
|
||||
// All done with nothing to do.
|
||||
Bind(&Done);
|
||||
(void)Bind(&Done);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -815,27 +821,27 @@ DEF_OP(PExt) {
|
||||
ARMEmitter::BackwardLabel NextBit;
|
||||
ARMEmitter::ForwardLabel Done;
|
||||
|
||||
cbz(EmitSize, Mask, &EarlyExit);
|
||||
(void)cbz(EmitSize, Mask, &EarlyExit);
|
||||
mov(EmitSize, MaskReg, Mask);
|
||||
mov(EmitSize, ValueReg, Input);
|
||||
mov(EmitSize, Dest, ARMEmitter::Reg::zr);
|
||||
|
||||
// Main loop
|
||||
Bind(&NextBit);
|
||||
cbz(EmitSize, MaskReg, &Done);
|
||||
(void)Bind(&NextBit);
|
||||
(void)cbz(EmitSize, MaskReg, &Done);
|
||||
clz(EmitSize, BitReg, MaskReg);
|
||||
lslv(EmitSize, ValueReg, ValueReg, BitReg);
|
||||
lslv(EmitSize, MaskReg, MaskReg, BitReg);
|
||||
extr(EmitSize, Dest, Dest, ValueReg, OpSizeBitsM1);
|
||||
bfc(EmitSize, MaskReg, OpSizeBitsM1, 1);
|
||||
b(&NextBit);
|
||||
(void)b(&NextBit);
|
||||
|
||||
// Early exit
|
||||
Bind(&EarlyExit);
|
||||
(void)Bind(&EarlyExit);
|
||||
mov(EmitSize, Dest, ARMEmitter::Reg::zr);
|
||||
|
||||
// All done with nothing to do.
|
||||
Bind(&Done);
|
||||
(void)Bind(&Done);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -903,7 +909,7 @@ DEF_OP(Div) {
|
||||
eor(EmitSize, TMP1, TMP1, Upper);
|
||||
|
||||
// If the sign bit matches then the result is zero
|
||||
cbz(EmitSize, TMP1, &Only64Bit);
|
||||
(void)cbz(EmitSize, TMP1, &Only64Bit);
|
||||
|
||||
// Long divide
|
||||
{
|
||||
@@ -922,17 +928,17 @@ DEF_OP(Div) {
|
||||
mov(EmitSize, Remainder, TMP2);
|
||||
|
||||
// Skip 64-bit path
|
||||
b(&LongDIVRet);
|
||||
(void)b(&LongDIVRet);
|
||||
}
|
||||
|
||||
Bind(&Only64Bit);
|
||||
(void)Bind(&Only64Bit);
|
||||
// 64-Bit only
|
||||
{
|
||||
sdiv(EmitSize, Quotient, Lower, Divisor);
|
||||
msub(EmitSize, Remainder, Quotient, Divisor, Lower);
|
||||
}
|
||||
|
||||
Bind(&LongDIVRet);
|
||||
(void)Bind(&LongDIVRet);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown DIV Size: {}", OpSize); break;
|
||||
@@ -986,7 +992,7 @@ DEF_OP(UDiv) {
|
||||
|
||||
// Check the upper bits for zero
|
||||
// If the upper bits are zero then we can do a 64-bit divide
|
||||
cbz(EmitSize, Upper, &Only64Bit);
|
||||
(void)cbz(EmitSize, Upper, &Only64Bit);
|
||||
|
||||
// Long divide
|
||||
{
|
||||
@@ -1005,17 +1011,17 @@ DEF_OP(UDiv) {
|
||||
mov(EmitSize, Remainder, TMP2);
|
||||
|
||||
// Skip 64-bit path
|
||||
b(&LongDIVRet);
|
||||
(void)b(&LongDIVRet);
|
||||
}
|
||||
|
||||
Bind(&Only64Bit);
|
||||
(void)Bind(&Only64Bit);
|
||||
// 64-Bit only
|
||||
{
|
||||
udiv(EmitSize, Quotient, Lower, Divisor);
|
||||
msub(EmitSize, Remainder, Quotient, Divisor, Lower);
|
||||
}
|
||||
|
||||
Bind(&LongDIVRet);
|
||||
(void)Bind(&LongDIVRet);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown LUDIV Size: {}", OpSize); break;
|
||||
@@ -1040,24 +1046,19 @@ DEF_OP(Popcount) {
|
||||
|
||||
if (CTX->HostFeatures.SupportsCSSC) {
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit:
|
||||
uxtb(ARMEmitter::Size::i32Bit, Dst, Src);
|
||||
cnt(ARMEmitter::Size::i32Bit, Dst, Dst);
|
||||
break;
|
||||
case IR::OpSize::i16Bit:
|
||||
uxth(ARMEmitter::Size::i32Bit, Dst, Src);
|
||||
cnt(ARMEmitter::Size::i32Bit, Dst, Dst);
|
||||
break;
|
||||
case IR::OpSize::i32Bit:
|
||||
cnt(ARMEmitter::Size::i32Bit, Dst, Src);
|
||||
break;
|
||||
case IR::OpSize::i64Bit:
|
||||
cnt(ARMEmitter::Size::i64Bit, Dst, Src);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unsupported Popcount size: {}", OpSize);
|
||||
case IR::OpSize::i8Bit:
|
||||
uxtb(ARMEmitter::Size::i32Bit, Dst, Src);
|
||||
cnt(ARMEmitter::Size::i32Bit, Dst, Dst);
|
||||
break;
|
||||
case IR::OpSize::i16Bit:
|
||||
uxth(ARMEmitter::Size::i32Bit, Dst, Src);
|
||||
cnt(ARMEmitter::Size::i32Bit, Dst, Dst);
|
||||
break;
|
||||
case IR::OpSize::i32Bit: cnt(ARMEmitter::Size::i32Bit, Dst, Src); break;
|
||||
case IR::OpSize::i64Bit: cnt(ARMEmitter::Size::i64Bit, Dst, Src); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unsupported Popcount size: {}", OpSize);
|
||||
}
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit:
|
||||
fmov(ARMEmitter::Size::i32Bit, VTMP1.S(), Src);
|
||||
@@ -1368,12 +1369,12 @@ DEF_OP(VExtractToGPR) {
|
||||
const auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
[[maybe_unused]] constexpr auto AVXRegBitSize = Core::CPUState::XMM_AVX_REG_SIZE * 8;
|
||||
constexpr auto AVXRegBitSize = Core::CPUState::XMM_AVX_REG_SIZE * 8;
|
||||
constexpr auto SSERegBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
|
||||
const auto ElementSizeBits = IR::OpSizeAsBits(Op->Header.ElementSize);
|
||||
|
||||
const auto Offset = ElementSizeBits * Op->Index;
|
||||
[[maybe_unused]] const auto Is256Bit = Offset >= SSERegBitSize;
|
||||
const auto Is256Bit = Offset >= SSERegBitSize;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
@@ -33,7 +33,7 @@ void Arm64JITCore::InsertNamedThunkRelocation(ARMEmitter::Register Reg, const IR
|
||||
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Sum));
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Pointer, EmitterCTX->Config.CacheObjectCodeCompilation());
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Pointer, false);
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
@@ -63,7 +63,7 @@ void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair& Lit) {
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t*>();
|
||||
Lit.MoveABI.NamedSymbolLiteral.Offset = CurrentCursor - CodeData.BlockBegin;
|
||||
|
||||
Bind(&Lit.Loc);
|
||||
BindOrRestart(&Lit.Loc);
|
||||
dc64(Lit.Lit);
|
||||
Relocations.emplace_back(Lit.MoveABI);
|
||||
}
|
||||
@@ -77,39 +77,36 @@ void Arm64JITCore::InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constan
|
||||
MoveABI.GuestRIPMove.GuestRIP = Constant;
|
||||
MoveABI.GuestRIPMove.RegisterIndex = Reg.Idx();
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Constant, EmitterCTX->Config.CacheObjectCodeCompilation());
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Constant, false);
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations,
|
||||
const char* EntryRelocations) {
|
||||
size_t DataIndex {};
|
||||
for (size_t j = 0; j < NumRelocations; ++j) {
|
||||
const FEXCore::CPU::Relocation* Reloc = reinterpret_cast<const FEXCore::CPU::Relocation*>(&EntryRelocations[DataIndex]);
|
||||
LOGMAN_THROW_A_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
|
||||
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, std::span<std::byte> Code, std::span<const FEXCore::CPU::Relocation> Relocations) {
|
||||
const auto OrigBase = GetBufferBase();
|
||||
const auto OrigSize = GetBufferSize();
|
||||
const auto OrigOffset = GetCursorOffset();
|
||||
|
||||
switch (Reloc->Header.Type) {
|
||||
SetBuffer(reinterpret_cast<std::uint8_t*>(Code.data()), Code.size_bytes());
|
||||
for (auto& Reloc : Relocations) {
|
||||
switch (Reloc.Header.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Reloc->NamedSymbolLiteral.Symbol);
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Reloc.NamedSymbolLiteral.Symbol);
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor
|
||||
SetCursorOffset(CursorEntry + Reloc->NamedSymbolLiteral.Offset);
|
||||
SetCursorOffset(Reloc.NamedSymbolLiteral.Offset);
|
||||
|
||||
// Generate a literal so we can place it
|
||||
dc64(Pointer);
|
||||
|
||||
DataIndex += sizeof(Reloc->NamedSymbolLiteral);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc->NamedThunkMove.Symbol));
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc.NamedThunkMove.Symbol));
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
SetCursorOffset(CursorEntry + Reloc->NamedThunkMove.Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->NamedThunkMove.RegisterIndex), Pointer, true);
|
||||
DataIndex += sizeof(Reloc->NamedThunkMove);
|
||||
SetCursorOffset(Reloc.NamedThunkMove.Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.NamedThunkMove.RegisterIndex), Pointer, true);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
|
||||
@@ -117,18 +114,27 @@ bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uin
|
||||
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
|
||||
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
|
||||
if (Pointer == ~0ULL) {
|
||||
SetBuffer(OrigBase, OrigSize);
|
||||
SetCursorOffset(OrigOffset);
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
SetCursorOffset(CursorEntry + Reloc->GuestRIPMove.Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc->GuestRIPMove.RegisterIndex), Pointer, true);
|
||||
DataIndex += sizeof(Reloc->GuestRIPMove);
|
||||
SetCursorOffset(Reloc.GuestRIPMove.Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.GuestRIPMove.RegisterIndex), Pointer, true);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
SetBuffer(OrigBase, OrigSize);
|
||||
SetCursorOffset(OrigOffset);
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
fextl::vector<FEXCore::CPU::Relocation> Arm64JITCore::TakeRelocations() {
|
||||
return std::move(Relocations);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -62,27 +62,27 @@ DEF_OP(CASPair) {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
ARMEmitter::ForwardLabel LoopNotExpected;
|
||||
ARMEmitter::ForwardLabel LoopExpected;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
|
||||
// This instruction sequence must be synced with HandleCASPAL_Armv8.
|
||||
ldaxp(EmitSize, TMP2, TMP3, MemSrc);
|
||||
cmp(EmitSize, TMP2, Expected0);
|
||||
ccmp(EmitSize, TMP3, Expected1, ARMEmitter::StatusFlags::None, ARMEmitter::Condition::CC_EQ);
|
||||
b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
|
||||
(void)b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
|
||||
stlxp(EmitSize, TMP2, Desired0, Desired1, MemSrc);
|
||||
cbnz(EmitSize, TMP2, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP2, &LoopTop);
|
||||
mov(EmitSize, Dst0, Expected0);
|
||||
mov(EmitSize, Dst1, Expected1);
|
||||
|
||||
b(&LoopExpected);
|
||||
(void)b(&LoopExpected);
|
||||
|
||||
Bind(&LoopNotExpected);
|
||||
(void)Bind(&LoopNotExpected);
|
||||
mov(EmitSize, Dst0, TMP2.R());
|
||||
mov(EmitSize, Dst1, TMP3.R());
|
||||
// exclusive monitor needs to be cleared here
|
||||
// Might have hit the case where ldaxr was hit but stlxr wasn't
|
||||
clrex();
|
||||
Bind(&LoopExpected);
|
||||
(void)Bind(&LoopExpected);
|
||||
|
||||
// Restore
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
|
||||
@@ -114,7 +114,7 @@ DEF_OP(CAS) {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
ARMEmitter::ForwardLabel LoopNotExpected;
|
||||
ARMEmitter::ForwardLabel LoopExpected;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
if (IROp->Size == IR::OpSize::i8Bit) {
|
||||
cmp(EmitSize, TMP2, Expected, ARMEmitter::ExtendedType::UXTB, 0);
|
||||
@@ -123,18 +123,18 @@ DEF_OP(CAS) {
|
||||
} else {
|
||||
cmp(EmitSize, TMP2, Expected);
|
||||
}
|
||||
b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
|
||||
(void)b(ARMEmitter::Condition::CC_NE, &LoopNotExpected);
|
||||
stlxr(SubEmitSize, TMP3, Desired, MemSrc);
|
||||
cbnz(EmitSize, TMP3, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP3, &LoopTop);
|
||||
mov(EmitSize, Dst, Expected);
|
||||
b(&LoopExpected);
|
||||
(void)b(&LoopExpected);
|
||||
|
||||
Bind(&LoopNotExpected);
|
||||
(void)Bind(&LoopNotExpected);
|
||||
mov(EmitSize, Dst, TMP2.R());
|
||||
// exclusive monitor needs to be cleared here
|
||||
// Might have hit the case where ldaxr was hit but stlxr wasn't
|
||||
clrex();
|
||||
Bind(&LoopExpected);
|
||||
(void)Bind(&LoopExpected);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -150,11 +150,11 @@ DEF_OP(AtomicXor) {
|
||||
steorl(SubEmitSize, Src, MemSrc);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
eor(EmitSize, TMP2, TMP2, Src);
|
||||
stlxr(SubEmitSize, TMP2, TMP2, MemSrc);
|
||||
cbnz(EmitSize, TMP2, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP2, &LoopTop);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -179,10 +179,10 @@ DEF_OP(AtomicSwap) {
|
||||
ldswpal(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
stlxr(SubEmitSize, TMP4, Src, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP4, &LoopTop);
|
||||
ubfm(EmitSize, GetReg(Node), TMP2, 0, IR::OpSizeAsBits(OpSize) - 1);
|
||||
}
|
||||
}
|
||||
@@ -199,11 +199,11 @@ DEF_OP(AtomicFetchAdd) {
|
||||
ldaddal(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
add(EmitSize, TMP3, TMP2, Src);
|
||||
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP4, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
}
|
||||
}
|
||||
@@ -221,11 +221,11 @@ DEF_OP(AtomicFetchSub) {
|
||||
ldaddal(SubEmitSize, TMP2, GetReg(Node), MemSrc);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
sub(EmitSize, TMP3, TMP2, Src);
|
||||
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP4, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
}
|
||||
}
|
||||
@@ -243,11 +243,11 @@ DEF_OP(AtomicFetchAnd) {
|
||||
ldclral(SubEmitSize, TMP2, GetReg(Node), MemSrc);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
and_(EmitSize, TMP3, TMP2, Src);
|
||||
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP4, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
}
|
||||
}
|
||||
@@ -264,11 +264,11 @@ DEF_OP(AtomicFetchCLR) {
|
||||
ldclral(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
bic(EmitSize, TMP3, TMP2, Src);
|
||||
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP4, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
}
|
||||
}
|
||||
@@ -285,11 +285,11 @@ DEF_OP(AtomicFetchOr) {
|
||||
ldsetal(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
orr(EmitSize, TMP3, TMP2, Src);
|
||||
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP4, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
}
|
||||
}
|
||||
@@ -306,11 +306,11 @@ DEF_OP(AtomicFetchXor) {
|
||||
ldeoral(SubEmitSize, Src, GetReg(Node), MemSrc);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
eor(EmitSize, TMP3, TMP2, Src);
|
||||
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
(void)cbnz(EmitSize, TMP4, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
}
|
||||
}
|
||||
@@ -322,13 +322,26 @@ DEF_OP(AtomicFetchNeg) {
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
neg(EmitSize, TMP3, TMP2);
|
||||
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
|
||||
cbnz(EmitSize, TMP4, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
// Use a CAS loop to avoid needing to emulate unaligned LLSC atomics
|
||||
ldr(SubEmitSize, TMP2, MemSrc);
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
(void)Bind(&LoopTop);
|
||||
mov(EmitSize, TMP4, TMP2);
|
||||
neg(EmitSize, TMP3, TMP2);
|
||||
casal(SubEmitSize, TMP2, TMP3, MemSrc);
|
||||
sub(EmitSize, TMP3, TMP2, TMP4);
|
||||
(void)cbnz(EmitSize, TMP3, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
neg(EmitSize, TMP3, TMP2);
|
||||
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
|
||||
(void)cbnz(EmitSize, TMP4, &LoopTop);
|
||||
mov(EmitSize, GetReg(Node), TMP2.R());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(TelemetrySetValue) {
|
||||
@@ -346,11 +359,11 @@ DEF_OP(TelemetrySetValue) {
|
||||
stsetl(ARMEmitter::SubRegSize::i64Bit, TMP1, TMP2);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
Bind(&LoopTop);
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(ARMEmitter::SubRegSize::i64Bit, TMP3, TMP2);
|
||||
orr(ARMEmitter::Size::i32Bit, TMP3, TMP3, Src);
|
||||
stlxr(ARMEmitter::SubRegSize::i64Bit, TMP3, TMP3, TMP2);
|
||||
cbnz(ARMEmitter::Size::i32Bit, TMP3, &LoopTop);
|
||||
(void)cbnz(ARMEmitter::Size::i32Bit, TMP3, &LoopTop);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
@@ -141,15 +141,24 @@ DEF_OP(ExitFunction) {
|
||||
if (!Op->CallReturnBlock.IsInvalid()) {
|
||||
auto CallReturnAddressReg = GetReg(Op->CallReturnAddress).X();
|
||||
PendingCallReturnTargetLabel = &CallReturnTargets.try_emplace(Op->CallReturnBlock.ID()).first->second;
|
||||
adr(TMP1, &l_CallReturn);
|
||||
(void)adr(TMP1, &l_CallReturn);
|
||||
stp<ARMEmitter::IndexType::PRE>(CallReturnAddressReg, TMP1, REG_CALLRET_SP, -0x10);
|
||||
} else {
|
||||
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::zr, ARMEmitter::XReg::zr, REG_CALLRET_SP, -0x10);
|
||||
}
|
||||
} else if (Op->Hint == IR::BranchHint::CheckTF) {
|
||||
ARMEmitter::ForwardLabel TFUnset;
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
(void)cbz(ARMEmitter::Size::i32Bit, TMP1, &TFUnset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, NewRIP);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
|
||||
ldr(TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop));
|
||||
blr(TMP2);
|
||||
(void)Bind(&TFUnset);
|
||||
}
|
||||
|
||||
EmitLinkedBranch(NewRIP, Op->Hint == IR::BranchHint::Call);
|
||||
Bind(&l_CallReturn);
|
||||
(void)Bind(&l_CallReturn);
|
||||
#ifdef _M_ARM_64EC
|
||||
}
|
||||
#endif
|
||||
@@ -161,7 +170,7 @@ DEF_OP(ExitFunction) {
|
||||
// First try to pop from the call-ret stack, otherwise follow the normal path (but ending in a ret)
|
||||
ldp<ARMEmitter::IndexType::POST>(TMP1, TMP2, REG_CALLRET_SP, 0x10);
|
||||
sub(TMP1, TMP1, RipReg.X());
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &SkipFullLookup);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &SkipFullLookup);
|
||||
}
|
||||
|
||||
// L1 Cache
|
||||
@@ -178,23 +187,23 @@ DEF_OP(ExitFunction) {
|
||||
|
||||
// Note: sub+cbnz used over cmp+br to preserve flags.
|
||||
sub(TMP1, TMP1, RipReg.X());
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &SkipFullLookup);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &SkipFullLookup);
|
||||
ldr(TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.DispatcherLoopTop));
|
||||
str(RipReg.X(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
|
||||
|
||||
Bind(&SkipFullLookup);
|
||||
(void)Bind(&SkipFullLookup);
|
||||
if (Op->Hint == IR::BranchHint::Call) {
|
||||
ARMEmitter::ForwardLabel l_CallReturn;
|
||||
if (!Op->CallReturnBlock.IsInvalid()) {
|
||||
auto CallReturnAddressReg = GetReg(Op->CallReturnAddress).X();
|
||||
PendingCallReturnTargetLabel = &CallReturnTargets.try_emplace(Op->CallReturnBlock.ID()).first->second;
|
||||
adr(TMP1, &l_CallReturn);
|
||||
(void)adr(TMP1, &l_CallReturn);
|
||||
stp<ARMEmitter::IndexType::PRE>(CallReturnAddressReg, TMP1, REG_CALLRET_SP, -0x10);
|
||||
} else {
|
||||
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::zr, ARMEmitter::XReg::zr, REG_CALLRET_SP, -0x10);
|
||||
}
|
||||
blr(TMP2);
|
||||
Bind(&l_CallReturn);
|
||||
(void)Bind(&l_CallReturn);
|
||||
} else if (Op->Hint == IR::BranchHint::Return) {
|
||||
ret(TMP2);
|
||||
} else {
|
||||
@@ -205,21 +214,20 @@ DEF_OP(ExitFunction) {
|
||||
|
||||
DEF_OP(Jump) {
|
||||
const auto Op = IROp->C<IR::IROp_Jump>();
|
||||
const auto Target = Op->TargetBlock;
|
||||
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(Target.ID()).first->second;
|
||||
PendingTargetLabel = JumpTarget(Op->TargetBlock);
|
||||
}
|
||||
|
||||
DEF_OP(CondJump) {
|
||||
auto Op = IROp->C<IR::IROp_CondJump>();
|
||||
|
||||
auto TrueTargetLabel = &JumpTargets.try_emplace(Op->TrueBlock.ID()).first->second;
|
||||
auto TrueTargetLabel = JumpTarget(Op->TrueBlock);
|
||||
|
||||
if (Op->FromNZCV) {
|
||||
b(MapCC(Op->Cond), TrueTargetLabel);
|
||||
b_OrRestart(MapCC(Op->Cond), TrueTargetLabel);
|
||||
} else {
|
||||
[[maybe_unused]] uint64_t Const;
|
||||
[[maybe_unused]] const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
|
||||
uint64_t Const;
|
||||
const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
|
||||
|
||||
auto Reg = GetReg(Op->Cmp1);
|
||||
const auto Size = Op->CompareSize == IR::OpSize::i32Bit ? ARMEmitter::Size::i32Bit : ARMEmitter::Size::i64Bit;
|
||||
@@ -229,22 +237,22 @@ DEF_OP(CondJump) {
|
||||
|
||||
if (Op->Cond.Val == FEXCore::IR::COND_EQ) {
|
||||
LOGMAN_THROW_A_FMT(Const == 0, "CondJump: Expected 0 source");
|
||||
cbz(Size, Reg, TrueTargetLabel);
|
||||
cbz_OrRestart(Size, Reg, TrueTargetLabel);
|
||||
} else if (Op->Cond.Val == FEXCore::IR::COND_NEQ) {
|
||||
LOGMAN_THROW_A_FMT(Const == 0, "CondJump: Expected 0 source");
|
||||
cbnz(Size, Reg, TrueTargetLabel);
|
||||
cbnz_OrRestart(Size, Reg, TrueTargetLabel);
|
||||
} else if (Op->Cond.Val == FEXCore::IR::COND_TSTZ) {
|
||||
LOGMAN_THROW_A_FMT(Const < 64, "CondJump: Expected valid bit source");
|
||||
tbz(Reg, Const, TrueTargetLabel);
|
||||
tbz_OrRestart(Reg, Const, TrueTargetLabel);
|
||||
} else if (Op->Cond.Val == FEXCore::IR::COND_TSTNZ) {
|
||||
LOGMAN_THROW_A_FMT(Const < 64, "CondJump: Expected valid bit source");
|
||||
tbnz(Reg, Const, TrueTargetLabel);
|
||||
tbnz_OrRestart(Reg, Const, TrueTargetLabel);
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "CondJump expected simple condition");
|
||||
}
|
||||
}
|
||||
|
||||
PendingTargetLabel = &JumpTargets.try_emplace(Op->FalseBlock.ID()).first->second;
|
||||
PendingTargetLabel = JumpTarget(Op->FalseBlock);
|
||||
}
|
||||
|
||||
DEF_OP(Syscall) {
|
||||
@@ -438,48 +446,50 @@ DEF_OP(Thunk) {
|
||||
|
||||
DEF_OP(ValidateCode) {
|
||||
auto Op = IROp->C<IR::IROp_ValidateCode>();
|
||||
const auto* OldCode = (const uint8_t*)&Op->CodeOriginalLow;
|
||||
auto OldCode = Op->CodeOriginal.data();
|
||||
auto Base = GetReg(Op->Header.Args[0]).X();
|
||||
int len = Op->CodeLength;
|
||||
int idx = 0;
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, GetReg(Node), 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Entry + Op->Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP2, 1);
|
||||
int Offset = 0;
|
||||
ARMEmitter::ForwardLabel Fail;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
while (len >= 8) {
|
||||
ldr(ARMEmitter::XReg::x2, TMP1, idx);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint64_t*)(OldCode + idx));
|
||||
cmp(ARMEmitter::Size::i64Bit, TMP3, TMP4);
|
||||
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
|
||||
len -= 8;
|
||||
idx += 8;
|
||||
}
|
||||
while (len >= 4) {
|
||||
ldr(ARMEmitter::WReg::w2, TMP1, idx);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint32_t*)(OldCode + idx));
|
||||
cmp(ARMEmitter::Size::i32Bit, TMP3, TMP4);
|
||||
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
|
||||
len -= 4;
|
||||
idx += 4;
|
||||
}
|
||||
while (len >= 2) {
|
||||
ldrh(TMP3, TMP1, idx);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint16_t*)(OldCode + idx));
|
||||
cmp(ARMEmitter::Size::i32Bit, TMP3, TMP4);
|
||||
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
|
||||
len -= 2;
|
||||
idx += 2;
|
||||
}
|
||||
while (len >= 1) {
|
||||
ldrb(TMP3, TMP1, idx);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, *(const uint8_t*)(OldCode + idx));
|
||||
cmp(ARMEmitter::Size::i32Bit, TMP3, TMP4);
|
||||
csel(ARMEmitter::Size::i64Bit, Dst, Dst, TMP2, ARMEmitter::Condition::CC_EQ);
|
||||
len -= 1;
|
||||
idx += 1;
|
||||
}
|
||||
auto EmitCheck = [&](size_t Size, auto&& LoadData) {
|
||||
while (len >= Size) {
|
||||
LoadData();
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP2);
|
||||
cbnz_OrRestart(ARMEmitter::Size::i64Bit, TMP1, &Fail);
|
||||
len -= Size;
|
||||
Offset += Size;
|
||||
}
|
||||
};
|
||||
|
||||
EmitCheck(8, [&]() {
|
||||
ldr(TMP1, Base, Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP2, *(const uint64_t*)(OldCode + Offset));
|
||||
});
|
||||
|
||||
EmitCheck(4, [&]() {
|
||||
ldr(TMP1.W(), Base, Offset);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint32_t*)(OldCode + Offset));
|
||||
});
|
||||
|
||||
EmitCheck(2, [&]() {
|
||||
ldrh(TMP1.W(), Base, Offset);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint16_t*)(OldCode + Offset));
|
||||
});
|
||||
|
||||
EmitCheck(1, [&]() {
|
||||
ldrb(TMP1.W(), Base, Offset);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint8_t*)(OldCode + Offset));
|
||||
});
|
||||
|
||||
ARMEmitter::ForwardLabel End;
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, Dst, 0);
|
||||
b_OrRestart(&End);
|
||||
BindOrRestart(&Fail);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, Dst, 1);
|
||||
BindOrRestart(&End);
|
||||
}
|
||||
|
||||
DEF_OP(ThreadRemoveCodeEntry) {
|
||||
|
||||
@@ -0,0 +1,35 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/AllocatorHooks.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
union Relocation;
|
||||
} // namespace FEXCore::CPU
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct DebugDataSubblock {
|
||||
uint32_t HostCodeOffset;
|
||||
uint32_t HostCodeSize;
|
||||
};
|
||||
|
||||
struct DebugDataGuestOpcode {
|
||||
uint64_t GuestEntryOffset;
|
||||
ptrdiff_t HostEntryOffset;
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Contains debug data for a block of code for later debugger analysis
|
||||
*
|
||||
* Needs to remain around for as long as the code could be executed at least
|
||||
*/
|
||||
struct DebugData : public FEXCore::Allocator::FEXAllocOperators {
|
||||
uint64_t HostCodeSize; ///< The size of the code generated in the host JIT
|
||||
fextl::vector<DebugDataSubblock> Subblocks;
|
||||
fextl::vector<DebugDataGuestOpcode> GuestOpcodes;
|
||||
fextl::vector<FEXCore::CPU::Relocation>* Relocations;
|
||||
};
|
||||
} // namespace FEXCore::Core
|
||||
@@ -16,7 +16,7 @@ DEF_OP(VAESImc) {
|
||||
|
||||
DEF_OP(VAESEnc) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESEnc>();
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Key = GetVReg(Op->Key);
|
||||
@@ -41,7 +41,7 @@ DEF_OP(VAESEnc) {
|
||||
|
||||
DEF_OP(VAESEncLast) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESEncLast>();
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Key = GetVReg(Op->Key);
|
||||
@@ -64,7 +64,7 @@ DEF_OP(VAESEncLast) {
|
||||
|
||||
DEF_OP(VAESDec) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESDec>();
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Key = GetVReg(Op->Key);
|
||||
@@ -89,7 +89,7 @@ DEF_OP(VAESDec) {
|
||||
|
||||
DEF_OP(VAESDecLast) {
|
||||
const auto Op = IROp->C<IR::IROp_VAESDecLast>();
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Key = GetVReg(Op->Key);
|
||||
@@ -322,7 +322,7 @@ DEF_OP(VSha256U1) {
|
||||
|
||||
DEF_OP(PCLMUL) {
|
||||
const auto Op = IROp->C<IR::IROp_PCLMUL>();
|
||||
[[maybe_unused]] const auto OpSize = IROp->Size;
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
|
||||
@@ -11,15 +11,12 @@ desc: Main glue logic of the arm64 splatter backend
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Common/SoftFloat.h"
|
||||
#include "FEXCore/Utils/Telemetry.h"
|
||||
#include "FEXCore/Utils/TypeDefines.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
#include "Interface/Core/JIT/DebugData.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include "Utils/MemberFunctionToPointer.h"
|
||||
@@ -30,15 +27,16 @@ $end_info$
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/LongJump.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
#include <FEXCore/Utils/TypeDefines.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
|
||||
#include <stdio.h>
|
||||
#include <cstdio>
|
||||
#include <cstring>
|
||||
#include <unistd.h>
|
||||
#include <string.h>
|
||||
#include <limits>
|
||||
|
||||
namespace {
|
||||
struct DivRem {
|
||||
@@ -222,7 +220,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
FillF64Result();
|
||||
} break;
|
||||
|
||||
case FABI_F64_I16_F64_PTR: {
|
||||
case FABI_F64_F64_PTR: {
|
||||
// Linux Reg/Win32 Reg:
|
||||
// tmp4 (x4/x13): FallbackHandler
|
||||
// x30: return
|
||||
@@ -239,7 +237,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
FillF64Result();
|
||||
} break;
|
||||
case FABI_F64x2_I16_F64_PTR: {
|
||||
case FABI_F64x2_F64_PTR: {
|
||||
// Linux Reg/Win32 Reg:
|
||||
// tmp4 (x4/x13): FallbackHandler
|
||||
// x30: return
|
||||
@@ -264,7 +262,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
FillF64x2Result(DstLo, DstHi);
|
||||
} break;
|
||||
|
||||
case FABI_F64_I16_F64_F64_PTR: {
|
||||
case FABI_F64_F64_F64_PTR: {
|
||||
// Linux Reg/Win32 Reg:
|
||||
// tmp4 (x4/x13): FallbackHandler
|
||||
// x30: return
|
||||
@@ -538,7 +536,8 @@ uint64_t Arm64JITCore::ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEX
|
||||
} else {
|
||||
{
|
||||
// Guard the LookupCache lock with the code invalidation mutex, to avoid issues with forking
|
||||
auto lk_inval = GuardSignalDeferringSection<std::shared_lock>(static_cast<Context::ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
auto lk_inval =
|
||||
GuardSignalDeferringSection<std::shared_lock>(static_cast<Context::ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
}
|
||||
if (!HostCode) {
|
||||
@@ -639,6 +638,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
|
||||
Common.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
|
||||
Common.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
|
||||
Common.ThreadRemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::ContextImpl::ThreadRemoveCodeEntryFromJit);
|
||||
Common.MonoBackpatcherWrite = reinterpret_cast<uint64_t>(&Context::ContextImpl::MonoBackpatcherWrite);
|
||||
Common.CPUIDObj = reinterpret_cast<uint64_t>(&CTX->CPUID);
|
||||
|
||||
{
|
||||
@@ -667,15 +667,6 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
|
||||
|
||||
CurrentCodeBuffer = CodeBuffers.GetLatest();
|
||||
ThreadState->LookupCache->Shared = CurrentCodeBuffer->LookupCache.get();
|
||||
|
||||
// Setup dynamic dispatch.
|
||||
if (ParanoidTSO()) {
|
||||
RT_LoadMemTSO = &Arm64JITCore::Op_ParanoidLoadMemTSO;
|
||||
RT_StoreMemTSO = &Arm64JITCore::Op_ParanoidStoreMemTSO;
|
||||
} else {
|
||||
RT_LoadMemTSO = &Arm64JITCore::Op_LoadMemTSO;
|
||||
RT_StoreMemTSO = &Arm64JITCore::Op_StoreMemTSO;
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64JITCore::EmitDetectionString() {
|
||||
@@ -739,48 +730,48 @@ bool Arm64JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode,
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64JITCore::EmitInterruptChecks(bool CheckTF) {
|
||||
if (CheckTF) {
|
||||
ARMEmitter::ForwardLabel l_TFUnset;
|
||||
ARMEmitter::ForwardLabel l_TFBlocked;
|
||||
void Arm64JITCore::EmitTFCheck() {
|
||||
ARMEmitter::ForwardLabel l_TFUnset;
|
||||
ARMEmitter::ForwardLabel l_TFBlocked;
|
||||
|
||||
// Note that this needs to be before the below suspend checks, as X86 checks this flag immediately after executing an instruction.
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
// Note that this needs to be before the below suspend checks, as X86 checks this flag immediately after executing an instruction.
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
|
||||
cbz(ARMEmitter::Size::i32Bit, TMP1, &l_TFUnset);
|
||||
(void)cbz(ARMEmitter::Size::i32Bit, TMP1, &l_TFUnset);
|
||||
|
||||
// X86 semantically checks TF after executing each instruction, so e.g. setting a context with TF set will execute a single instruction
|
||||
// and then raise an exception. However on the FEX side this is simpler to implement by checking at the start of each instruction, handle this by having bit 1 being unset in the flag state indicate that TF is blocked for a single instruction.
|
||||
tbz(TMP1, 1, &l_TFBlocked);
|
||||
// X86 semantically checks TF after executing each instruction, so e.g. setting a context with TF set will execute a single instruction
|
||||
// and then raise an exception. However on the FEX side this is simpler to implement by checking at the start of each instruction, handle this by having bit 1 being unset in the flag state indicate that TF is blocked for a single instruction.
|
||||
(void)tbz(TMP1, 1, &l_TFBlocked);
|
||||
|
||||
// Block TF for a single instruction when the frontend jumps to a new context by unsetting bit 1.
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
and_(ARMEmitter::Size::i32Bit, TMP1, TMP1, ~(1 << 1));
|
||||
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
// Block TF for a single instruction when the frontend jumps to a new context by unsetting bit 1.
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
and_(ARMEmitter::Size::i32Bit, TMP1, TMP1, ~(1 << 1));
|
||||
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
|
||||
Core::CpuStateFrame::SynchronousFaultDataStruct State = {
|
||||
.FaultToTopAndGeneratedException = 1,
|
||||
.Signal = Core::FAULT_SIGTRAP,
|
||||
.TrapNo = X86State::X86_TRAPNO_DB,
|
||||
.si_code = 2,
|
||||
.err_code = 0,
|
||||
};
|
||||
Core::CpuStateFrame::SynchronousFaultDataStruct State = {
|
||||
.FaultToTopAndGeneratedException = 1,
|
||||
.Signal = Core::FAULT_SIGTRAP,
|
||||
.TrapNo = X86State::X86_TRAPNO_DB,
|
||||
.si_code = 2,
|
||||
.err_code = 0,
|
||||
};
|
||||
|
||||
uint64_t Constant {};
|
||||
memcpy(&Constant, &State, sizeof(State));
|
||||
uint64_t Constant {};
|
||||
memcpy(&Constant, &State, sizeof(State));
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Constant);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
|
||||
br(TMP1);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Constant);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
|
||||
br(TMP1);
|
||||
|
||||
Bind(&l_TFBlocked);
|
||||
// If TF was blocked for this instruction, unblock it for the next.
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 0b11);
|
||||
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
Bind(&l_TFUnset);
|
||||
}
|
||||
(void)Bind(&l_TFBlocked);
|
||||
// If TF was blocked for this instruction, unblock it for the next.
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 0b11);
|
||||
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
(void)Bind(&l_TFUnset);
|
||||
}
|
||||
|
||||
void Arm64JITCore::EmitSuspendInterruptCheck() {
|
||||
if (CTX->Config.NeedsPendingInterruptFaultCheck) {
|
||||
// Trigger a fault if there are any pending interrupts
|
||||
// Used only for suspend on WIN32 at the moment
|
||||
@@ -795,17 +786,19 @@ void Arm64JITCore::EmitInterruptChecks(bool CheckTF) {
|
||||
ARMEmitter::ForwardLabel l_NoSuspend;
|
||||
cbz(ARMEmitter::Size::i32Bit, TMP2, &l_NoSuspend);
|
||||
brk(SuspendMagic);
|
||||
Bind(&l_NoSuspend);
|
||||
(void)Bind(&l_NoSuspend);
|
||||
#endif
|
||||
}
|
||||
|
||||
void Arm64JITCore::EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool CheckTF) {
|
||||
// Get the address of the JITCodeHeader and store in to the core state.
|
||||
// Two instruction cost, each 1 cycle.
|
||||
adr(TMP1, &HeaderLabel);
|
||||
adr_OrRestart(TMP1, &HeaderLabel);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, InlineJITBlockHeader));
|
||||
|
||||
EmitInterruptChecks(CheckTF);
|
||||
if (CheckTF) {
|
||||
EmitTFCheck();
|
||||
}
|
||||
|
||||
if (SpillSlots) {
|
||||
const auto TotalSpillSlotsSize = SpillSlots * MaxSpillSlotSize;
|
||||
@@ -822,15 +815,25 @@ void Arm64JITCore::EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool C
|
||||
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, bool CheckTF) {
|
||||
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
|
||||
|
||||
JumpTargets.clear();
|
||||
CallReturnTargets.clear();
|
||||
PendingJumpThunks.clear();
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
|
||||
this->Entry = Entry;
|
||||
this->DebugData = DebugData;
|
||||
this->IR = IR;
|
||||
RequiresFarARM64Jumps = false;
|
||||
|
||||
switch (static_cast<RestartOptions::Control>(FEXCore::LongJump::SetJump(RestartControl.RestartJump))) {
|
||||
case RestartOptions::Control::Incoming:
|
||||
// Nothing
|
||||
break;
|
||||
case RestartOptions::Control::EnableFarARM64Jumps: RequiresFarARM64Jumps = true; break;
|
||||
default: ERROR_AND_DIE_FMT("Unhandled Arm64 restart condition!");
|
||||
}
|
||||
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
JumpTargets.clear();
|
||||
CallReturnTargets.clear();
|
||||
PendingJumpThunks.clear();
|
||||
JumpTargets.resize(IR->GetHeader()->BlockCount, {});
|
||||
|
||||
CodeData.EntryPoints.clear();
|
||||
|
||||
// Fairly excessive buffer range to make sure we don't overflow
|
||||
@@ -845,7 +848,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
|
||||
// Put the code header at the start of the data block.
|
||||
ARMEmitter::BackwardLabel JITCodeHeaderLabel {};
|
||||
Bind(&JITCodeHeaderLabel);
|
||||
(void)Bind(&JITCodeHeaderLabel);
|
||||
JITCodeHeader* CodeHeader = GetCursorAddress<JITCodeHeader*>();
|
||||
CursorIncrement(sizeof(JITCodeHeader));
|
||||
|
||||
@@ -886,11 +889,14 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
auto BlockStartHostCode = GetCursorAddress<uint8_t*>();
|
||||
{
|
||||
const auto Node = IR->GetID(BlockNode);
|
||||
const auto IsTarget = JumpTargets.try_emplace(Node).first;
|
||||
const auto Target = &JumpTargets[BlockIROp->ID];
|
||||
|
||||
// if there's a pending branch, and it is not fall-through
|
||||
if (PendingTargetLabel && PendingTargetLabel != &IsTarget->second) {
|
||||
b(PendingTargetLabel);
|
||||
if (PendingTargetLabel && PendingTargetLabel != Target) {
|
||||
if (PendingTargetLabel->Backward.Location) {
|
||||
EmitSuspendInterruptCheck();
|
||||
}
|
||||
b_OrRestart(PendingTargetLabel);
|
||||
PendingTargetLabel = nullptr;
|
||||
}
|
||||
|
||||
@@ -900,14 +906,14 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
const auto IsReturnTarget = CallReturnTargets.try_emplace(Node).first;
|
||||
if (PendingTargetLabel) {
|
||||
// If there is a fallthrough branch to this block, skip over the entrypoint code.
|
||||
b(&IsTarget->second);
|
||||
b_OrRestart(Target);
|
||||
} else if (PendingCallReturnTargetLabel && PendingCallReturnTargetLabel != &IsReturnTarget->second) {
|
||||
// If we just emitted a call, but the block we're now emitting is not the return block so don't fallthrough.
|
||||
b(PendingCallReturnTargetLabel);
|
||||
b_OrRestart(PendingCallReturnTargetLabel);
|
||||
}
|
||||
PendingCallReturnTargetLabel = nullptr;
|
||||
|
||||
Bind(&IsReturnTarget->second);
|
||||
BindOrRestart(&IsReturnTarget->second);
|
||||
CodeData.EntryPoints.emplace(BlockStartRIP, GetCursorAddress<uint8_t*>());
|
||||
DebugData->GuestOpcodes.push_back({BlockIROp->GuestEntryOffset, GetCursorAddress<uint8_t*>() - CodeData.BlockBegin});
|
||||
|
||||
@@ -916,12 +922,12 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
|
||||
if (PendingCallReturnTargetLabel) {
|
||||
// If there is still a pending call return target, then the block we're emitting is not the return block so don't fallthrough.
|
||||
b(PendingCallReturnTargetLabel);
|
||||
b_OrRestart(PendingCallReturnTargetLabel);
|
||||
PendingCallReturnTargetLabel = nullptr;
|
||||
}
|
||||
PendingTargetLabel = nullptr;
|
||||
|
||||
Bind(&IsTarget->second);
|
||||
BindOrRestart(Target);
|
||||
}
|
||||
|
||||
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
|
||||
@@ -945,7 +951,10 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
|
||||
// Make sure last branch is generated. It certainly can't be eliminated here.
|
||||
if (PendingTargetLabel) {
|
||||
b(PendingTargetLabel);
|
||||
if (PendingTargetLabel->Backward.Location) {
|
||||
EmitSuspendInterruptCheck();
|
||||
}
|
||||
b_OrRestart(PendingTargetLabel);
|
||||
}
|
||||
PendingTargetLabel = nullptr;
|
||||
|
||||
@@ -956,21 +965,21 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
|
||||
ARMEmitter::ForwardLabel l_DoLink;
|
||||
uint64_t ThunkAddress = GetCursorAddress<uint64_t>();
|
||||
Bind(&PendingJumpThunk.Label);
|
||||
b(&l_DoLink);
|
||||
BindOrRestart(&PendingJumpThunk.Label);
|
||||
b_OrRestart(&l_DoLink);
|
||||
br(TMP1);
|
||||
Bind(&l_DoLink);
|
||||
BindOrRestart(&l_DoLink);
|
||||
ldr(TMP1, &l_ExitLink);
|
||||
blr(TMP1);
|
||||
|
||||
// This is a ExitFunctionLinkData struct
|
||||
Bind(&l_ExitLink);
|
||||
BindOrRestart(&l_ExitLink);
|
||||
dc64(0); // HostCode
|
||||
dc64(PendingJumpThunk.GuestRIP); // GuestRIP
|
||||
dc64(PendingJumpThunk.CallerAddress - ThunkAddress); // CallerOffset
|
||||
}
|
||||
|
||||
Bind(&l_ExitLink);
|
||||
BindOrRestart(&l_ExitLink);
|
||||
dc64(ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker);
|
||||
|
||||
// CodeSize not including the header or tail data.
|
||||
|
||||
@@ -19,11 +19,13 @@ $end_info$
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXCore/Utils/LongJump.h>
|
||||
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <functional>
|
||||
#include <utility>
|
||||
#include <variant>
|
||||
|
||||
@@ -53,6 +55,7 @@ public:
|
||||
|
||||
private:
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
FEX_CONFIG_OPT(HalfBarrierTSOEnabled, HALFBARRIERTSOENABLED);
|
||||
|
||||
const bool HostSupportsSVE128 {};
|
||||
const bool HostSupportsSVE256 {};
|
||||
@@ -60,6 +63,19 @@ private:
|
||||
const bool HostSupportsRPRES {};
|
||||
const bool HostSupportsAFP {};
|
||||
|
||||
struct RestartOptions {
|
||||
FEXCore::LongJump::JumpBuf RestartJump;
|
||||
enum class Control : uint64_t {
|
||||
Incoming = 0,
|
||||
EnableFarARM64Jumps = 1,
|
||||
};
|
||||
};
|
||||
|
||||
// FEXCore makes assumptions in the JIT about certain conditions being true.
|
||||
// In the rare case when those assumptions are broken, FEX needs to safely restart the JIT.
|
||||
RestartOptions RestartControl {};
|
||||
bool RequiresFarARM64Jumps {};
|
||||
|
||||
ARMEmitter::BiDirectionalLabel* PendingTargetLabel {};
|
||||
ARMEmitter::BiDirectionalLabel* PendingCallReturnTargetLabel {};
|
||||
FEXCore::Context::ContextImpl* CTX {};
|
||||
@@ -67,7 +83,13 @@ private:
|
||||
uint64_t Entry {};
|
||||
CPUBackend::CompiledCode CodeData {};
|
||||
|
||||
fextl::map<IR::NodeID, ARMEmitter::BiDirectionalLabel> JumpTargets;
|
||||
fextl::vector<ARMEmitter::BiDirectionalLabel> JumpTargets;
|
||||
|
||||
ARMEmitter::BiDirectionalLabel* JumpTarget(IR::OrderedNodeWrapper Node) {
|
||||
auto Block = IR->GetOp<IR::IROp_CodeBlock>(Node);
|
||||
return &JumpTargets[Block->ID];
|
||||
}
|
||||
|
||||
fextl::map<IR::NodeID, ARMEmitter::BiDirectionalLabel> CallReturnTargets;
|
||||
|
||||
struct PendingJumpThunk {
|
||||
@@ -238,10 +260,10 @@ private:
|
||||
case FEXCore::IR::COND_FGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_FLEU: return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_FGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_FU: return ARMEmitter::Condition::CC_VS;
|
||||
case FEXCore::IR::COND_FNU: return ARMEmitter::Condition::CC_VC;
|
||||
case FEXCore::IR::COND_VS:
|
||||
case FEXCore::IR::COND_VC:
|
||||
case FEXCore::IR::COND_FU:
|
||||
case FEXCore::IR::COND_VS: return ARMEmitter::Condition::CC_VS;
|
||||
case FEXCore::IR::COND_FNU:
|
||||
case FEXCore::IR::COND_VC: return ARMEmitter::Condition::CC_VC;
|
||||
case FEXCore::IR::COND_MI: return ARMEmitter::Condition::CC_MI;
|
||||
case FEXCore::IR::COND_PL: return ARMEmitter::Condition::CC_PL;
|
||||
default: LOGMAN_MSG_A_FMT("Unsupported compare type"); return ARMEmitter::Condition::CC_NV;
|
||||
@@ -308,14 +330,199 @@ private:
|
||||
void EmitLinkedBranch(uint64_t GuestRIP, bool Call) {
|
||||
PendingJumpThunks.push_back({GetCursorAddress<uint64_t>(), GuestRIP, {}});
|
||||
auto& Thunk = PendingJumpThunks.back();
|
||||
Bind(&Thunk.Label);
|
||||
BindOrRestart(&Thunk.Label);
|
||||
if (Call) {
|
||||
bl(&Thunk.Label);
|
||||
bl_OrRestart(&Thunk.Label);
|
||||
} else {
|
||||
b(&Thunk.Label);
|
||||
b_OrRestart(&Thunk.Label);
|
||||
}
|
||||
}
|
||||
|
||||
// Restart helpers
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
void bl_OrRestart(T* Label) {
|
||||
if (bl(Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
// We can support this but currently unnecessary.
|
||||
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
void b_OrRestart(T* Label) {
|
||||
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
// We can support this but currently unnecessary.
|
||||
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
void b_OrRestart(ARMEmitter::Condition Cond, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
|
||||
(void)b(InvertCondition(Cond), &Skip);
|
||||
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
|
||||
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
|
||||
}
|
||||
|
||||
(void)Bind(&Skip);
|
||||
return;
|
||||
}
|
||||
|
||||
if (b(Cond, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
void cbz_OrRestart(ARMEmitter::Size s, ARMEmitter::Register rt, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
|
||||
(void)cbnz(s, rt, &Skip);
|
||||
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
|
||||
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
|
||||
}
|
||||
|
||||
(void)Bind(&Skip);
|
||||
return;
|
||||
}
|
||||
|
||||
if (cbz(s, rt, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
void cbnz_OrRestart(ARMEmitter::Size s, ARMEmitter::Register rt, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
|
||||
(void)cbz(s, rt, &Skip);
|
||||
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
|
||||
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
|
||||
}
|
||||
|
||||
(void)Bind(&Skip);
|
||||
return;
|
||||
}
|
||||
|
||||
if (cbnz(s, rt, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
void tbz_OrRestart(ARMEmitter::Register rt, uint32_t Bit, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
|
||||
(void)tbnz(rt, Bit, &Skip);
|
||||
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
|
||||
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
|
||||
}
|
||||
|
||||
(void)Bind(&Skip);
|
||||
return;
|
||||
}
|
||||
|
||||
if (tbz(rt, Bit, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
void tbnz_OrRestart(ARMEmitter::Register rt, uint32_t Bit, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
// Wrap a manual Cond check around an unconditional branch; this can encode larger offsets
|
||||
(void)tbz(rt, Bit, &Skip);
|
||||
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
|
||||
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
|
||||
}
|
||||
|
||||
(void)Bind(&Skip);
|
||||
return;
|
||||
}
|
||||
|
||||
if (tbnz(rt, Bit, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
void adr_OrRestart(ARMEmitter::Register rd, T* Label) {
|
||||
if (adr(rd, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
// We can support this but currently unnecessary.
|
||||
ERROR_AND_DIE_FMT("Long ADR currently unsupported!");
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
void adrp_OrRestart(ARMEmitter::Register rd, T* Label) {
|
||||
if (adrp(rd, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
// We can support this but currently unnecessary.
|
||||
ERROR_AND_DIE_FMT("Long ADRP currently unsupported!");
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
void BindOrRestart(T* Label) {
|
||||
if (Bind(Label)) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (RequiresFarARM64Jumps) {
|
||||
// This should have been caught before this point.
|
||||
ERROR_AND_DIE_FMT("Oops. Unhandled long bind.");
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
|
||||
void EmitDetectionString();
|
||||
IR::RegisterAllocationPass* RAPass {};
|
||||
@@ -374,7 +581,9 @@ private:
|
||||
fextl::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
|
||||
///< Relocation code loading
|
||||
bool ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations);
|
||||
bool ApplyRelocations(uint64_t GuestEntry, std::span<std::byte> Code, std::span<const FEXCore::CPU::Relocation>);
|
||||
|
||||
fextl::vector<FEXCore::CPU::Relocation> TakeRelocations() override;
|
||||
|
||||
/** @} */
|
||||
|
||||
@@ -399,21 +608,14 @@ private:
|
||||
std::optional<ARMEmitter::VRegister> VectorIndexHigh, ARMEmitter::VRegister MaskReg, IR::OpSize VectorIndexSize,
|
||||
size_t DataElementOffsetStart, size_t IndexElementOffsetStart, uint8_t OffsetScale);
|
||||
|
||||
void EmitInterruptChecks(bool CheckTF);
|
||||
void EmitTFCheck();
|
||||
|
||||
void EmitSuspendInterruptCheck();
|
||||
|
||||
void EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool CheckTF);
|
||||
|
||||
// Runtime selection;
|
||||
// Load and store TSO memory style
|
||||
OpType RT_LoadMemTSO;
|
||||
OpType RT_StoreMemTSO;
|
||||
|
||||
#define DEF_OP(x) void Op_##x(IR::IROp_Header const* IROp, IR::Ref Node)
|
||||
|
||||
// Dynamic Dispatcher supporting operations
|
||||
DEF_OP(ParanoidLoadMemTSO);
|
||||
DEF_OP(ParanoidStoreMemTSO);
|
||||
|
||||
///< Unhandled handler
|
||||
DEF_OP(Unhandled);
|
||||
|
||||
|
||||
@@ -140,7 +140,7 @@ DEF_OP(LoadRegister) {
|
||||
|
||||
mov(GetReg(Node).X(), StaticRegisters[Op->Reg].X());
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
[[maybe_unused]] const auto regSize = HostSupportsAVX256 ? IR::OpSize::i256Bit : IR::OpSize::i128Bit;
|
||||
const auto regSize = HostSupportsAVX256 ? IR::OpSize::i256Bit : IR::OpSize::i128Bit;
|
||||
LOGMAN_THROW_A_FMT(Op->Reg < StaticFPRegisters.size(), "out of range reg");
|
||||
LOGMAN_THROW_A_FMT(IROp->Size == regSize, "expected sized");
|
||||
|
||||
@@ -181,7 +181,7 @@ DEF_OP(StoreRegister) {
|
||||
// Always use 64-bit, it's faster. Upper bits ignored for 32-bit mode.
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Reg), GetReg(Op->Value));
|
||||
} else if (Reg.Class == IR::FPRFixedClass) {
|
||||
[[maybe_unused]] const auto regSize = HostSupportsAVX256 ? IR::OpSize::i256Bit : IR::OpSize::i128Bit;
|
||||
const auto regSize = HostSupportsAVX256 ? IR::OpSize::i256Bit : IR::OpSize::i128Bit;
|
||||
LOGMAN_THROW_A_FMT(IROp->Size == regSize, "expected sized");
|
||||
|
||||
const auto guest = GetVReg(Reg);
|
||||
@@ -348,6 +348,25 @@ DEF_OP(StoreContextIndexed) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(FormContextAddress) {
|
||||
const auto Op = IROp->C<IR::IROp_FormContextAddress>();
|
||||
const auto Index = GetReg(Op->Index);
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
switch (Op->Stride) {
|
||||
case 1:
|
||||
case 2:
|
||||
case 4:
|
||||
case 8:
|
||||
case 16:
|
||||
case 32: {
|
||||
add(ARMEmitter::Size::i64Bit, Dst, STATE, Index, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Op->Stride));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled FormContextAddress stride: {}", Op->Stride); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(SpillRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_SpillRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -742,7 +761,7 @@ DEF_OP(LoadMemTSO) {
|
||||
const auto Dst = GetReg(Node);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
[[maybe_unused]] bool IsInline = IsInlineConstant(Op->Offset, &Offset);
|
||||
bool IsInline = IsInlineConstant(Op->Offset, &Offset);
|
||||
LOGMAN_THROW_A_FMT(IsInline, "expected immediate");
|
||||
}
|
||||
|
||||
@@ -757,8 +776,10 @@ DEF_OP(LoadMemTSO) {
|
||||
case IR::OpSize::i64Bit: ldapur(Dst.X(), MemReg, Offset); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
}
|
||||
} else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
@@ -772,8 +793,10 @@ DEF_OP(LoadMemTSO) {
|
||||
case IR::OpSize::i64Bit: ldapr(Dst.X(), MemReg); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
@@ -787,8 +810,10 @@ DEF_OP(LoadMemTSO) {
|
||||
case IR::OpSize::i64Bit: ldar(Dst.X(), MemReg); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
}
|
||||
} else {
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -892,7 +917,7 @@ DEF_OP(VLoadVectorMasked) {
|
||||
|
||||
// If the sign bit is zero then skip the load
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
|
||||
(void)tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
|
||||
// Do the gather load for this element into the destination
|
||||
switch (IROp->ElementSize) {
|
||||
case IR::OpSize::i8Bit: ld1<ARMEmitter::SubRegSize::i8Bit>(TempDst.Q(), i, TempMemReg); break;
|
||||
@@ -903,7 +928,7 @@ DEF_OP(VLoadVectorMasked) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, IROp->ElementSize); return;
|
||||
}
|
||||
|
||||
Bind(&Skip);
|
||||
(void)Bind(&Skip);
|
||||
|
||||
if ((i + 1) != NumElements) {
|
||||
// Handle register rename to save a move.
|
||||
@@ -993,7 +1018,7 @@ DEF_OP(VStoreVectorMasked) {
|
||||
|
||||
// If the sign bit is zero then skip the load
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
|
||||
(void)tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
|
||||
// Do the gather load for this element into the destination
|
||||
switch (IROp->ElementSize) {
|
||||
case IR::OpSize::i8Bit: st1<ARMEmitter::SubRegSize::i8Bit>(RegData.Q(), i, TempMemReg); break;
|
||||
@@ -1004,7 +1029,7 @@ DEF_OP(VStoreVectorMasked) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, IROp->ElementSize); return;
|
||||
}
|
||||
|
||||
Bind(&Skip);
|
||||
(void)Bind(&Skip);
|
||||
|
||||
if ((i + 1) != NumElements) {
|
||||
// Handle register rename to save a move.
|
||||
@@ -1082,7 +1107,7 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
|
||||
PerformMove(ElementSize, WorkingReg, MaskReg, i);
|
||||
|
||||
// Skip if the mask's sign bit isn't set
|
||||
tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
|
||||
(void)tbz(WorkingReg, ElementSizeInBits - 1, &Skip);
|
||||
|
||||
// Extract Index Element
|
||||
if ((IndexElement * IR::OpSizeToSize(VectorIndexSize)) >= 16) {
|
||||
@@ -1120,7 +1145,7 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, ElementSize); FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
Bind(&Skip);
|
||||
(void)Bind(&Skip);
|
||||
}
|
||||
|
||||
if (NeedsDestTmp) {
|
||||
@@ -1750,7 +1775,7 @@ DEF_OP(StoreMemTSO) {
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
[[maybe_unused]] bool IsInline = IsInlineConstant(Op->Offset, &Offset);
|
||||
bool IsInline = IsInlineConstant(Op->Offset, &Offset);
|
||||
LOGMAN_THROW_A_FMT(IsInline, "expected immediate");
|
||||
}
|
||||
|
||||
@@ -1758,8 +1783,10 @@ DEF_OP(StoreMemTSO) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
stlurb(Src, MemReg, Offset);
|
||||
} else {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i16Bit: stlurh(Src, MemReg, Offset); break;
|
||||
case IR::OpSize::i32Bit: stlur(Src.W(), MemReg, Offset); break;
|
||||
@@ -1774,8 +1801,10 @@ DEF_OP(StoreMemTSO) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
stlrb(Src, MemReg);
|
||||
} else {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i16Bit: stlrh(Src, MemReg); break;
|
||||
case IR::OpSize::i32Bit: stlr(Src.W(), MemReg); break;
|
||||
@@ -1851,7 +1880,7 @@ DEF_OP(MemSet) {
|
||||
|
||||
if (!DirectionIsInline) {
|
||||
// Backward or forwards implementation depends on flag
|
||||
tbnz(DirectionReg, 1, &BackwardImpl);
|
||||
(void)tbnz(DirectionReg, 1, &BackwardImpl);
|
||||
}
|
||||
|
||||
auto MemStore = [this](auto Value, uint32_t OpSize, int32_t Size) {
|
||||
@@ -1869,7 +1898,9 @@ DEF_OP(MemSet) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
stlrb(Value.W(), TMP2);
|
||||
} else {
|
||||
nop();
|
||||
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
|
||||
nop();
|
||||
}
|
||||
switch (OpSize) {
|
||||
case 2: stlrh(Value.W(), TMP2); break;
|
||||
case 4: stlr(Value.W(), TMP2); break;
|
||||
@@ -1899,7 +1930,7 @@ DEF_OP(MemSet) {
|
||||
ARMEmitter::ForwardLabel DoneInternal {};
|
||||
|
||||
// Early exit if zero count.
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
|
||||
if (!IsAtomic) {
|
||||
ARMEmitter::ForwardLabel AgainInternal256Exit {};
|
||||
@@ -1916,50 +1947,50 @@ DEF_OP(MemSet) {
|
||||
// Do this in two parts, to fallback to the byte by byte loop if size < 32, and to the
|
||||
// single copy loop if size < 64.
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
tbnz(TMP1, 63, &AgainInternal128Exit);
|
||||
(void)tbnz(TMP1, 63, &AgainInternal128Exit);
|
||||
|
||||
// Fill VTMP2 with the set pattern
|
||||
dup(SubRegSize, VTMP2.Q(), Value);
|
||||
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
tbnz(TMP1, 63, &AgainInternal256Exit);
|
||||
(void)tbnz(TMP1, 63, &AgainInternal256Exit);
|
||||
|
||||
Bind(&AgainInternal256);
|
||||
(void)Bind(&AgainInternal256);
|
||||
stp<ARMEmitter::IndexType::POST>(VTMP2.Q(), VTMP2.Q(), TMP2, 32 * Direction);
|
||||
stp<ARMEmitter::IndexType::POST>(VTMP2.Q(), VTMP2.Q(), TMP2, 32 * Direction);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
|
||||
tbz(TMP1, 63, &AgainInternal256);
|
||||
(void)tbz(TMP1, 63, &AgainInternal256);
|
||||
|
||||
Bind(&AgainInternal256Exit);
|
||||
(void)Bind(&AgainInternal256Exit);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
tbnz(TMP1, 63, &AgainInternal128Exit);
|
||||
Bind(&AgainInternal128);
|
||||
(void)tbnz(TMP1, 63, &AgainInternal128Exit);
|
||||
(void)Bind(&AgainInternal128);
|
||||
stp<ARMEmitter::IndexType::POST>(VTMP2.Q(), VTMP2.Q(), TMP2, 32 * Direction);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
tbz(TMP1, 63, &AgainInternal128);
|
||||
(void)tbz(TMP1, 63, &AgainInternal128);
|
||||
|
||||
Bind(&AgainInternal128Exit);
|
||||
(void)Bind(&AgainInternal128Exit);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
|
||||
if (Direction == -1) {
|
||||
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
|
||||
}
|
||||
}
|
||||
|
||||
Bind(&AgainInternal);
|
||||
(void)Bind(&AgainInternal);
|
||||
if (IsAtomic) {
|
||||
MemStoreTSO(Value, OpSize, SizeDirection);
|
||||
} else {
|
||||
MemStore(Value, OpSize, SizeDirection);
|
||||
}
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 1);
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
|
||||
(void)cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
|
||||
|
||||
Bind(&DoneInternal);
|
||||
(void)Bind(&DoneInternal);
|
||||
|
||||
if (SizeDirection >= 0) {
|
||||
switch (OpSize) {
|
||||
@@ -1989,12 +2020,12 @@ DEF_OP(MemSet) {
|
||||
EmitMemset(Direction);
|
||||
|
||||
if (Direction == 1) {
|
||||
b(&Done);
|
||||
Bind(&BackwardImpl);
|
||||
(void)b(&Done);
|
||||
(void)Bind(&BackwardImpl);
|
||||
}
|
||||
}
|
||||
|
||||
Bind(&Done);
|
||||
(void)Bind(&Done);
|
||||
// Destination already set to the final pointer.
|
||||
}
|
||||
}
|
||||
@@ -2044,7 +2075,7 @@ DEF_OP(MemCpy) {
|
||||
|
||||
if (!DirectionIsInline) {
|
||||
// Backward or forwards implementation depends on flag
|
||||
tbnz(DirectionReg, 1, &BackwardImpl);
|
||||
(void)tbnz(DirectionReg, 1, &BackwardImpl);
|
||||
}
|
||||
|
||||
auto MemCpy = [this](uint32_t OpSize, int32_t Size) {
|
||||
@@ -2087,9 +2118,11 @@ DEF_OP(MemCpy) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size); break;
|
||||
}
|
||||
|
||||
// Placeholders for backpatching barriers (one per load/store)
|
||||
nop();
|
||||
nop();
|
||||
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
|
||||
// Placeholders for backpatching barriers (one per load/store)
|
||||
nop();
|
||||
nop();
|
||||
}
|
||||
|
||||
switch (OpSize) {
|
||||
case 2: stlrh(TMP4.W(), TMP2); break;
|
||||
@@ -2111,9 +2144,11 @@ DEF_OP(MemCpy) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size); break;
|
||||
}
|
||||
|
||||
// Placeholders for backpatching barriers (one per load/store)
|
||||
nop();
|
||||
nop();
|
||||
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
|
||||
// Placeholders for backpatching barriers (one per load/store)
|
||||
nop();
|
||||
nop();
|
||||
}
|
||||
|
||||
switch (OpSize) {
|
||||
case 2: stlrh(TMP4.W(), TMP2); break;
|
||||
@@ -2141,7 +2176,7 @@ DEF_OP(MemCpy) {
|
||||
ARMEmitter::ForwardLabel DoneInternal {};
|
||||
|
||||
// Early exit if zero count.
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
|
||||
if (!IsAtomic) {
|
||||
ARMEmitter::ForwardLabel AbsPos {};
|
||||
@@ -2151,11 +2186,11 @@ DEF_OP(MemCpy) {
|
||||
ARMEmitter::BackwardLabel AgainInternal256 {};
|
||||
|
||||
sub(ARMEmitter::Size::i64Bit, TMP4, TMP2, TMP3);
|
||||
tbz(TMP4, 63, &AbsPos);
|
||||
(void)tbz(TMP4, 63, &AbsPos);
|
||||
neg(ARMEmitter::Size::i64Bit, TMP4, TMP4);
|
||||
Bind(&AbsPos);
|
||||
(void)Bind(&AbsPos);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP4, TMP4, 32);
|
||||
tbnz(TMP4, 63, &AgainInternal);
|
||||
(void)tbnz(TMP4, 63, &AgainInternal);
|
||||
|
||||
if (Direction == -1) {
|
||||
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
|
||||
@@ -2167,30 +2202,30 @@ DEF_OP(MemCpy) {
|
||||
// Do this in two parts, to fallback to the byte by byte loop if size < 32, and to the
|
||||
// single copy loop if size < 64.
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
tbnz(TMP1, 63, &AgainInternal128Exit);
|
||||
(void)tbnz(TMP1, 63, &AgainInternal128Exit);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
tbnz(TMP1, 63, &AgainInternal256Exit);
|
||||
(void)tbnz(TMP1, 63, &AgainInternal256Exit);
|
||||
|
||||
Bind(&AgainInternal256);
|
||||
(void)Bind(&AgainInternal256);
|
||||
MemCpy(32, 32 * Direction);
|
||||
MemCpy(32, 32 * Direction);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
|
||||
tbz(TMP1, 63, &AgainInternal256);
|
||||
(void)tbz(TMP1, 63, &AgainInternal256);
|
||||
|
||||
Bind(&AgainInternal256Exit);
|
||||
(void)Bind(&AgainInternal256Exit);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 64 / Size);
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
tbnz(TMP1, 63, &AgainInternal128Exit);
|
||||
Bind(&AgainInternal128);
|
||||
(void)tbnz(TMP1, 63, &AgainInternal128Exit);
|
||||
(void)Bind(&AgainInternal128);
|
||||
MemCpy(32, 32 * Direction);
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
tbz(TMP1, 63, &AgainInternal128);
|
||||
(void)tbz(TMP1, 63, &AgainInternal128);
|
||||
|
||||
Bind(&AgainInternal128Exit);
|
||||
(void)Bind(&AgainInternal128Exit);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, 32 / Size);
|
||||
cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &DoneInternal);
|
||||
|
||||
if (Direction == -1) {
|
||||
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 32 - Size);
|
||||
@@ -2198,16 +2233,16 @@ DEF_OP(MemCpy) {
|
||||
}
|
||||
}
|
||||
|
||||
Bind(&AgainInternal);
|
||||
(void)Bind(&AgainInternal);
|
||||
if (IsAtomic) {
|
||||
MemCpyTSO(OpSize, SizeDirection);
|
||||
} else {
|
||||
MemCpy(OpSize, SizeDirection);
|
||||
}
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, TMP1, 1);
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
|
||||
(void)cbnz(ARMEmitter::Size::i64Bit, TMP1, &AgainInternal);
|
||||
|
||||
Bind(&DoneInternal);
|
||||
(void)Bind(&DoneInternal);
|
||||
|
||||
// Needs to use temporaries just in case of overwrite
|
||||
mov(TMP1, MemRegDest.X());
|
||||
@@ -2265,186 +2300,15 @@ DEF_OP(MemCpy) {
|
||||
for (int32_t Direction : {1, -1}) {
|
||||
EmitMemcpy(Direction);
|
||||
if (Direction == 1) {
|
||||
b(&Done);
|
||||
Bind(&BackwardImpl);
|
||||
(void)b(&Done);
|
||||
(void)Bind(&BackwardImpl);
|
||||
}
|
||||
}
|
||||
Bind(&Done);
|
||||
(void)Bind(&Done);
|
||||
// Destination already set to the final pointer.
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(ParanoidLoadMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
if (!IsInlineConstant(Op->Offset, &Offset)) {
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
}
|
||||
}
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
const auto Dst = GetReg(Node);
|
||||
ldapurb(Dst, MemReg, Offset);
|
||||
} else {
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i16Bit: ldapurh(Dst, MemReg, Offset); break;
|
||||
case IR::OpSize::i32Bit: ldapur(Dst.W(), MemReg, Offset); break;
|
||||
case IR::OpSize::i64Bit: ldapur(Dst.X(), MemReg, Offset); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
} else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
ldaprb(Dst.W(), MemReg);
|
||||
} else {
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i16Bit: ldaprh(Dst.W(), MemReg); break;
|
||||
case IR::OpSize::i32Bit: ldapr(Dst.W(), MemReg); break;
|
||||
case IR::OpSize::i64Bit: ldapr(Dst.X(), MemReg); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: ldarb(Dst, MemReg); break;
|
||||
case IR::OpSize::i16Bit: ldarh(Dst, MemReg); break;
|
||||
case IR::OpSize::i32Bit: ldar(Dst.W(), MemReg); break;
|
||||
case IR::OpSize::i64Bit: ldar(Dst.X(), MemReg); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
const auto Dst = GetVReg(Node);
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit:
|
||||
ldarb(TMP1, MemReg);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
|
||||
break;
|
||||
case IR::OpSize::i16Bit:
|
||||
ldarh(TMP1, MemReg);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
|
||||
break;
|
||||
case IR::OpSize::i32Bit:
|
||||
ldar(TMP1.W(), MemReg);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
|
||||
break;
|
||||
case IR::OpSize::i64Bit:
|
||||
ldar(TMP1, MemReg);
|
||||
fmov(ARMEmitter::Size::i64Bit, Dst.D(), TMP1);
|
||||
break;
|
||||
case IR::OpSize::i128Bit:
|
||||
ldaxp(ARMEmitter::Size::i64Bit, TMP1, TMP2, MemReg);
|
||||
clrex();
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, TMP1);
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 1, TMP2);
|
||||
break;
|
||||
case IR::OpSize::i256Bit:
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), PRED_TMP_32B.Zeroing(), MemReg);
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(ParanoidStoreMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
if (!IsInlineConstant(Op->Offset, &Offset)) {
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP1, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
}
|
||||
}
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
stlurb(Src, MemReg, Offset);
|
||||
} else {
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i16Bit: stlurh(Src, MemReg, Offset); break;
|
||||
case IR::OpSize::i32Bit: stlur(Src.W(), MemReg, Offset); break;
|
||||
case IR::OpSize::i64Bit: stlur(Src.X(), MemReg, Offset); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP1, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: stlrb(Src, MemReg); break;
|
||||
case IR::OpSize::i16Bit: stlrh(Src, MemReg); break;
|
||||
case IR::OpSize::i32Bit: stlr(Src.W(), MemReg); break;
|
||||
case IR::OpSize::i64Bit: stlr(Src.X(), MemReg); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
const auto Src = GetVReg(Op->Value);
|
||||
|
||||
MemReg = ApplyMemOperand(OpSize, MemReg, TMP4, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit:
|
||||
umov<ARMEmitter::SubRegSize::i8Bit>(TMP1, Src, 0);
|
||||
stlrb(TMP1, MemReg);
|
||||
break;
|
||||
case IR::OpSize::i16Bit:
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(TMP1, Src, 0);
|
||||
stlrh(TMP1, MemReg);
|
||||
break;
|
||||
case IR::OpSize::i32Bit:
|
||||
umov<ARMEmitter::SubRegSize::i32Bit>(TMP1, Src, 0);
|
||||
stlr(TMP1.W(), MemReg);
|
||||
break;
|
||||
case IR::OpSize::i64Bit:
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(TMP1, Src, 0);
|
||||
stlr(TMP1, MemReg);
|
||||
break;
|
||||
case IR::OpSize::i128Bit: {
|
||||
// Move vector to GPRs
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(TMP1, Src, 0);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(TMP2, Src, 1);
|
||||
ARMEmitter::BackwardLabel B;
|
||||
Bind(&B);
|
||||
|
||||
// ldaxp must not have both the destination registers be the same
|
||||
ldaxp(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::zr, TMP3, MemReg); // <- Can hit SIGBUS. Overwritten with DMB
|
||||
stlxp(ARMEmitter::Size::i64Bit, TMP3, TMP1, TMP2, MemReg); // <- Can also hit SIGBUS
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP3, &B); // < Overwritten with DMB
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i256Bit: {
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(Src.Z(), PRED_TMP_32B, MemReg, 0);
|
||||
dmb(ARMEmitter::BarrierScope::ISH);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(CacheLineClear) {
|
||||
if (!CTX->HostFeatures.SupportsCacheMaintenanceOps) {
|
||||
dmb(ARMEmitter::BarrierScope::SY);
|
||||
@@ -2586,7 +2450,7 @@ DEF_OP(VStoreNonTemporalPair) {
|
||||
const auto Op = IROp->C<IR::IROp_VStoreNonTemporalPair>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
[[maybe_unused]] const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
|
||||
const auto Is128Bit = OpSize == IR::OpSize::i128Bit;
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "This IR operation only operates at 128-bit wide");
|
||||
|
||||
const auto ValueLow = GetVReg(Op->ValueLow);
|
||||
|
||||
@@ -10,10 +10,11 @@ $end_info$
|
||||
#endif
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/JIT/DebugData.h"
|
||||
#include "Interface/Core/JIT/JITClass.h"
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
@@ -286,4 +287,43 @@ DEF_OP(Yield) {
|
||||
yield();
|
||||
}
|
||||
|
||||
DEF_OP(MonoBackpatcherWrite) {
|
||||
auto Op = IROp->C<IR::IROp_MonoBackpatcherWrite>();
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, TMP3, GetReg(Op->Addr));
|
||||
mov(ARMEmitter::Size::i64Bit, TMP4, GetReg(Op->Value));
|
||||
|
||||
PushDynamicRegs(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, STATE.R());
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, IR::OpSizeToSize(Op->Size));
|
||||
|
||||
if (!TMP_ABIARGS) {
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, TMP3);
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, TMP4);
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ldr(TMP2, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 1);
|
||||
strb(TMP1.W(), TMP2, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
|
||||
#endif
|
||||
|
||||
ldr(ARMEmitter::XReg::x4, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.MonoBackpatcherWrite));
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<void, void*, uint8_t, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
|
||||
} else {
|
||||
blr(ARMEmitter::Reg::r4);
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ldr(TMP2, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
|
||||
strb(ARMEmitter::WReg::zr, TMP2, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
|
||||
#endif
|
||||
|
||||
FillStaticRegs();
|
||||
PopDynamicRegs();
|
||||
}
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
File renamed without changes.
@@ -1352,7 +1352,7 @@ DEF_OP(VFMin) {
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
[[maybe_unused]] const auto IsScalar = ElementSize == OpSize;
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
@@ -1425,7 +1425,7 @@ DEF_OP(VFMax) {
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
[[maybe_unused]] const auto IsScalar = ElementSize == OpSize;
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
|
||||
@@ -1,86 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
// If any of the config options mismatch on load then the cache won't be used
|
||||
// Any of these will result in codegen changes
|
||||
struct FEX_PACKED CodeObjectSerializationConfig {
|
||||
// Cookie in the header of the file, isn't part of the config hash
|
||||
uint64_t Cookie {};
|
||||
|
||||
// Instructions per block configuration
|
||||
int32_t MaxInstPerBlock {};
|
||||
|
||||
// Follows CPUID 4000_0001_EAX[3:0]
|
||||
unsigned Arch : 4;
|
||||
|
||||
// Multiblock enabled
|
||||
unsigned MultiBlock : 1;
|
||||
|
||||
// Hardware TSO enabled
|
||||
unsigned HardwareTSOEnabled : 1;
|
||||
|
||||
// TSO enabled
|
||||
unsigned TSOEnabled : 1;
|
||||
|
||||
// ABI local flag unsafe optimization
|
||||
unsigned ABILocalFlags : 1;
|
||||
|
||||
// Paranoid TSO mode enabled
|
||||
unsigned ParanoidTSO : 1;
|
||||
|
||||
// Guest code execution mode (We don't support live mode switch)
|
||||
unsigned Is64BitMode : 1;
|
||||
|
||||
// SMC checks style
|
||||
unsigned SMCChecks : 2;
|
||||
|
||||
// x87 reduced precision
|
||||
unsigned x87ReducedPrecision : 1;
|
||||
|
||||
// Padding to remove uninitialized data warning from asan
|
||||
// Shows remaining amount of bits available for config
|
||||
unsigned _Pad : 19;
|
||||
|
||||
bool operator==(const CodeObjectSerializationConfig& other) const {
|
||||
return Cookie == other.Cookie && MaxInstPerBlock == other.MaxInstPerBlock && Arch == other.Arch && MultiBlock == other.MultiBlock &&
|
||||
HardwareTSOEnabled == other.HardwareTSOEnabled && TSOEnabled == other.TSOEnabled && ABILocalFlags == other.ABILocalFlags &&
|
||||
ParanoidTSO == other.ParanoidTSO && Is64BitMode == other.Is64BitMode && SMCChecks == other.SMCChecks &&
|
||||
x87ReducedPrecision == other.x87ReducedPrecision;
|
||||
}
|
||||
static uint64_t GetHash(const CodeObjectSerializationConfig& other) {
|
||||
// For < 64-bits of data just pack directly
|
||||
// Skip the cookie
|
||||
uint64_t Hash {};
|
||||
Hash <<= 32;
|
||||
Hash |= other.MaxInstPerBlock;
|
||||
Hash <<= 1;
|
||||
Hash |= other.Arch;
|
||||
Hash <<= 1;
|
||||
Hash |= other.MultiBlock;
|
||||
Hash <<= 1;
|
||||
Hash |= other.HardwareTSOEnabled;
|
||||
Hash <<= 1;
|
||||
Hash |= other.TSOEnabled;
|
||||
Hash <<= 1;
|
||||
Hash |= other.ABILocalFlags;
|
||||
Hash <<= 1;
|
||||
Hash |= other.ParanoidTSO;
|
||||
Hash <<= 1;
|
||||
Hash |= other.Is64BitMode;
|
||||
Hash <<= 2;
|
||||
Hash |= other.SMCChecks;
|
||||
Hash <<= 1;
|
||||
Hash |= other.x87ReducedPrecision;
|
||||
return Hash;
|
||||
}
|
||||
};
|
||||
|
||||
static_assert(sizeof(CodeObjectSerializationConfig) == 16, "Size changed");
|
||||
static_assert((sizeof(CodeObjectSerializationConfig) - sizeof(uint64_t)) == 8, "Config size exceeded 64its. Need to change how the hash is "
|
||||
"generated!");
|
||||
} // namespace FEXCore::CodeSerialize
|
||||
@@ -1,121 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXHeaderUtils/Filesystem.h>
|
||||
|
||||
#include <fcntl.h>
|
||||
#include <xxhash.h>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
void AsyncJobHandler::AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string& filename) {
|
||||
#ifndef _WIN32
|
||||
// This function adds a named region *JOB* to our named region handler
|
||||
// This needs to be as fast as possible to keep out of the way of the JIT
|
||||
|
||||
const fextl::string BaseFilename = FHU::Filesystem::GetFilename(filename);
|
||||
|
||||
if (!BaseFilename.empty()) {
|
||||
// Create a new entry that once set up will be put in to our section object map
|
||||
auto Entry = fextl::make_unique<CodeRegionEntry>(Base, Size, Offset, filename, NamedRegionHandler->DefaultCodeHeader(Base, Offset));
|
||||
|
||||
// Lock the job ref counter so we can block anything attempting to use the entry before it is loaded
|
||||
Entry->NamedJobRefCountMutex.lock();
|
||||
|
||||
CodeRegionMapType::iterator EntryIterator;
|
||||
{
|
||||
std::unique_lock lk {CodeObjectCacheService->GetEntryMapMutex()};
|
||||
|
||||
auto& EntryMap = CodeObjectCacheService->GetEntryMap();
|
||||
|
||||
auto it = EntryMap.emplace(Base, std::move(Entry));
|
||||
if (!it.second) {
|
||||
// This happens when an application overwrites a previous region without unmapping what was there
|
||||
|
||||
// Lock this entry's Named job reference counter.
|
||||
// Once this passes then we know that this section has been loaded.
|
||||
it.first->second->NamedJobRefCountMutex.lock();
|
||||
|
||||
// Finalize anything the region needs to do first.
|
||||
CodeObjectCacheService->DoCodeRegionClosure(it.first->second->Base, it.first->second.get());
|
||||
|
||||
// munmap the file that was mapped
|
||||
FEXCore::Allocator::munmap(it.first->second->CodeData, it.first->second->FileSize);
|
||||
|
||||
// Remove this entry from the unrelocated map as well
|
||||
{
|
||||
std::unique_lock lk2 {CodeObjectCacheService->GetUnrelocatedEntryMapMutex()};
|
||||
CodeObjectCacheService->GetUnrelocatedEntryMap().erase(it.first->second->EntryHeader.OriginalBase);
|
||||
}
|
||||
|
||||
// Now overwrite the entry in the map
|
||||
it = EntryMap.insert_or_assign(Base, std::move(Entry));
|
||||
EntryIterator = it.first;
|
||||
} else {
|
||||
// No overwrite, just insert
|
||||
EntryIterator = it.first;
|
||||
}
|
||||
}
|
||||
|
||||
// Now that this entry has been added to the map, we can insert a load job using the entry iterator.
|
||||
// This allows us to quickly unblock the JIT thread when it is loading multiple regions and have the async thread
|
||||
// do the loading for us.
|
||||
//
|
||||
// Create the async work queue job now so it can load
|
||||
NamedRegionHandler->AsyncAddNamedRegionWorkItem(BaseFilename, filename, true, EntryIterator);
|
||||
|
||||
// Tell the async thread that it has work to do
|
||||
CodeObjectCacheService->NotifyWork();
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void AsyncJobHandler::AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size) {
|
||||
#ifndef _WIN32
|
||||
// Removing a named region through the job system
|
||||
// We need to find the entry that we are deleting first
|
||||
fextl::unique_ptr<CodeRegionEntry> EntryPointer;
|
||||
{
|
||||
std::unique_lock lk {CodeObjectCacheService->GetEntryMapMutex()};
|
||||
|
||||
auto& EntryMap = CodeObjectCacheService->GetEntryMap();
|
||||
auto it = EntryMap.find(Base);
|
||||
if (it != EntryMap.end()) {
|
||||
// Lock the job ref counter since we are erasing it
|
||||
// Once this passes it will have been loaded
|
||||
it->second->NamedJobRefCountMutex.lock();
|
||||
|
||||
// Take the pointer from the map
|
||||
EntryPointer = std::move(it->second);
|
||||
|
||||
// We can now unmap the file data
|
||||
FEXCore::Allocator::munmap(EntryPointer->CodeData, EntryPointer->FileSize);
|
||||
|
||||
// Remove this from the entry map
|
||||
EntryMap.erase(it);
|
||||
|
||||
// Remove this entry from the unrelocated map as well
|
||||
{
|
||||
std::unique_lock lk2 {CodeObjectCacheService->GetUnrelocatedEntryMapMutex()};
|
||||
CodeObjectCacheService->GetUnrelocatedEntryMap().erase(EntryPointer->EntryHeader.OriginalBase);
|
||||
}
|
||||
} else {
|
||||
// Tried to remove something that wasn't in our code object tracking
|
||||
return;
|
||||
}
|
||||
|
||||
// Create the async work queue job now so it can finalize what it needs to do
|
||||
NamedRegionHandler->AsyncRemoveNamedRegionWorkItem(Base, Size, std::move(EntryPointer));
|
||||
|
||||
// Tell the async thread that it has work to do
|
||||
CodeObjectCacheService->NotifyWork();
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void AsyncJobHandler::AsyncAddSerializationJob(fextl::unique_ptr<SerializationJobData> Data) {
|
||||
// XXX: Actually add serialization job
|
||||
}
|
||||
} // namespace FEXCore::CodeSerialize
|
||||
@@ -1,71 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
NamedRegionObjectHandler::NamedRegionObjectHandler(FEXCore::Context::ContextImpl* ctx) {
|
||||
DefaultSerializationConfig.Cookie = CODE_COOKIE;
|
||||
|
||||
// Initialize the Arch from CPUID
|
||||
uint32_t Arch = ctx->CPUID.RunFunction(0x4000'0001, 0).eax & 0xF;
|
||||
DefaultSerializationConfig.Arch = Arch;
|
||||
|
||||
DefaultSerializationConfig.MaxInstPerBlock = ctx->Config.MaxInstPerBlock;
|
||||
DefaultSerializationConfig.MultiBlock = ctx->Config.Multiblock;
|
||||
DefaultSerializationConfig.TSOEnabled = ctx->Config.TSOEnabled;
|
||||
DefaultSerializationConfig.ABILocalFlags = ctx->Config.ABILocalFlags;
|
||||
DefaultSerializationConfig.ParanoidTSO = ctx->Config.ParanoidTSO;
|
||||
DefaultSerializationConfig.Is64BitMode = ctx->Config.Is64BitMode;
|
||||
DefaultSerializationConfig.SMCChecks = ctx->Config.SMCChecks;
|
||||
DefaultSerializationConfig.x87ReducedPrecision = ctx->Config.x87ReducedPrecision;
|
||||
}
|
||||
|
||||
void NamedRegionObjectHandler::AddNamedRegionObject(CodeRegionMapType::iterator Entry, const fextl::string& base_filename,
|
||||
const fextl::string& filename, bool Executable) {
|
||||
// XXX: Add named region objects
|
||||
|
||||
// XXX: Until entry loading is complete just claim it is loaded
|
||||
Entry->second->NamedJobRefCountMutex.unlock();
|
||||
}
|
||||
|
||||
void NamedRegionObjectHandler::RemoveNamedRegionObject(uintptr_t Base, uintptr_t Size, fextl::unique_ptr<CodeRegionEntry> Entry) {
|
||||
// XXX: Remove named region objects
|
||||
|
||||
// XXX: Until entry loading is complete just claim it is loaded
|
||||
Entry->NamedJobRefCountMutex.unlock();
|
||||
}
|
||||
|
||||
void NamedRegionObjectHandler::HandleNamedRegionObjectJobs() {
|
||||
// Walk through all of our jobs sequentially until the work queue is empty
|
||||
while (NamedWorkQueueJobs.load()) {
|
||||
fextl::unique_ptr<AsyncJobHandler::NamedRegionWorkItem> WorkItem;
|
||||
|
||||
{
|
||||
// Lock the work queue mutex for a short moment and grab an item from the list
|
||||
std::unique_lock lk {NamedWorkQueueMutex};
|
||||
size_t WorkItems = WorkQueue.size();
|
||||
if (WorkItems != 0) {
|
||||
WorkItem = std::move(WorkQueue.front());
|
||||
WorkQueue.pop();
|
||||
}
|
||||
|
||||
// Atomically update the number of jobs
|
||||
--NamedWorkQueueJobs;
|
||||
}
|
||||
|
||||
if (WorkItem) {
|
||||
if (WorkItem->GetType() == AsyncJobHandler::NamedRegionJobType::JOB_ADD_NAMED_REGION) {
|
||||
auto WorkAdd = static_cast<AsyncJobHandler::WorkItemAddNamedRegion*>(WorkItem.get());
|
||||
AddNamedRegionObject(WorkAdd->Entry, WorkAdd->BaseFilename, WorkAdd->Filename, WorkAdd->Executable);
|
||||
}
|
||||
|
||||
if (WorkItem->GetType() == AsyncJobHandler::NamedRegionJobType::JOB_REMOVE_NAMED_REGION) {
|
||||
auto WorkRemove = static_cast<AsyncJobHandler::WorkItemRemoveNamedRegion*>(WorkItem.get());
|
||||
RemoveNamedRegionObject(WorkRemove->Base, WorkRemove->Size, std::move(WorkRemove->Entry));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
} // namespace FEXCore::CodeSerialize
|
||||
@@ -1,85 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
|
||||
namespace {
|
||||
static void* ThreadHandler(void* Arg) {
|
||||
FEXCore::CodeSerialize::CodeObjectSerializeService* This = reinterpret_cast<FEXCore::CodeSerialize::CodeObjectSerializeService*>(Arg);
|
||||
This->ExecutionThread();
|
||||
return nullptr;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
CodeObjectSerializeService::CodeObjectSerializeService(FEXCore::Context::ContextImpl* ctx)
|
||||
: CTX {ctx}
|
||||
, AsyncHandler {&NamedRegionHandler, this}
|
||||
, NamedRegionHandler {ctx} {
|
||||
Initialize();
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::Shutdown() {
|
||||
if (CTX->Config.CacheObjectCodeCompilation() == FEXCore::Config::ConfigObjectCodeHandler::CONFIG_NONE) {
|
||||
return;
|
||||
}
|
||||
|
||||
WorkerThreadShuttingDown = true;
|
||||
|
||||
// Kick the working thread
|
||||
WorkAvailable.NotifyAll();
|
||||
|
||||
if (WorkerThread->joinable()) {
|
||||
// Wait for worker thread to close down
|
||||
WorkerThread->join(nullptr);
|
||||
}
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::Initialize() {
|
||||
// Add a canary so we don't crash on empty map iterator handling
|
||||
auto it = AddressToEntryMap.insert_or_assign(~0ULL, fextl::make_unique<CodeRegionEntry>());
|
||||
UnrelocatedAddressToEntryMap.insert_or_assign(~0ULL, it.first->second.get());
|
||||
|
||||
uint64_t OldMask = FEXCore::Threads::SetSignalMask(~0ULL);
|
||||
WorkerThread = FEXCore::Threads::Thread::Create(ThreadHandler, this);
|
||||
FEXCore::Threads::SetSignalMask(OldMask);
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::DoCodeRegionClosure(uint64_t Base, CodeRegionEntry* it) {
|
||||
if (Base == ~0ULL) {
|
||||
// Don't do closure on canary
|
||||
return;
|
||||
}
|
||||
// XXX: Do code region closure
|
||||
}
|
||||
|
||||
const CodeObjectFileSection* CodeObjectSerializeService::FetchCodeObjectFromCache(uint64_t GuestRIP) {
|
||||
// XXX: Actually fetch code objects from cache
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
void CodeObjectSerializeService::ExecutionThread() {
|
||||
// Set our thread name so we can see its relation
|
||||
FEXCore::Threads::SetThreadName("ObjectCodeSeri\0");
|
||||
while (WorkerThreadShuttingDown.load() != true) {
|
||||
// Wait for work
|
||||
WorkAvailable.Wait();
|
||||
|
||||
// Handle named region async jobs first. Highest priority
|
||||
NamedRegionHandler.HandleNamedRegionObjectJobs();
|
||||
|
||||
// XXX: Handle code serialization jobs second.
|
||||
}
|
||||
|
||||
// Do final code region closures on thread shutdown
|
||||
for (auto& it : AddressToEntryMap) {
|
||||
DoCodeRegionClosure(it.first, it.second.get());
|
||||
}
|
||||
|
||||
// Safely clear our maps now
|
||||
AddressToEntryMap.clear();
|
||||
UnrelocatedAddressToEntryMap.clear();
|
||||
}
|
||||
} // namespace FEXCore::CodeSerialize
|
||||
@@ -1,457 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ObjectCache/Relocations.h"
|
||||
#include "Interface/Core/ObjectCache/CodeObjectSerializationConfig.h"
|
||||
#include "Interface/IR/AOTIR.h"
|
||||
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/queue.h>
|
||||
#include <FEXCore/fextl/robin_map.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <shared_mutex>
|
||||
|
||||
namespace FEXCore::CodeSerialize {
|
||||
// XXX: Does this need to be signal safe?
|
||||
using CodeSerializationMutex = std::shared_mutex;
|
||||
struct CodeSerializationData {};
|
||||
|
||||
struct CodeObjectFileSection {
|
||||
bool Serialized;
|
||||
bool Invalid;
|
||||
const CodeSerializationData* Data;
|
||||
const char* HostCode;
|
||||
uint64_t NumRelocations;
|
||||
const char* Relocations;
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief This is the file header that lives at the start of an object cache file
|
||||
*
|
||||
* This header is updated from multiple processes!
|
||||
* Care must be taken to use OS locks when updating the file backing including this header
|
||||
*/
|
||||
struct CodeObjectSerializationHeader {
|
||||
// The configuration that this file has
|
||||
CodeObjectSerializationConfig Config;
|
||||
// The original RIP that this object section was mapped at
|
||||
uint64_t OriginalBase {};
|
||||
// The original offset in to the file that this object section was loaded from
|
||||
uint64_t OriginalOffset {};
|
||||
// Total amount of code that should be in this file
|
||||
uint64_t TotalCodeSize {};
|
||||
// Used to reserve the TSL map
|
||||
uint64_t NumCodeEntries {};
|
||||
// The number of relocations that point to this section
|
||||
uint64_t NumRelocationsTo {};
|
||||
// Total relocations in this file
|
||||
uint64_t TotalRelocationsCount {};
|
||||
};
|
||||
|
||||
struct CodeRegionEntry {
|
||||
/**
|
||||
* @name Threaded initialization objects for the initial object creation
|
||||
* @{ */
|
||||
// Base address in memory where the code region is at
|
||||
uint64_t Base {};
|
||||
|
||||
// Size of this code entry
|
||||
uint64_t Size {};
|
||||
|
||||
// The offset inside the file that is mapped to Base
|
||||
uint64_t Offset {};
|
||||
|
||||
// Filename of the object
|
||||
fextl::string Filename {};
|
||||
|
||||
CodeObjectSerializationHeader EntryHeader {};
|
||||
/** @} */
|
||||
|
||||
// The filename of the object cache for this entry
|
||||
fextl::string ObjectEntrySourceFilename {};
|
||||
|
||||
// In the case of file corruption that we can detect, we can disable serialization early for an entry
|
||||
// We should be resiliant to corruption but things happen
|
||||
bool StillSerializing {true};
|
||||
|
||||
// Long lived FD for serialization if we have multiple jobs to serialize
|
||||
// Bursts of code entries are common and this reduces file lock overhead
|
||||
//
|
||||
// Especially useful over network mounts where file locks are very slow
|
||||
int CurrentSerializedFD {-1};
|
||||
|
||||
/**
|
||||
* @name Objects required to sync objects between threads
|
||||
* @{ */
|
||||
// Refcount for the number of outstanding code entries waiting to be written for this object section
|
||||
CodeSerializationMutex ObjectJobRefCountMutex;
|
||||
|
||||
// Refcount for outstanding named object region entry loading itself
|
||||
// Will block JIT code cache look up when this has a unique_lock held
|
||||
CodeSerializationMutex NamedJobRefCountMutex;
|
||||
/** @} */
|
||||
|
||||
/**
|
||||
* @name Object Entry data management
|
||||
* @{ */
|
||||
|
||||
/**
|
||||
* @name This is the raw file data that we loaded from the code region entry file
|
||||
* @{ */
|
||||
char* CodeData {};
|
||||
size_t FileSize {};
|
||||
|
||||
fextl::vector<CodeObjectFileSection> FileCodeSections;
|
||||
/** @} */
|
||||
|
||||
// This per section map takes the most time to load and needs to be quick
|
||||
// This is the map of all code segments for this entry
|
||||
fextl::robin_map<uint64_t, CodeObjectFileSection*> SectionLookupMap {};
|
||||
/** @} */
|
||||
|
||||
// Default initialization
|
||||
CodeRegionEntry() = default;
|
||||
|
||||
// Initializer specifically for threaded loading
|
||||
CodeRegionEntry(uint64_t Base, uint64_t Size, uint64_t Offset, const fextl::string& Filename, const CodeObjectSerializationHeader& DefaultHeader)
|
||||
: Base {Base}
|
||||
, Size {Size}
|
||||
, Offset {Offset}
|
||||
, Filename {Filename}
|
||||
, EntryHeader {DefaultHeader} {}
|
||||
};
|
||||
|
||||
// Map type must use an interator that isn't invalidation on erase/insert
|
||||
using CodeRegionMapType = fextl::map<uint64_t, fextl::unique_ptr<CodeRegionEntry>>;
|
||||
using CodeRegionPtrMapType = fextl::map<uint64_t, CodeRegionEntry*>;
|
||||
|
||||
class NamedRegionObjectHandler;
|
||||
class CodeObjectSerializeService;
|
||||
|
||||
class AsyncJobHandler final {
|
||||
public:
|
||||
/**
|
||||
* @brief Structure containing all the data required to async serialize code objects
|
||||
*/
|
||||
struct SerializationJobData {
|
||||
uint64_t GuestRIP; ///< The RIP for the guest
|
||||
// XXX: Support multiblock
|
||||
uint64_t GuestCodeLength; ///< The Guest's code length
|
||||
uint64_t GuestCodeHash; ///< Hash of the guest code
|
||||
|
||||
void* HostCodeBegin; ///< Host JIT code starting memory address
|
||||
size_t HostCodeLength; ///< Host JIT code length
|
||||
uint64_t HostCodeHash; ///< Host JIT code hash before any backpatching
|
||||
|
||||
// This is the thread specific ref counter for outstanding jobs.
|
||||
// This shared mutex is incremented when the job is added, then decremented when the job is complete.
|
||||
// If a thread is shutting down or clearing code cache then the thread will pull a unique lock on this mutex.
|
||||
// This way it will wait until the async job handler is complete with it.
|
||||
CodeSerializationMutex* ThreadJobRefCount;
|
||||
|
||||
// These are the reolocations for this serialization job
|
||||
// Relatively small number of entries most of the time
|
||||
fextl::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
|
||||
/**
|
||||
* @name Objects filled in from the Code Object Serialization service when a job is added
|
||||
* @{ */
|
||||
// This is the code region's ref counter for outstanding jobs.
|
||||
// This shared mutex is incremented when the job is added, then decremented when the job is complete.
|
||||
// If a named region is being removed then a unique lock will be pulled to wait for all jobs to complete and no new jobs to be added.
|
||||
CodeSerializationMutex* ObjectJobRefCountMutexPtr;
|
||||
|
||||
// This is the code region iterator to reduce the number of map lookups
|
||||
// This will remain valid while jobs are outstanding for this region
|
||||
CodeRegionMapType::iterator CodeRegionIterator;
|
||||
/** @} */
|
||||
};
|
||||
|
||||
AsyncJobHandler(NamedRegionObjectHandler* NamedRegionHandler, CodeObjectSerializeService* CodeObjectCacheService)
|
||||
: NamedRegionHandler {NamedRegionHandler}
|
||||
, CodeObjectCacheService {CodeObjectCacheService} {}
|
||||
|
||||
protected:
|
||||
friend class CodeObjectSerializeService;
|
||||
friend class NamedRegionObjectHandler;
|
||||
/**
|
||||
* @name Async job submission functions
|
||||
* @{ */
|
||||
void AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string& filename);
|
||||
void AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size);
|
||||
void AsyncAddSerializationJob(fextl::unique_ptr<SerializationJobData> Data);
|
||||
/** @} */
|
||||
|
||||
/**
|
||||
* @name Async named region handling
|
||||
* @{ */
|
||||
/**
|
||||
* @brief The async named region jobs to handle.
|
||||
*
|
||||
* Only two, Code serialization goes in to a different queue.
|
||||
*/
|
||||
enum class NamedRegionJobType {
|
||||
JOB_ADD_NAMED_REGION,
|
||||
JOB_REMOVE_NAMED_REGION,
|
||||
};
|
||||
|
||||
class NamedRegionWorkItem {
|
||||
public:
|
||||
NamedRegionJobType GetType() const {
|
||||
return Type;
|
||||
}
|
||||
|
||||
protected:
|
||||
friend class WorkItemAddNamedRegion;
|
||||
NamedRegionWorkItem(NamedRegionJobType type)
|
||||
: Type {type} {}
|
||||
|
||||
private:
|
||||
NamedRegionJobType Type;
|
||||
};
|
||||
|
||||
class WorkItemAddNamedRegion : public NamedRegionWorkItem {
|
||||
public:
|
||||
WorkItemAddNamedRegion(const fextl::string& base, const fextl::string& filename, bool executable, CodeRegionMapType::iterator entry)
|
||||
: NamedRegionWorkItem {NamedRegionJobType::JOB_ADD_NAMED_REGION}
|
||||
, BaseFilename {base}
|
||||
, Filename {filename}
|
||||
, Executable {executable}
|
||||
, Entry {entry} {}
|
||||
const fextl::string BaseFilename;
|
||||
const fextl::string Filename;
|
||||
bool Executable;
|
||||
CodeRegionMapType::iterator Entry;
|
||||
};
|
||||
|
||||
class WorkItemRemoveNamedRegion : public NamedRegionWorkItem {
|
||||
public:
|
||||
WorkItemRemoveNamedRegion(uint64_t base, uint64_t size, fextl::unique_ptr<CodeRegionEntry> entry)
|
||||
: NamedRegionWorkItem {NamedRegionJobType::JOB_REMOVE_NAMED_REGION}
|
||||
, Base {base}
|
||||
, Size {size}
|
||||
, Entry {std::move(entry)} {}
|
||||
|
||||
uint64_t Base;
|
||||
uint64_t Size;
|
||||
fextl::unique_ptr<CodeRegionEntry> Entry;
|
||||
};
|
||||
/** @} */
|
||||
|
||||
private:
|
||||
NamedRegionObjectHandler* NamedRegionHandler;
|
||||
CodeObjectSerializeService* CodeObjectCacheService;
|
||||
};
|
||||
|
||||
class NamedRegionObjectHandler final {
|
||||
public:
|
||||
NamedRegionObjectHandler(FEXCore::Context::ContextImpl* ctx);
|
||||
|
||||
void HandleNamedRegionObjectJobs();
|
||||
|
||||
const CodeObjectSerializationConfig& GetDefaultSerializationConfig() const {
|
||||
return DefaultSerializationConfig;
|
||||
}
|
||||
|
||||
protected:
|
||||
friend class AsyncJobHandler;
|
||||
|
||||
// Return a default code header based off the default serialization config
|
||||
CodeObjectSerializationHeader DefaultCodeHeader(uint64_t Base, uint64_t Offset) const {
|
||||
return CodeObjectSerializationHeader {
|
||||
.Config = DefaultSerializationConfig,
|
||||
.OriginalBase = Base,
|
||||
.OriginalOffset = Offset,
|
||||
.NumCodeEntries = 0,
|
||||
.NumRelocationsTo = 0,
|
||||
.TotalRelocationsCount = 0,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Adds an asynchronous add named region work item to the object queue
|
||||
*
|
||||
* This adds the job that will do the loading of file resources and data tracking.
|
||||
*/
|
||||
void AsyncAddNamedRegionWorkItem(const fextl::string& base, const fextl::string& filename, bool executable, CodeRegionMapType::iterator entry) {
|
||||
std::unique_lock lk {NamedWorkQueueMutex};
|
||||
WorkQueue.emplace(fextl::make_unique<AsyncJobHandler::WorkItemAddNamedRegion>(base, filename, executable, entry));
|
||||
++NamedWorkQueueJobs;
|
||||
}
|
||||
|
||||
void AsyncRemoveNamedRegionWorkItem(uint64_t Base, uint64_t Size, fextl::unique_ptr<CodeRegionEntry> Entry) {
|
||||
std::unique_lock lk {NamedWorkQueueMutex};
|
||||
WorkQueue.emplace(fextl::make_unique<AsyncJobHandler::WorkItemRemoveNamedRegion>(Base, Size, std::move(Entry)));
|
||||
++NamedWorkQueueJobs;
|
||||
}
|
||||
|
||||
private:
|
||||
// Code version. If the code emission changes then this needs to increment
|
||||
constexpr static uint32_t CODE_VERSION = 0x0;
|
||||
|
||||
// Default cookie header for the file header
|
||||
constexpr static uint64_t CODE_COOKIE = FEXCore::IR::COOKIE_VERSION("FEXC", CODE_VERSION);
|
||||
|
||||
// Code serialization config for our current process configuration
|
||||
CodeObjectSerializationConfig DefaultSerializationConfig;
|
||||
|
||||
// Atomic counter for number of jobs in the queue without needing to pull the mutex to check
|
||||
std::atomic<uint64_t> NamedWorkQueueJobs {};
|
||||
|
||||
// Mutex for ading new jobs to the work queue
|
||||
std::mutex NamedWorkQueueMutex {};
|
||||
|
||||
// The job queue itself
|
||||
// Jobs get consumed as a FIFO
|
||||
// Jobs always get appended to the end
|
||||
fextl::queue<fextl::unique_ptr<AsyncJobHandler::NamedRegionWorkItem>> WorkQueue {};
|
||||
|
||||
/**
|
||||
* @name Named Region object handling
|
||||
* @{ */
|
||||
void AddNamedRegionObject(CodeRegionMapType::iterator Entry, const fextl::string& base_filename, const fextl::string& filename, bool Executable);
|
||||
void RemoveNamedRegionObject(uintptr_t Base, uintptr_t Size, fextl::unique_ptr<CodeRegionEntry> Entry);
|
||||
/** @} */
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Context specific code object serialization class
|
||||
*
|
||||
* Contains everything required for FEXCore to serialize code objects
|
||||
*/
|
||||
class CodeObjectSerializeService final {
|
||||
public:
|
||||
CodeObjectSerializeService(FEXCore::Context::ContextImpl* ctx);
|
||||
|
||||
/**
|
||||
* @brief Initialize the internal interface
|
||||
*
|
||||
* Is a public interface to allow the service to reinitialize after forking
|
||||
*/
|
||||
void Initialize();
|
||||
|
||||
/**
|
||||
* @brief Safely shut down the Code Object serialization service.
|
||||
*
|
||||
* This service needs to be resiliant to application crashes, but shutting down safely is still preferred.
|
||||
*/
|
||||
void Shutdown();
|
||||
|
||||
/**
|
||||
* @name Async interface
|
||||
* @{ */
|
||||
/**
|
||||
* @brief Loads a named region in to the code serialization service. As async as possible.
|
||||
*
|
||||
* @param Base - Virtual address that this named region is loaded
|
||||
* @param Size - The size of the region
|
||||
* @param Offset - The offset from the file
|
||||
* @param filename - The filename itself
|
||||
*/
|
||||
void AsyncAddNamedRegionJob(uintptr_t Base, uintptr_t Size, uintptr_t Offset, const fextl::string& filename) {
|
||||
AsyncHandler.AsyncAddNamedRegionJob(Base, Size, Offset, filename);
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Unloads a named region from the code serialization service. As async as possible.
|
||||
*
|
||||
* @param Base - Virtual address of the named region
|
||||
* @param Size - The size of the region
|
||||
*/
|
||||
void AsyncRemoveNamedRegionJob(uintptr_t Base, uintptr_t Size) {
|
||||
AsyncHandler.AsyncRemoveNamedRegionJob(Base, Size);
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Adds a code object serialization job. As async as possible.
|
||||
* Code hashing happens prior to async job serialization to catch invalidations due to backpatching.
|
||||
*
|
||||
* @param Data - A fully filled out struct containing all the code serialization
|
||||
*/
|
||||
void AsyncAddSerializationJob(fextl::unique_ptr<AsyncJobHandler::SerializationJobData> Data) {
|
||||
AsyncHandler.AsyncAddSerializationJob(std::move(Data));
|
||||
}
|
||||
/** @} */
|
||||
|
||||
/**
|
||||
* @name Synchronous interface
|
||||
* @{ */
|
||||
/**
|
||||
* @brief Synchronously waits for this thread's job queue to become empty.
|
||||
*
|
||||
* This is necessary for when a thread is shutting down
|
||||
*
|
||||
* @param ThreadJobRefCount - The shared mutex to wait on until to be empty
|
||||
*/
|
||||
static void WaitForEmptyJobQueue(CodeSerializationMutex* ThreadJobRefCount) {
|
||||
// Once the shared mutex is empty this unique lock will be gained
|
||||
std::unique_lock lk {*ThreadJobRefCount};
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Fetches object code from the Code Object Cache for JIT.
|
||||
*
|
||||
* @param GuestRIP - Which GuestRIP to search the cache for
|
||||
*
|
||||
* @return Data required for the JIT to relocate the Object code.
|
||||
*/
|
||||
const CodeObjectFileSection* FetchCodeObjectFromCache(uint64_t GuestRIP);
|
||||
/** @} */
|
||||
|
||||
// Public for threading
|
||||
void ExecutionThread();
|
||||
|
||||
protected:
|
||||
friend class AsyncJobHandler;
|
||||
|
||||
/**
|
||||
* @brief Safely closes out code object regions from the map
|
||||
*
|
||||
* @param it - iterator to do a closure on
|
||||
*/
|
||||
void DoCodeRegionClosure(uint64_t Base, CodeRegionEntry* it);
|
||||
|
||||
CodeSerializationMutex& GetEntryMapMutex() {
|
||||
return EntryMapMutex;
|
||||
}
|
||||
CodeSerializationMutex& GetUnrelocatedEntryMapMutex() {
|
||||
return EntryMapMutex;
|
||||
}
|
||||
|
||||
CodeRegionMapType& GetEntryMap() {
|
||||
return AddressToEntryMap;
|
||||
}
|
||||
CodeRegionPtrMapType& GetUnrelocatedEntryMap() {
|
||||
return UnrelocatedAddressToEntryMap;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Notify the async thread that it has work to do
|
||||
*/
|
||||
void NotifyWork() {
|
||||
WorkAvailable.NotifyOne();
|
||||
}
|
||||
|
||||
private:
|
||||
FEXCore::Context::ContextImpl* CTX;
|
||||
|
||||
Event WorkAvailable {};
|
||||
fextl::unique_ptr<FEXCore::Threads::Thread> WorkerThread;
|
||||
std::atomic_bool WorkerThreadShuttingDown {false};
|
||||
AsyncJobHandler AsyncHandler;
|
||||
NamedRegionObjectHandler NamedRegionHandler;
|
||||
|
||||
// Mutex to hold when modifying the entry maps
|
||||
CodeSerializationMutex EntryMapMutex;
|
||||
CodeSerializationMutex UnrelocatedEntryMapMutex;
|
||||
|
||||
// Entry maps
|
||||
CodeRegionMapType AddressToEntryMap;
|
||||
CodeRegionPtrMapType UnrelocatedAddressToEntryMap;
|
||||
};
|
||||
} // namespace FEXCore::CodeSerialize
|
||||
File diff suppressed because it is too large.
Load diff
@@ -27,8 +27,6 @@
|
||||
#include <xxhash.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class Pass;
|
||||
class PassManager;
|
||||
|
||||
enum class MemoryAccessType {
|
||||
// Choose TSO or Non-TSO depending on access type
|
||||
@@ -87,9 +85,6 @@ struct DispatchTableEntry {
|
||||
};
|
||||
|
||||
class OpDispatchBuilder final : public IREmitter {
|
||||
friend class FEXCore::IR::Pass;
|
||||
friend class FEXCore::IR::PassManager;
|
||||
|
||||
public:
|
||||
Ref GetNewJumpBlock(uint64_t RIP) {
|
||||
auto it = JumpTargets.find(RIP);
|
||||
@@ -201,8 +196,7 @@ public:
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
// If we don't have a jump target to a new block then we have to leave
|
||||
// Set the RIP to the next instruction and leave
|
||||
auto RelocatedNextRIP = _EntrypointOffset(GPRSize, NextRIP - Entry);
|
||||
ExitFunction(RelocatedNextRIP);
|
||||
ExitFunction(_InlineEntrypointOffset(GPRSize, NextRIP - Entry));
|
||||
} else if (it != JumpTargets.end()) {
|
||||
Jump(it->second.BlockEntry);
|
||||
return true;
|
||||
@@ -215,10 +209,17 @@ public:
|
||||
}
|
||||
|
||||
static bool CanHaveSideEffects(const FEXCore::X86Tables::X86InstInfo* TableInfo, FEXCore::X86Tables::DecodedOp Op) {
|
||||
if (TableInfo && TableInfo->Flags & X86Tables::InstFlags::FLAGS_DEBUG_MEM_ACCESS) {
|
||||
// If it is marked as having memory access then always say it has a side-effect.
|
||||
// Not always true but better to be safe.
|
||||
return true;
|
||||
if (TableInfo) {
|
||||
if (TableInfo->Flags & X86Tables::InstFlags::FLAGS_DEBUG_MEM_ACCESS) {
|
||||
// If it is marked as having memory access then always say it has a side-effect.
|
||||
// Not always true but better to be safe.
|
||||
return true;
|
||||
}
|
||||
|
||||
if (TableInfo->Flags & (X86Tables::InstFlags::FLAGS_SETS_RIP | X86Tables::InstFlags::FLAGS_BLOCK_END)) {
|
||||
// Cooperative suspend interrupts can be triggered at any back-edge, the RIP must be reconstructed correctly in such cases
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
auto CanHaveSideEffects = false;
|
||||
@@ -266,7 +267,6 @@ public:
|
||||
}
|
||||
|
||||
OpDispatchBuilder(FEXCore::Context::ContextImpl* ctx);
|
||||
OpDispatchBuilder(FEXCore::Utils::IntrusivePooledAllocator& Allocator);
|
||||
|
||||
void ResetWorkingList();
|
||||
void ResetDecodeFailure() {
|
||||
@@ -300,7 +300,8 @@ public:
|
||||
return ShouldDump;
|
||||
}
|
||||
|
||||
void BeginFunction(uint64_t RIP, const fextl::vector<FEXCore::Frontend::Decoder::DecodedBlocks>* Blocks, uint32_t NumInstructions, bool Is64BitMode);
|
||||
void BeginFunction(uint64_t RIP, const fextl::vector<FEXCore::Frontend::Decoder::DecodedBlocks>* Blocks, uint32_t NumInstructions,
|
||||
bool Is64BitMode, bool MonoBackpatcherBlock);
|
||||
void Finalize();
|
||||
|
||||
// Dispatch builder functions
|
||||
@@ -317,6 +318,7 @@ public:
|
||||
|
||||
void UnhandledOp(OpcodeArgs);
|
||||
void MOVGPROp(OpcodeArgs, uint32_t SrcIndex);
|
||||
void MOVGPRImmediate(OpcodeArgs);
|
||||
void MOVGPRNTOp(OpcodeArgs);
|
||||
void MOVVectorAlignedOp(OpcodeArgs);
|
||||
void MOVVectorUnalignedOp(OpcodeArgs);
|
||||
@@ -350,6 +352,9 @@ public:
|
||||
void LoopOp(OpcodeArgs);
|
||||
void JUMPOp(OpcodeArgs);
|
||||
void JUMPAbsoluteOp(OpcodeArgs);
|
||||
void JUMPFARIndirectOp(OpcodeArgs);
|
||||
void CALLFARIndirectOp(OpcodeArgs);
|
||||
void RETFARIndirectOp(OpcodeArgs);
|
||||
void TESTOp(OpcodeArgs, uint32_t SrcIndex);
|
||||
void MOVSXDOp(OpcodeArgs);
|
||||
void MOVSXOp(OpcodeArgs);
|
||||
@@ -735,6 +740,7 @@ public:
|
||||
void FADD(OpcodeArgs, IR::OpSize Width, bool Integer, OpResult ResInST0);
|
||||
void FDIV(OpcodeArgs, IR::OpSize Width, bool Integer, bool Reverse, OpResult ResInST0);
|
||||
void FMUL(OpcodeArgs, IR::OpSize Width, bool Integer, OpResult ResInST0);
|
||||
void FNCLEX(OpcodeArgs);
|
||||
void FNINIT(OpcodeArgs);
|
||||
void FSUB(OpcodeArgs, IR::OpSize Width, bool Integer, bool Reverse, OpResult ResInST0);
|
||||
void FTST(OpcodeArgs);
|
||||
@@ -849,8 +855,6 @@ public:
|
||||
|
||||
void PSADBW(OpcodeArgs);
|
||||
|
||||
Ref BitwiseAtLeastTwo(Ref A, Ref B, Ref C);
|
||||
|
||||
void SHA1NEXTEOp(OpcodeArgs);
|
||||
void SHA1MSG1Op(OpcodeArgs);
|
||||
void SHA1MSG2Op(OpcodeArgs);
|
||||
@@ -940,7 +944,6 @@ public:
|
||||
RefVSIB AVX128_LoadVSIB(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags, bool NeedsHigh);
|
||||
void AVX128_StoreResult_WithOpSize(FEXCore::X86Tables::DecodedOp Op, const FEXCore::X86Tables::DecodedOperand& Operand, const RefPair Src,
|
||||
MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
|
||||
void InstallAVX128Handlers();
|
||||
void AVX128_VMOVScalarImpl(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void AVX128_VectorALU(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
|
||||
void AVX128_VectorUnary(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
|
||||
@@ -969,39 +972,26 @@ public:
|
||||
void AVX128_VMOVDDUP(OpcodeArgs);
|
||||
void AVX128_VMOVSLDUP(OpcodeArgs);
|
||||
void AVX128_VMOVSHDUP(OpcodeArgs);
|
||||
template<IR::OpSize ElementSize>
|
||||
void AVX128_VBROADCAST(OpcodeArgs);
|
||||
template<IR::OpSize ElementSize>
|
||||
void AVX128_VPUNPCKL(OpcodeArgs);
|
||||
template<IR::OpSize ElementSize>
|
||||
void AVX128_VPUNPCKH(OpcodeArgs);
|
||||
void AVX128_VBROADCAST(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void AVX128_VPUNPCKL(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void AVX128_VPUNPCKH(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void AVX128_MOVVectorUnaligned(OpcodeArgs);
|
||||
template<IR::OpSize DstElementSize>
|
||||
void AVX128_InsertCVTGPR_To_FPR(OpcodeArgs);
|
||||
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
|
||||
void AVX128_CVTFPR_To_GPR(OpcodeArgs);
|
||||
void AVX128_InsertCVTGPR_To_FPR(OpcodeArgs, IR::OpSize DstElementSize);
|
||||
void AVX128_CVTFPR_To_GPR(OpcodeArgs, IR::OpSize SrcElementSize, bool HostRoundingMode);
|
||||
void AVX128_VANDN(OpcodeArgs);
|
||||
template<IR::OpSize ElementSize>
|
||||
void AVX128_VPACKSS(OpcodeArgs);
|
||||
template<IR::OpSize ElementSize>
|
||||
void AVX128_VPACKUS(OpcodeArgs);
|
||||
void AVX128_VPACKSS(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void AVX128_VPACKUS(OpcodeArgs, IR::OpSize ElementSize);
|
||||
Ref AVX128_PSIGNImpl(IR::OpSize ElementSize, Ref Src1, Ref Src2);
|
||||
template<IR::OpSize ElementSize>
|
||||
void AVX128_VPSIGN(OpcodeArgs);
|
||||
template<IR::OpSize ElementSize>
|
||||
void AVX128_UCOMISx(OpcodeArgs);
|
||||
void AVX128_VPSIGN(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void AVX128_UCOMISx(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void AVX128_VectorScalarInsertALU(OpcodeArgs, FEXCore::IR::IROps IROp, IR::OpSize ElementSize);
|
||||
Ref AVX128_VFCMPImpl(IR::OpSize ElementSize, Ref Src1, Ref Src2, uint8_t CompType);
|
||||
template<IR::OpSize ElementSize>
|
||||
void AVX128_VFCMP(OpcodeArgs);
|
||||
template<IR::OpSize ElementSize>
|
||||
void AVX128_InsertScalarFCMP(OpcodeArgs);
|
||||
void AVX128_VFCMP(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void AVX128_InsertScalarFCMP(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void AVX128_MOVBetweenGPR_FPR(OpcodeArgs);
|
||||
template<IR::OpSize ElementSize>
|
||||
void AVX128_PExtr(OpcodeArgs);
|
||||
void AVX128_PExtr(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void AVX128_ExtendVectorElements(OpcodeArgs, IR::OpSize ElementSize, IR::OpSize DstElementSize, bool Signed);
|
||||
template<IR::OpSize ElementSize>
|
||||
void AVX128_MOVMSK(OpcodeArgs);
|
||||
void AVX128_MOVMSK(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void AVX128_MOVMSKB(OpcodeArgs);
|
||||
void AVX128_PINSRImpl(OpcodeArgs, IR::OpSize ElementSize, const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op, const X86Tables::DecodedOperand& Imm);
|
||||
@@ -1015,33 +1005,25 @@ public:
|
||||
void AVX128_VINSERTPS(OpcodeArgs);
|
||||
|
||||
Ref AVX128_PHSUBImpl(Ref Src1, Ref Src2, size_t ElementSize);
|
||||
template<IR::OpSize ElementSize>
|
||||
void AVX128_VPHSUB(OpcodeArgs);
|
||||
void AVX128_VPHSUB(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
void AVX128_VPHSUBSW(OpcodeArgs);
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void AVX128_VADDSUBP(OpcodeArgs);
|
||||
void AVX128_VADDSUBP(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
template<IR::OpSize ElementSize, bool Signed>
|
||||
void AVX128_VPMULL(OpcodeArgs);
|
||||
void AVX128_VPMULL(OpcodeArgs, IR::OpSize ElementSize, bool Signed);
|
||||
|
||||
void AVX128_VPMULHRSW(OpcodeArgs);
|
||||
|
||||
template<bool Signed>
|
||||
void AVX128_VPMULHW(OpcodeArgs);
|
||||
void AVX128_VPMULHW(OpcodeArgs, bool Signed);
|
||||
|
||||
template<IR::OpSize DstElementSize, IR::OpSize SrcElementSize>
|
||||
void AVX128_InsertScalar_CVT_Float_To_Float(OpcodeArgs);
|
||||
void AVX128_InsertScalar_CVT_Float_To_Float(OpcodeArgs, IR::OpSize DstElementSize, IR::OpSize SrcElementSize);
|
||||
|
||||
template<IR::OpSize DstElementSize, IR::OpSize SrcElementSize>
|
||||
void AVX128_Vector_CVT_Float_To_Float(OpcodeArgs);
|
||||
void AVX128_Vector_CVT_Float_To_Float(OpcodeArgs, IR::OpSize DstElementSize, IR::OpSize SrcElementSize);
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
|
||||
void AVX128_Vector_CVT_Float_To_Int(OpcodeArgs);
|
||||
void AVX128_Vector_CVT_Float_To_Int(OpcodeArgs, IR::OpSize SrcElementSize, bool HostRoundingMode);
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool Widen>
|
||||
void AVX128_Vector_CVT_Int_To_Float(OpcodeArgs);
|
||||
void AVX128_Vector_CVT_Int_To_Float(OpcodeArgs, IR::OpSize SrcElementSize, bool Widen);
|
||||
|
||||
void AVX128_VEXTRACT128(OpcodeArgs);
|
||||
void AVX128_VAESImc(OpcodeArgs);
|
||||
@@ -1058,36 +1040,28 @@ public:
|
||||
|
||||
void AVX128_PHMINPOSUW(OpcodeArgs);
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void AVX128_VectorRound(OpcodeArgs);
|
||||
template<IR::OpSize ElementSize>
|
||||
void AVX128_InsertScalarRound(OpcodeArgs);
|
||||
void AVX128_VectorRound(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void AVX128_InsertScalarRound(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void AVX128_VDPP(OpcodeArgs);
|
||||
void AVX128_VDPP(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void AVX128_VPERMQ(OpcodeArgs);
|
||||
|
||||
void AVX128_VPSHUFW(OpcodeArgs, bool Low);
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void AVX128_VSHUF(OpcodeArgs);
|
||||
void AVX128_VSHUF(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void AVX128_VPERMILImm(OpcodeArgs);
|
||||
void AVX128_VPERMILImm(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
template<IROps IROp, IR::OpSize ElementSize>
|
||||
void AVX128_VHADDP(OpcodeArgs);
|
||||
void AVX128_VHADDP(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
|
||||
|
||||
void AVX128_VPHADDSW(OpcodeArgs);
|
||||
|
||||
void AVX128_VPMADDUBSW(OpcodeArgs);
|
||||
void AVX128_VPMADDWD(OpcodeArgs);
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void AVX128_VBLEND(OpcodeArgs);
|
||||
void AVX128_VBLEND(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void AVX128_VHSUBP(OpcodeArgs);
|
||||
void AVX128_VHSUBP(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
void AVX128_VPSHUFB(OpcodeArgs);
|
||||
void AVX128_VPSADBW(OpcodeArgs);
|
||||
@@ -1098,28 +1072,23 @@ public:
|
||||
void AVX128_VMASKMOVImpl(OpcodeArgs, IR::OpSize ElementSize, IR::OpSize DstSize, bool IsStore, const X86Tables::DecodedOperand& MaskOp,
|
||||
const X86Tables::DecodedOperand& DataOp);
|
||||
|
||||
template<bool IsStore>
|
||||
void AVX128_VPMASKMOV(OpcodeArgs);
|
||||
void AVX128_VPMASKMOV(OpcodeArgs, bool IsStore);
|
||||
|
||||
template<IR::OpSize ElementSize, bool IsStore>
|
||||
void AVX128_VMASKMOV(OpcodeArgs);
|
||||
void AVX128_VMASKMOV(OpcodeArgs, IR::OpSize ElementSize, bool IsStore);
|
||||
|
||||
void AVX128_MASKMOV(OpcodeArgs);
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void AVX128_VectorVariableBlend(OpcodeArgs);
|
||||
void AVX128_VectorVariableBlend(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
void AVX128_SaveAVXState(Ref MemBase);
|
||||
void AVX128_RestoreAVXState(Ref MemBase);
|
||||
void AVX128_DefaultAVXState();
|
||||
|
||||
void AVX128_VPERM2(OpcodeArgs);
|
||||
template<IR::OpSize ElementSize>
|
||||
void AVX128_VTESTP(OpcodeArgs);
|
||||
void AVX128_VTESTP(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void AVX128_PTest(OpcodeArgs);
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void AVX128_VPERMILReg(OpcodeArgs);
|
||||
void AVX128_VPERMILReg(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
void AVX128_VPERMD(OpcodeArgs);
|
||||
|
||||
@@ -1132,8 +1101,7 @@ public:
|
||||
RefPair AVX128_VPGatherQPSImpl(Ref Dest, Ref Mask, RefVSIB VSIB);
|
||||
RefPair AVX128_VPGatherImpl(OpSize Size, OpSize ElementLoadSize, OpSize AddrElementSize, RefPair Dest, RefPair Mask, RefVSIB VSIB);
|
||||
|
||||
template<OpSize AddrElementSize>
|
||||
void AVX128_VPGATHER(OpcodeArgs);
|
||||
void AVX128_VPGATHER(OpcodeArgs, OpSize AddrElementSize);
|
||||
|
||||
void AVX128_VCVTPH2PS(OpcodeArgs);
|
||||
void AVX128_VCVTPS2PH(OpcodeArgs);
|
||||
@@ -1167,6 +1135,10 @@ public:
|
||||
StoreXMMRegister(XMM, Value);
|
||||
}
|
||||
|
||||
void AVXVectorALUOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
|
||||
void AVXVectorUnaryOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
|
||||
void AVXVectorVariableBlend(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
// End of AVX 256-bit implementation
|
||||
|
||||
void InvalidOp(OpcodeArgs);
|
||||
@@ -1189,13 +1161,42 @@ public:
|
||||
}
|
||||
}
|
||||
|
||||
void FlushRegisterCache(bool SRAOnly = false) {
|
||||
void StoreContextHelper(IR::OpSize Size, RegisterClassType Class, Ref Value, uint32_t Offset) {
|
||||
// For i128Bit, we won't see a normal Constant to inline, but as a special
|
||||
// case we can replace with a 2x64-bit store which can use inline zeroes.
|
||||
if (Size == OpSize::i128Bit) {
|
||||
auto Header = GetOpHeader(WrapNode(Value));
|
||||
const auto MAX_STP_OFFSET = (252 * 4);
|
||||
|
||||
if (Offset <= MAX_STP_OFFSET && Header->Op == OP_LOADNAMEDVECTORCONSTANT) {
|
||||
auto Const = Header->C<IR::IROp_LoadNamedVectorConstant>();
|
||||
|
||||
if (Const->Constant == IR::NamedVectorConstant::NAMED_VECTOR_ZERO) {
|
||||
Ref Zero = _Constant(0);
|
||||
Ref STP = _StoreContextPair(IR::OpSize::i64Bit, GPRClass, Zero, Zero, Offset);
|
||||
|
||||
// XXX: This works around InlineConstant not having an associated
|
||||
// register class, else we'd just do InlineConstant above.
|
||||
Ref InlineZero = _InlineConstant(0);
|
||||
ReplaceNodeArgument(STP, 0, InlineZero);
|
||||
ReplaceNodeArgument(STP, 1, InlineZero);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
_StoreContext(Size, Class, Value, Offset);
|
||||
}
|
||||
|
||||
void FlushRegisterCache(bool SRAOnly = false, bool MMXOnly = false) {
|
||||
// At block boundaries, fix up the carry flag.
|
||||
if (!SRAOnly) {
|
||||
RectifyCarryInvert(CFInvertedABI);
|
||||
}
|
||||
|
||||
CalculateDeferredFlags();
|
||||
if (!MMXOnly) {
|
||||
CalculateDeferredFlags();
|
||||
}
|
||||
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
const auto VectorSize = GetGuestVectorLength();
|
||||
@@ -1217,6 +1218,11 @@ public:
|
||||
Bits &= Mask;
|
||||
}
|
||||
|
||||
if (MMXOnly) {
|
||||
Mask &= ((1ull << (MM7Index - MM0Index + 1)) - 1) << MM0Index;
|
||||
Bits &= Mask;
|
||||
}
|
||||
|
||||
while (Bits != 0) {
|
||||
uint32_t Index = 63 - std::countl_zero(Bits);
|
||||
Ref Value = RegCache.Value[Index];
|
||||
@@ -1254,7 +1260,7 @@ public:
|
||||
_StoreContextPair(Size, Class, ValueNext, Value, Offset - SizeInt);
|
||||
Bits &= ~NextBit;
|
||||
} else {
|
||||
_StoreContext(Size, Class, Value, Offset);
|
||||
StoreContextHelper(Size, Class, Value, Offset);
|
||||
// If Partial and MMX register, then we need to store all 1s in bits 64-80
|
||||
if (Partial && Index >= MM0Index && Index <= MM7Index) {
|
||||
_StoreContext(OpSize::i16Bit, IR::GPRClass, Constant(0xFFFF), Offset + 8);
|
||||
@@ -1376,11 +1382,6 @@ private:
|
||||
|
||||
Ref ADDSUBPOpImpl(OpSize Size, IR::OpSize ElementSize, Ref Src1, Ref Src2);
|
||||
|
||||
void AVXVectorALUOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
|
||||
void AVXVectorUnaryOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
|
||||
|
||||
void AVXVectorVariableBlend(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
void AVXVariableShiftImpl(OpcodeArgs, IROps IROp);
|
||||
|
||||
Ref AESKeyGenAssistImpl(OpcodeArgs);
|
||||
@@ -1521,7 +1522,23 @@ private:
|
||||
void StoreGPRRegister(uint32_t GPR, const Ref Src, IR::OpSize Size = OpSize::iInvalid, uint8_t Offset = 0);
|
||||
void StoreXMMRegister(uint32_t XMM, const Ref Src);
|
||||
|
||||
Ref GetRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset = 0);
|
||||
Ref _GetRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset, bool Inline) {
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
const auto Offs = Op->PC + Op->InstSize + Offset - Entry;
|
||||
return Inline ? _InlineEntrypointOffset(GPRSize, Offs) : _EntrypointOffset(GPRSize, Offs);
|
||||
}
|
||||
|
||||
Ref GetRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset = 0) {
|
||||
return _GetRelocatedPC(Op, Offset, false);
|
||||
}
|
||||
|
||||
void ExitRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset = 0) {
|
||||
ExitFunction(_GetRelocatedPC(Op, Offset, true /* Inline */));
|
||||
}
|
||||
|
||||
void ExitRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset, BranchHint Hint, Ref CallReturnAddress, Ref CallReturnBlock) {
|
||||
ExitFunction(_GetRelocatedPC(Op, Offset, true /* Inline */), Hint, CallReturnAddress, CallReturnBlock);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static bool IsOperandMem(const X86Tables::DecodedOperand& Operand, bool Load) {
|
||||
@@ -1650,7 +1667,7 @@ private:
|
||||
// This is currently worse for 8/16-bit, but that should be optimized. TODO
|
||||
if (SrcSize >= OpSize::i32Bit) {
|
||||
if (SetPF) {
|
||||
CalculatePF(_SubWithFlags(SrcSize, Res, Constant(0)));
|
||||
CalculatePF(SubWithFlags(SrcSize, Res, (uint64_t)0));
|
||||
} else {
|
||||
_SubNZCV(SrcSize, Res, Constant(0));
|
||||
}
|
||||
@@ -1837,10 +1854,10 @@ private:
|
||||
static const int PFIndex = 16;
|
||||
static const int AFIndex = 17;
|
||||
/* Gap 18..19 */
|
||||
/* Note this range is only valid if MMXState = MMXState_MMX */
|
||||
static const int MM0Index = 20;
|
||||
static const int MM7Index = 27;
|
||||
static const int AbridgedFTWIndex = 28;
|
||||
/* Gap 29..30 */
|
||||
/* Gap 28..30 */
|
||||
static const int DFIndex = 31;
|
||||
static const int FPR0Index = 32;
|
||||
static const int FPR15Index = 47;
|
||||
@@ -1852,7 +1869,6 @@ private:
|
||||
switch (Index) {
|
||||
case MM0Index ... MM7Index: return offsetof(FEXCore::Core::CPUState, mm[Index - MM0Index]);
|
||||
case AVXHigh0Index ... AVXHigh15Index: return offsetof(FEXCore::Core::CPUState, avx_high[Index - AVXHigh0Index][0]);
|
||||
case AbridgedFTWIndex: return offsetof(FEXCore::Core::CPUState, AbridgedFTW);
|
||||
default: return ~0U;
|
||||
}
|
||||
}
|
||||
@@ -1914,7 +1930,7 @@ private:
|
||||
if (!(RegCache.Cached & Bit)) {
|
||||
if (Index == DFIndex) {
|
||||
RegCache.Value[Index] = _LoadDF();
|
||||
} else if ((Index >= MM0Index && Index <= AbridgedFTWIndex) || Index >= AVXHigh0Index) {
|
||||
} else if ((Index >= MM0Index && Index <= MM7Index) || Index >= AVXHigh0Index) {
|
||||
RegCache.Value[Index] = _LoadContext(Size, RegClass, Offset);
|
||||
|
||||
// We may have done a partial load, this requires special handling.
|
||||
@@ -2036,7 +2052,7 @@ private:
|
||||
} else {
|
||||
// Because we explicitly inverted for CF above, we use the unsafe
|
||||
// _NZCVSelect rather than the safe CF-aware version.
|
||||
return _NZCVSelect(OpSize::i32Bit, CondForNZCVBit(BitOffset, Invert), Constant(1), Constant(0));
|
||||
return _NZCVSelect01(CondForNZCVBit(BitOffset, Invert));
|
||||
}
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) {
|
||||
return LoadGPR(Core::CPUState::PF_AS_GREG);
|
||||
@@ -2133,7 +2149,7 @@ private:
|
||||
void ConvertNZCVToX87() {
|
||||
LOGMAN_THROW_A_FMT(NZCVDirty && CachedNZCV, "NZCV must be saved");
|
||||
|
||||
Ref V = _NZCVSelect(OpSize::i32Bit, CondForNZCVBit(FEXCore::X86State::RFLAG_OF_RAW_LOC, false), Constant(1), Constant(0));
|
||||
Ref V = _NZCVSelect01(CondForNZCVBit(FEXCore::X86State::RFLAG_OF_RAW_LOC, false));
|
||||
|
||||
if (CTX->HostFeatures.SupportsFlagM2) {
|
||||
// Convert to x86 flags, saves us from or'ing after.
|
||||
@@ -2141,8 +2157,8 @@ private:
|
||||
}
|
||||
|
||||
// CF is inverted after FCMP
|
||||
Ref C = _NZCVSelect(OpSize::i32Bit, CondForNZCVBit(FEXCore::X86State::RFLAG_CF_RAW_LOC, true), Constant(1), Constant(0));
|
||||
Ref Z = _NZCVSelect(OpSize::i32Bit, CondForNZCVBit(FEXCore::X86State::RFLAG_ZF_RAW_LOC, false), Constant(1), Constant(0));
|
||||
Ref C = _NZCVSelect01(CondForNZCVBit(FEXCore::X86State::RFLAG_CF_RAW_LOC, true));
|
||||
Ref Z = _NZCVSelect01(CondForNZCVBit(FEXCore::X86State::RFLAG_ZF_RAW_LOC, false));
|
||||
|
||||
if (!CTX->HostFeatures.SupportsFlagM2) {
|
||||
C = _Or(OpSize::i32Bit, C, V);
|
||||
@@ -2238,7 +2254,7 @@ private:
|
||||
|
||||
std::optional<CondClassType> DecodeNZCVCondition(uint8_t OP);
|
||||
Ref SelectBit(Ref Cmp, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue);
|
||||
Ref SelectCC(uint8_t OP, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue);
|
||||
Ref SelectCC0All1(uint8_t OP);
|
||||
|
||||
/**
|
||||
* @brief Flushes NZCV. Mostly vestigial.
|
||||
@@ -2313,7 +2329,6 @@ private:
|
||||
* @name These functions are used by the deferred flag handling while it is calculating and storing flags in to RFLAGs.
|
||||
* @{ */
|
||||
Ref LoadPFRaw(bool Mask, bool Invert);
|
||||
Ref SelectPF(bool Invert, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue);
|
||||
Ref LoadAF();
|
||||
void FixupAF();
|
||||
void SetAFAndFixup(Ref AF);
|
||||
@@ -2349,17 +2364,19 @@ private:
|
||||
void ChgStateX87_MMX() override {
|
||||
LOGMAN_THROW_A_FMT(MMXState == MMXState_X87, "Expected state to be x87");
|
||||
_StackForceSlow();
|
||||
SetX87Top(Constant(0)); // top reset to zero
|
||||
StoreContext(AbridgedFTWIndex, Constant(0xFFFFUL)); // all valid
|
||||
SetX87Top(Constant(0)); // top reset to zero
|
||||
_StoreContext(OpSize::i8Bit, GPRClass, Constant(0xFFFFUL), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
MMXState = MMXState_MMX;
|
||||
}
|
||||
|
||||
void ChgStateMMX_X87() override {
|
||||
LOGMAN_THROW_A_FMT(MMXState == MMXState_MMX, "Expected state to be MMX");
|
||||
// The opcode dispatcher register cache is used for MMX, but the x87 pass register cache is used for x87, spill to
|
||||
// context to ensure coherence.
|
||||
FlushRegisterCache(false, true);
|
||||
// We explicitly initialize to x87 state in StartNewBlock.
|
||||
// So if we ever change this to do something else, we need to
|
||||
// make sure that we consider if we need to explicitly set it there.
|
||||
FlushRegisterCache();
|
||||
MMXState = MMXState_X87;
|
||||
}
|
||||
|
||||
@@ -2377,6 +2394,11 @@ private:
|
||||
bool Multiblock {};
|
||||
bool Is64BitMode {};
|
||||
uint64_t Entry {};
|
||||
|
||||
// Set if mono hacks are enabled and the current block is the mono callsite backpatcher, in which case the
|
||||
// XCHG ops that would patch code are replaced with a hook that performs the write and manually invalidates
|
||||
// the target address.
|
||||
bool IsMonoBackpatcherBlock {false};
|
||||
IROp_IRHeader* CurrentHeader {};
|
||||
|
||||
[[nodiscard]]
|
||||
@@ -2546,7 +2568,7 @@ private:
|
||||
}
|
||||
|
||||
ArithRef Presub(uint64_t K) {
|
||||
return IsConstant ? ArithRef(E, K - C) : ArithRef(E, E->_Sub(OpSize::i64Bit, E->Constant(K), R));
|
||||
return IsConstant ? ArithRef(E, K - C) : ArithRef(E, E->Sub(OpSize::i64Bit, E->Constant(K), R));
|
||||
}
|
||||
|
||||
ArithRef Lshl(uint64_t Shift) {
|
||||
@@ -2638,8 +2660,6 @@ private:
|
||||
return ArithRef(this, K);
|
||||
}
|
||||
|
||||
void InstallHostSpecificOpcodeHandlers();
|
||||
|
||||
///< Segment telemetry tracking
|
||||
uint32_t SegmentsNeedReadCheck {~0U};
|
||||
void CheckLegacySegmentWrite(Ref NewNode, uint32_t SegmentReg);
|
||||
@@ -2653,16 +2673,14 @@ constexpr inline void InstallToTable(auto& FinalTable, const auto& LocalTable) {
|
||||
for (uint8_t i = 0; i < Op.Count; ++i) {
|
||||
auto& TableOp = FinalTable[OpNum + i];
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
if (TableOp.OpcodeDispatcher) {
|
||||
if (TableOp.OpcodeDispatcher.OpDispatch) {
|
||||
ERROR_AND_DIE_FMT("Duplicate Entry {}", TableOp.Name);
|
||||
}
|
||||
#endif
|
||||
|
||||
TableOp.OpcodeDispatcher = Dispatcher;
|
||||
TableOp.OpcodeDispatcher.OpDispatch = Dispatcher;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void InstallOpcodeHandlers(Context::OperatingMode Mode);
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -21,466 +21,6 @@ class OrderedNode;
|
||||
|
||||
#define OpcodeArgs [[maybe_unused]] FEXCore::X86Tables::DecodedOp Op
|
||||
|
||||
void OpDispatchBuilder::InstallAVX128Handlers() {
|
||||
#define OPD(map_select, pp, opcode) (((map_select - 1) << 10) | (pp << 8) | (opcode))
|
||||
static constexpr DispatchTableEntry AVX128Table[] = {
|
||||
{OPD(1, 0b00, 0x10), 1, &OpDispatchBuilder::AVX128_VMOVAPS},
|
||||
{OPD(1, 0b01, 0x10), 1, &OpDispatchBuilder::AVX128_VMOVAPS},
|
||||
{OPD(1, 0b10, 0x10), 1, &OpDispatchBuilder::AVX128_VMOVSS},
|
||||
{OPD(1, 0b11, 0x10), 1, &OpDispatchBuilder::AVX128_VMOVSD},
|
||||
{OPD(1, 0b00, 0x11), 1, &OpDispatchBuilder::AVX128_VMOVAPS},
|
||||
{OPD(1, 0b01, 0x11), 1, &OpDispatchBuilder::AVX128_VMOVAPS},
|
||||
{OPD(1, 0b10, 0x11), 1, &OpDispatchBuilder::AVX128_VMOVSS},
|
||||
{OPD(1, 0b11, 0x11), 1, &OpDispatchBuilder::AVX128_VMOVSD},
|
||||
|
||||
{OPD(1, 0b00, 0x12), 1, &OpDispatchBuilder::AVX128_VMOVLP},
|
||||
{OPD(1, 0b01, 0x12), 1, &OpDispatchBuilder::AVX128_VMOVLP},
|
||||
{OPD(1, 0b10, 0x12), 1, &OpDispatchBuilder::AVX128_VMOVSLDUP},
|
||||
{OPD(1, 0b11, 0x12), 1, &OpDispatchBuilder::AVX128_VMOVDDUP},
|
||||
{OPD(1, 0b00, 0x13), 1, &OpDispatchBuilder::AVX128_VMOVLP},
|
||||
{OPD(1, 0b01, 0x13), 1, &OpDispatchBuilder::AVX128_VMOVLP},
|
||||
|
||||
{OPD(1, 0b00, 0x14), 1, &OpDispatchBuilder::AVX128_VPUNPCKL<OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x14), 1, &OpDispatchBuilder::AVX128_VPUNPCKL<OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x15), 1, &OpDispatchBuilder::AVX128_VPUNPCKH<OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x15), 1, &OpDispatchBuilder::AVX128_VPUNPCKH<OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x16), 1, &OpDispatchBuilder::AVX128_VMOVHP},
|
||||
{OPD(1, 0b01, 0x16), 1, &OpDispatchBuilder::AVX128_VMOVHP},
|
||||
{OPD(1, 0b10, 0x16), 1, &OpDispatchBuilder::AVX128_VMOVSHDUP},
|
||||
{OPD(1, 0b00, 0x17), 1, &OpDispatchBuilder::AVX128_VMOVHP},
|
||||
{OPD(1, 0b01, 0x17), 1, &OpDispatchBuilder::AVX128_VMOVHP},
|
||||
|
||||
{OPD(1, 0b00, 0x28), 1, &OpDispatchBuilder::AVX128_VMOVAPS},
|
||||
{OPD(1, 0b01, 0x28), 1, &OpDispatchBuilder::AVX128_VMOVAPS},
|
||||
{OPD(1, 0b00, 0x29), 1, &OpDispatchBuilder::AVX128_VMOVAPS},
|
||||
{OPD(1, 0b01, 0x29), 1, &OpDispatchBuilder::AVX128_VMOVAPS},
|
||||
|
||||
{OPD(1, 0b10, 0x2A), 1, &OpDispatchBuilder::AVX128_InsertCVTGPR_To_FPR<OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x2A), 1, &OpDispatchBuilder::AVX128_InsertCVTGPR_To_FPR<OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x2B), 1, &OpDispatchBuilder::AVX128_MOVVectorNT},
|
||||
{OPD(1, 0b01, 0x2B), 1, &OpDispatchBuilder::AVX128_MOVVectorNT},
|
||||
|
||||
{OPD(1, 0b10, 0x2C), 1, &OpDispatchBuilder::AVX128_CVTFPR_To_GPR<OpSize::i32Bit, false>},
|
||||
{OPD(1, 0b11, 0x2C), 1, &OpDispatchBuilder::AVX128_CVTFPR_To_GPR<OpSize::i64Bit, false>},
|
||||
|
||||
{OPD(1, 0b10, 0x2D), 1, &OpDispatchBuilder::AVX128_CVTFPR_To_GPR<OpSize::i32Bit, true>},
|
||||
{OPD(1, 0b11, 0x2D), 1, &OpDispatchBuilder::AVX128_CVTFPR_To_GPR<OpSize::i64Bit, true>},
|
||||
|
||||
{OPD(1, 0b00, 0x2E), 1, &OpDispatchBuilder::AVX128_UCOMISx<OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x2E), 1, &OpDispatchBuilder::AVX128_UCOMISx<OpSize::i64Bit>},
|
||||
{OPD(1, 0b00, 0x2F), 1, &OpDispatchBuilder::AVX128_UCOMISx<OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x2F), 1, &OpDispatchBuilder::AVX128_UCOMISx<OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x50), 1, &OpDispatchBuilder::AVX128_MOVMSK<OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x50), 1, &OpDispatchBuilder::AVX128_MOVMSK<OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x51), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorUnary, IR::OP_VFSQRT, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x51), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorUnary, IR::OP_VFSQRT, OpSize::i64Bit>},
|
||||
{OPD(1, 0b10, 0x51), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorScalarInsertALU, IR::OP_VFSQRTSCALARINSERT, OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x51), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorScalarInsertALU, IR::OP_VFSQRTSCALARINSERT, OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x52), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorUnary, IR::OP_VFRSQRT, OpSize::i32Bit>},
|
||||
{OPD(1, 0b10, 0x52), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorScalarInsertALU, IR::OP_VFRSQRTSCALARINSERT, OpSize::i32Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x53), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorUnary, IR::OP_VFRECP, OpSize::i32Bit>},
|
||||
{OPD(1, 0b10, 0x53), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorScalarInsertALU, IR::OP_VFRECPSCALARINSERT, OpSize::i32Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x54), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VAND, OpSize::i128Bit>},
|
||||
{OPD(1, 0b01, 0x54), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VAND, OpSize::i128Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x55), 1, &OpDispatchBuilder::AVX128_VANDN},
|
||||
{OPD(1, 0b01, 0x55), 1, &OpDispatchBuilder::AVX128_VANDN},
|
||||
|
||||
{OPD(1, 0b00, 0x56), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VOR, OpSize::i128Bit>},
|
||||
{OPD(1, 0b01, 0x56), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VOR, OpSize::i128Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x57), 1, &OpDispatchBuilder::AVX128_VectorXOR},
|
||||
{OPD(1, 0b01, 0x57), 1, &OpDispatchBuilder::AVX128_VectorXOR},
|
||||
|
||||
{OPD(1, 0b00, 0x58), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VFADD, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x58), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VFADD, OpSize::i64Bit>},
|
||||
{OPD(1, 0b10, 0x58), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorScalarInsertALU, IR::OP_VFADDSCALARINSERT, OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x58), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorScalarInsertALU, IR::OP_VFADDSCALARINSERT, OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x59), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VFMUL, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x59), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VFMUL, OpSize::i64Bit>},
|
||||
{OPD(1, 0b10, 0x59), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorScalarInsertALU, IR::OP_VFMULSCALARINSERT, OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x59), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorScalarInsertALU, IR::OP_VFMULSCALARINSERT, OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x5A), 1, &OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Float<OpSize::i64Bit, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x5A), 1, &OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Float<OpSize::i32Bit, OpSize::i64Bit>},
|
||||
{OPD(1, 0b10, 0x5A), 1, &OpDispatchBuilder::AVX128_InsertScalar_CVT_Float_To_Float<OpSize::i64Bit, OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x5A), 1, &OpDispatchBuilder::AVX128_InsertScalar_CVT_Float_To_Float<OpSize::i32Bit, OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x5B), 1, &OpDispatchBuilder::AVX128_Vector_CVT_Int_To_Float<OpSize::i32Bit, false>},
|
||||
{OPD(1, 0b01, 0x5B), 1, &OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Int<OpSize::i32Bit, true>},
|
||||
{OPD(1, 0b10, 0x5B), 1, &OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Int<OpSize::i32Bit, false>},
|
||||
|
||||
{OPD(1, 0b00, 0x5C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VFSUB, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x5C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VFSUB, OpSize::i64Bit>},
|
||||
{OPD(1, 0b10, 0x5C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorScalarInsertALU, IR::OP_VFSUBSCALARINSERT, OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x5C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorScalarInsertALU, IR::OP_VFSUBSCALARINSERT, OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x5D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VFMIN, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x5D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VFMIN, OpSize::i64Bit>},
|
||||
{OPD(1, 0b10, 0x5D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorScalarInsertALU, IR::OP_VFMINSCALARINSERT, OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x5D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorScalarInsertALU, IR::OP_VFMINSCALARINSERT, OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x5E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VFDIV, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x5E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VFDIV, OpSize::i64Bit>},
|
||||
{OPD(1, 0b10, 0x5E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorScalarInsertALU, IR::OP_VFDIVSCALARINSERT, OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x5E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorScalarInsertALU, IR::OP_VFDIVSCALARINSERT, OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x5F), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VFMAX, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x5F), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VFMAX, OpSize::i64Bit>},
|
||||
{OPD(1, 0b10, 0x5F), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorScalarInsertALU, IR::OP_VFMAXSCALARINSERT, OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x5F), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorScalarInsertALU, IR::OP_VFMAXSCALARINSERT, OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b01, 0x60), 1, &OpDispatchBuilder::AVX128_VPUNPCKL<OpSize::i8Bit>},
|
||||
{OPD(1, 0b01, 0x61), 1, &OpDispatchBuilder::AVX128_VPUNPCKL<OpSize::i16Bit>},
|
||||
{OPD(1, 0b01, 0x62), 1, &OpDispatchBuilder::AVX128_VPUNPCKL<OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x63), 1, &OpDispatchBuilder::AVX128_VPACKSS<OpSize::i16Bit>},
|
||||
{OPD(1, 0b01, 0x64), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VCMPGT, OpSize::i8Bit>},
|
||||
{OPD(1, 0b01, 0x65), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VCMPGT, OpSize::i16Bit>},
|
||||
{OPD(1, 0b01, 0x66), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VCMPGT, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x67), 1, &OpDispatchBuilder::AVX128_VPACKUS<OpSize::i16Bit>},
|
||||
{OPD(1, 0b01, 0x68), 1, &OpDispatchBuilder::AVX128_VPUNPCKH<OpSize::i8Bit>},
|
||||
{OPD(1, 0b01, 0x69), 1, &OpDispatchBuilder::AVX128_VPUNPCKH<OpSize::i16Bit>},
|
||||
{OPD(1, 0b01, 0x6A), 1, &OpDispatchBuilder::AVX128_VPUNPCKH<OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x6B), 1, &OpDispatchBuilder::AVX128_VPACKSS<OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x6C), 1, &OpDispatchBuilder::AVX128_VPUNPCKL<OpSize::i64Bit>},
|
||||
{OPD(1, 0b01, 0x6D), 1, &OpDispatchBuilder::AVX128_VPUNPCKH<OpSize::i64Bit>},
|
||||
{OPD(1, 0b01, 0x6E), 1, &OpDispatchBuilder::AVX128_MOVBetweenGPR_FPR},
|
||||
|
||||
{OPD(1, 0b01, 0x6F), 1, &OpDispatchBuilder::AVX128_VMOVAPS},
|
||||
{OPD(1, 0b10, 0x6F), 1, &OpDispatchBuilder::AVX128_VMOVAPS},
|
||||
|
||||
{OPD(1, 0b01, 0x70), 1, &OpDispatchBuilder::AVX128_VPERMILImm<OpSize::i32Bit>},
|
||||
{OPD(1, 0b10, 0x70), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VPSHUFW, false>},
|
||||
{OPD(1, 0b11, 0x70), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VPSHUFW, true>},
|
||||
|
||||
{OPD(1, 0b01, 0x74), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VCMPEQ, OpSize::i8Bit>},
|
||||
{OPD(1, 0b01, 0x75), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VCMPEQ, OpSize::i16Bit>},
|
||||
{OPD(1, 0b01, 0x76), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VCMPEQ, OpSize::i32Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x77), 1, &OpDispatchBuilder::AVX128_VZERO},
|
||||
|
||||
{OPD(1, 0b01, 0x7C), 1, &OpDispatchBuilder::AVX128_VHADDP<IR::OP_VFADDP, OpSize::i64Bit>},
|
||||
{OPD(1, 0b11, 0x7C), 1, &OpDispatchBuilder::AVX128_VHADDP<IR::OP_VFADDP, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x7D), 1, &OpDispatchBuilder::AVX128_VHSUBP<OpSize::i64Bit>},
|
||||
{OPD(1, 0b11, 0x7D), 1, &OpDispatchBuilder::AVX128_VHSUBP<OpSize::i32Bit>},
|
||||
|
||||
{OPD(1, 0b01, 0x7E), 1, &OpDispatchBuilder::AVX128_MOVBetweenGPR_FPR},
|
||||
{OPD(1, 0b10, 0x7E), 1, &OpDispatchBuilder::AVX128_MOVQ},
|
||||
|
||||
{OPD(1, 0b01, 0x7F), 1, &OpDispatchBuilder::AVX128_VMOVAPS},
|
||||
{OPD(1, 0b10, 0x7F), 1, &OpDispatchBuilder::AVX128_VMOVAPS},
|
||||
|
||||
{OPD(1, 0b00, 0xC2), 1, &OpDispatchBuilder::AVX128_VFCMP<OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0xC2), 1, &OpDispatchBuilder::AVX128_VFCMP<OpSize::i64Bit>},
|
||||
{OPD(1, 0b10, 0xC2), 1, &OpDispatchBuilder::AVX128_InsertScalarFCMP<OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0xC2), 1, &OpDispatchBuilder::AVX128_InsertScalarFCMP<OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b01, 0xC4), 1, &OpDispatchBuilder::AVX128_VPINSRW},
|
||||
{OPD(1, 0b01, 0xC5), 1, &OpDispatchBuilder::AVX128_PExtr<OpSize::i16Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0xC6), 1, &OpDispatchBuilder::AVX128_VSHUF<OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0xC6), 1, &OpDispatchBuilder::AVX128_VSHUF<OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b01, 0xD0), 1, &OpDispatchBuilder::AVX128_VADDSUBP<OpSize::i64Bit>},
|
||||
{OPD(1, 0b11, 0xD0), 1, &OpDispatchBuilder::AVX128_VADDSUBP<OpSize::i32Bit>},
|
||||
|
||||
{OPD(1, 0b01, 0xD1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorShiftWideImpl, OpSize::i16Bit, IROps::OP_VUSHRSWIDE>}, // VPSRL
|
||||
{OPD(1, 0b01, 0xD2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorShiftWideImpl, OpSize::i32Bit, IROps::OP_VUSHRSWIDE>}, // VPSRL
|
||||
{OPD(1, 0b01, 0xD3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorShiftWideImpl, OpSize::i64Bit, IROps::OP_VUSHRSWIDE>}, // VPSRL
|
||||
{OPD(1, 0b01, 0xD4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VADD, OpSize::i64Bit>},
|
||||
{OPD(1, 0b01, 0xD5), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VMUL, OpSize::i16Bit>},
|
||||
{OPD(1, 0b01, 0xD6), 1, &OpDispatchBuilder::AVX128_MOVQ},
|
||||
{OPD(1, 0b01, 0xD7), 1, &OpDispatchBuilder::AVX128_MOVMSKB},
|
||||
|
||||
{OPD(1, 0b01, 0xD8), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VUQSUB, OpSize::i8Bit>},
|
||||
{OPD(1, 0b01, 0xD9), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VUQSUB, OpSize::i16Bit>},
|
||||
{OPD(1, 0b01, 0xDA), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VUMIN, OpSize::i8Bit>},
|
||||
{OPD(1, 0b01, 0xDB), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VAND, OpSize::i128Bit>},
|
||||
{OPD(1, 0b01, 0xDC), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VUQADD, OpSize::i8Bit>},
|
||||
{OPD(1, 0b01, 0xDD), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VUQADD, OpSize::i16Bit>},
|
||||
{OPD(1, 0b01, 0xDE), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VUMAX, OpSize::i8Bit>},
|
||||
{OPD(1, 0b01, 0xDF), 1, &OpDispatchBuilder::AVX128_VANDN},
|
||||
|
||||
{OPD(1, 0b01, 0xE0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VURAVG, OpSize::i8Bit>},
|
||||
{OPD(1, 0b01, 0xE1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorShiftWideImpl, OpSize::i16Bit, IROps::OP_VSSHRSWIDE>}, // VPSRA
|
||||
{OPD(1, 0b01, 0xE2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorShiftWideImpl, OpSize::i32Bit, IROps::OP_VSSHRSWIDE>}, // VPSRA
|
||||
{OPD(1, 0b01, 0xE3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VURAVG, OpSize::i16Bit>},
|
||||
{OPD(1, 0b01, 0xE4), 1, &OpDispatchBuilder::AVX128_VPMULHW<false>},
|
||||
{OPD(1, 0b01, 0xE5), 1, &OpDispatchBuilder::AVX128_VPMULHW<true>},
|
||||
|
||||
{OPD(1, 0b01, 0xE6), 1, &OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Int<OpSize::i64Bit, false>},
|
||||
{OPD(1, 0b10, 0xE6), 1, &OpDispatchBuilder::AVX128_Vector_CVT_Int_To_Float<OpSize::i32Bit, true>},
|
||||
{OPD(1, 0b11, 0xE6), 1, &OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Int<OpSize::i64Bit, true>},
|
||||
|
||||
{OPD(1, 0b01, 0xE7), 1, &OpDispatchBuilder::AVX128_MOVVectorNT},
|
||||
|
||||
{OPD(1, 0b01, 0xE8), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VSQSUB, OpSize::i8Bit>},
|
||||
{OPD(1, 0b01, 0xE9), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VSQSUB, OpSize::i16Bit>},
|
||||
{OPD(1, 0b01, 0xEA), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VSMIN, OpSize::i16Bit>},
|
||||
{OPD(1, 0b01, 0xEB), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VOR, OpSize::i128Bit>},
|
||||
{OPD(1, 0b01, 0xEC), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VSQADD, OpSize::i8Bit>},
|
||||
{OPD(1, 0b01, 0xED), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VSQADD, OpSize::i16Bit>},
|
||||
{OPD(1, 0b01, 0xEE), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VSMAX, OpSize::i16Bit>},
|
||||
{OPD(1, 0b01, 0xEF), 1, &OpDispatchBuilder::AVX128_VectorXOR},
|
||||
|
||||
{OPD(1, 0b11, 0xF0), 1, &OpDispatchBuilder::AVX128_MOVVectorUnaligned},
|
||||
{OPD(1, 0b01, 0xF1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorShiftWideImpl, OpSize::i16Bit, IROps::OP_VUSHLSWIDE>}, // VPSLL
|
||||
{OPD(1, 0b01, 0xF2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorShiftWideImpl, OpSize::i32Bit, IROps::OP_VUSHLSWIDE>}, // VPSLL
|
||||
{OPD(1, 0b01, 0xF3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorShiftWideImpl, OpSize::i64Bit, IROps::OP_VUSHLSWIDE>}, // VPSLL
|
||||
{OPD(1, 0b01, 0xF4), 1, &OpDispatchBuilder::AVX128_VPMULL<OpSize::i32Bit, false>},
|
||||
{OPD(1, 0b01, 0xF5), 1, &OpDispatchBuilder::AVX128_VPMADDWD},
|
||||
{OPD(1, 0b01, 0xF6), 1, &OpDispatchBuilder::AVX128_VPSADBW},
|
||||
{OPD(1, 0b01, 0xF7), 1, &OpDispatchBuilder::AVX128_MASKMOV},
|
||||
|
||||
{OPD(1, 0b01, 0xF8), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VSUB, OpSize::i8Bit>},
|
||||
{OPD(1, 0b01, 0xF9), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VSUB, OpSize::i16Bit>},
|
||||
{OPD(1, 0b01, 0xFA), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VSUB, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0xFB), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VSUB, OpSize::i64Bit>},
|
||||
{OPD(1, 0b01, 0xFC), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VADD, OpSize::i8Bit>},
|
||||
{OPD(1, 0b01, 0xFD), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VADD, OpSize::i16Bit>},
|
||||
{OPD(1, 0b01, 0xFE), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VADD, OpSize::i32Bit>},
|
||||
|
||||
{OPD(2, 0b01, 0x00), 1, &OpDispatchBuilder::AVX128_VPSHUFB},
|
||||
{OPD(2, 0b01, 0x01), 1, &OpDispatchBuilder::AVX128_VHADDP<IR::OP_VADDP, OpSize::i16Bit>},
|
||||
{OPD(2, 0b01, 0x02), 1, &OpDispatchBuilder::AVX128_VHADDP<IR::OP_VADDP, OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x03), 1, &OpDispatchBuilder::AVX128_VPHADDSW},
|
||||
{OPD(2, 0b01, 0x04), 1, &OpDispatchBuilder::AVX128_VPMADDUBSW},
|
||||
|
||||
{OPD(2, 0b01, 0x05), 1, &OpDispatchBuilder::AVX128_VPHSUB<OpSize::i16Bit>},
|
||||
{OPD(2, 0b01, 0x06), 1, &OpDispatchBuilder::AVX128_VPHSUB<OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x07), 1, &OpDispatchBuilder::AVX128_VPHSUBSW},
|
||||
|
||||
{OPD(2, 0b01, 0x08), 1, &OpDispatchBuilder::AVX128_VPSIGN<OpSize::i8Bit>},
|
||||
{OPD(2, 0b01, 0x09), 1, &OpDispatchBuilder::AVX128_VPSIGN<OpSize::i16Bit>},
|
||||
{OPD(2, 0b01, 0x0A), 1, &OpDispatchBuilder::AVX128_VPSIGN<OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x0B), 1, &OpDispatchBuilder::AVX128_VPMULHRSW},
|
||||
{OPD(2, 0b01, 0x0C), 1, &OpDispatchBuilder::AVX128_VPERMILReg<OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x0D), 1, &OpDispatchBuilder::AVX128_VPERMILReg<OpSize::i64Bit>},
|
||||
{OPD(2, 0b01, 0x0E), 1, &OpDispatchBuilder::AVX128_VTESTP<OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x0F), 1, &OpDispatchBuilder::AVX128_VTESTP<OpSize::i64Bit>},
|
||||
|
||||
|
||||
{OPD(2, 0b01, 0x13), 1, &OpDispatchBuilder::AVX128_VCVTPH2PS},
|
||||
{OPD(2, 0b01, 0x16), 1, &OpDispatchBuilder::AVX128_VPERMD},
|
||||
{OPD(2, 0b01, 0x17), 1, &OpDispatchBuilder::AVX128_PTest},
|
||||
{OPD(2, 0b01, 0x18), 1, &OpDispatchBuilder::AVX128_VBROADCAST<OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x19), 1, &OpDispatchBuilder::AVX128_VBROADCAST<OpSize::i64Bit>},
|
||||
{OPD(2, 0b01, 0x1A), 1, &OpDispatchBuilder::AVX128_VBROADCAST<OpSize::i128Bit>},
|
||||
{OPD(2, 0b01, 0x1C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorUnary, IR::OP_VABS, OpSize::i8Bit>},
|
||||
{OPD(2, 0b01, 0x1D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorUnary, IR::OP_VABS, OpSize::i16Bit>},
|
||||
{OPD(2, 0b01, 0x1E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorUnary, IR::OP_VABS, OpSize::i32Bit>},
|
||||
|
||||
{OPD(2, 0b01, 0x20), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_ExtendVectorElements, OpSize::i8Bit, OpSize::i16Bit, true>},
|
||||
{OPD(2, 0b01, 0x21), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_ExtendVectorElements, OpSize::i8Bit, OpSize::i32Bit, true>},
|
||||
{OPD(2, 0b01, 0x22), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_ExtendVectorElements, OpSize::i8Bit, OpSize::i64Bit, true>},
|
||||
{OPD(2, 0b01, 0x23), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_ExtendVectorElements, OpSize::i16Bit, OpSize::i32Bit, true>},
|
||||
{OPD(2, 0b01, 0x24), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_ExtendVectorElements, OpSize::i16Bit, OpSize::i64Bit, true>},
|
||||
{OPD(2, 0b01, 0x25), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_ExtendVectorElements, OpSize::i32Bit, OpSize::i64Bit, true>},
|
||||
|
||||
{OPD(2, 0b01, 0x28), 1, &OpDispatchBuilder::AVX128_VPMULL<OpSize::i32Bit, true>},
|
||||
{OPD(2, 0b01, 0x29), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VCMPEQ, OpSize::i64Bit>},
|
||||
{OPD(2, 0b01, 0x2A), 1, &OpDispatchBuilder::AVX128_MOVVectorNT},
|
||||
{OPD(2, 0b01, 0x2B), 1, &OpDispatchBuilder::AVX128_VPACKUS<OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x2C), 1, &OpDispatchBuilder::AVX128_VMASKMOV<OpSize::i32Bit, false>},
|
||||
{OPD(2, 0b01, 0x2D), 1, &OpDispatchBuilder::AVX128_VMASKMOV<OpSize::i64Bit, false>},
|
||||
{OPD(2, 0b01, 0x2E), 1, &OpDispatchBuilder::AVX128_VMASKMOV<OpSize::i32Bit, true>},
|
||||
{OPD(2, 0b01, 0x2F), 1, &OpDispatchBuilder::AVX128_VMASKMOV<OpSize::i64Bit, true>},
|
||||
|
||||
{OPD(2, 0b01, 0x30), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_ExtendVectorElements, OpSize::i8Bit, OpSize::i16Bit, false>},
|
||||
{OPD(2, 0b01, 0x31), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_ExtendVectorElements, OpSize::i8Bit, OpSize::i32Bit, false>},
|
||||
{OPD(2, 0b01, 0x32), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_ExtendVectorElements, OpSize::i8Bit, OpSize::i64Bit, false>},
|
||||
{OPD(2, 0b01, 0x33), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_ExtendVectorElements, OpSize::i16Bit, OpSize::i32Bit, false>},
|
||||
{OPD(2, 0b01, 0x34), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_ExtendVectorElements, OpSize::i16Bit, OpSize::i64Bit, false>},
|
||||
{OPD(2, 0b01, 0x35), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_ExtendVectorElements, OpSize::i32Bit, OpSize::i64Bit, false>},
|
||||
{OPD(2, 0b01, 0x36), 1, &OpDispatchBuilder::AVX128_VPERMD},
|
||||
|
||||
{OPD(2, 0b01, 0x37), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VCMPGT, OpSize::i64Bit>},
|
||||
{OPD(2, 0b01, 0x38), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VSMIN, OpSize::i8Bit>},
|
||||
{OPD(2, 0b01, 0x39), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VSMIN, OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x3A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VUMIN, OpSize::i16Bit>},
|
||||
{OPD(2, 0b01, 0x3B), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VUMIN, OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x3C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VSMAX, OpSize::i8Bit>},
|
||||
{OPD(2, 0b01, 0x3D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VSMAX, OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x3E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VUMAX, OpSize::i16Bit>},
|
||||
{OPD(2, 0b01, 0x3F), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VUMAX, OpSize::i32Bit>},
|
||||
|
||||
{OPD(2, 0b01, 0x40), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorALU, IR::OP_VMUL, OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x41), 1, &OpDispatchBuilder::AVX128_PHMINPOSUW},
|
||||
{OPD(2, 0b01, 0x45), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VariableShiftImpl, IROps::OP_VUSHR>}, // VPSRLV
|
||||
{OPD(2, 0b01, 0x46), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VariableShiftImpl, IROps::OP_VSSHR>}, // VPSRAVD
|
||||
{OPD(2, 0b01, 0x47), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VariableShiftImpl, IROps::OP_VUSHL>}, // VPSLLV
|
||||
|
||||
{OPD(2, 0b01, 0x58), 1, &OpDispatchBuilder::AVX128_VBROADCAST<OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x59), 1, &OpDispatchBuilder::AVX128_VBROADCAST<OpSize::i64Bit>},
|
||||
{OPD(2, 0b01, 0x5A), 1, &OpDispatchBuilder::AVX128_VBROADCAST<OpSize::i128Bit>},
|
||||
|
||||
{OPD(2, 0b01, 0x78), 1, &OpDispatchBuilder::AVX128_VBROADCAST<OpSize::i8Bit>},
|
||||
{OPD(2, 0b01, 0x79), 1, &OpDispatchBuilder::AVX128_VBROADCAST<OpSize::i16Bit>},
|
||||
|
||||
{OPD(2, 0b01, 0x8C), 1, &OpDispatchBuilder::AVX128_VPMASKMOV<false>},
|
||||
{OPD(2, 0b01, 0x8E), 1, &OpDispatchBuilder::AVX128_VPMASKMOV<true>},
|
||||
|
||||
{OPD(2, 0b01, 0x90), 1, &OpDispatchBuilder::AVX128_VPGATHER<OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x91), 1, &OpDispatchBuilder::AVX128_VPGATHER<OpSize::i64Bit>},
|
||||
{OPD(2, 0b01, 0x92), 1, &OpDispatchBuilder::AVX128_VPGATHER<OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x93), 1, &OpDispatchBuilder::AVX128_VPGATHER<OpSize::i64Bit>},
|
||||
|
||||
{OPD(2, 0b01, 0x96), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAddSubImpl, true, 1, 3, 2>}, // VFMADDSUB
|
||||
{OPD(2, 0b01, 0x97), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAddSubImpl, false, 1, 3, 2>}, // VFMSUBADD
|
||||
|
||||
{OPD(2, 0b01, 0x98), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAImpl, IR::OP_VFMLA, 1, 3, 2>}, // VFMADD
|
||||
{OPD(2, 0b01, 0x99), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAScalarImpl, IR::OP_VFMLASCALARINSERT, 1, 3, 2>}, // VFMADD
|
||||
{OPD(2, 0b01, 0x9A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAImpl, IR::OP_VFMLS, 1, 3, 2>}, // VFMSUB
|
||||
{OPD(2, 0b01, 0x9B), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAScalarImpl, IR::OP_VFMLSSCALARINSERT, 1, 3, 2>}, // VFMSUB
|
||||
{OPD(2, 0b01, 0x9C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAImpl, IR::OP_VFNMLA, 1, 3, 2>}, // VFNMADD
|
||||
{OPD(2, 0b01, 0x9D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAScalarImpl, IR::OP_VFNMLASCALARINSERT, 1, 3, 2>}, // VFNMADD
|
||||
{OPD(2, 0b01, 0x9E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAImpl, IR::OP_VFNMLS, 1, 3, 2>}, // VFNMSUB
|
||||
{OPD(2, 0b01, 0x9F), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAScalarImpl, IR::OP_VFNMLSSCALARINSERT, 1, 3, 2>}, // VFNMSUB
|
||||
|
||||
{OPD(2, 0b01, 0xA8), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAImpl, IR::OP_VFMLA, 2, 1, 3>}, // VFMADD
|
||||
{OPD(2, 0b01, 0xA9), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAScalarImpl, IR::OP_VFMLASCALARINSERT, 2, 1, 3>}, // VFMADD
|
||||
{OPD(2, 0b01, 0xAA), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAImpl, IR::OP_VFMLS, 2, 1, 3>}, // VFMSUB
|
||||
{OPD(2, 0b01, 0xAB), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAScalarImpl, IR::OP_VFMLSSCALARINSERT, 2, 1, 3>}, // VFMSUB
|
||||
{OPD(2, 0b01, 0xAC), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAImpl, IR::OP_VFNMLA, 2, 1, 3>}, // VFNMADD
|
||||
{OPD(2, 0b01, 0xAD), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAScalarImpl, IR::OP_VFNMLASCALARINSERT, 2, 1, 3>}, // VFNMADD
|
||||
{OPD(2, 0b01, 0xAE), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAImpl, IR::OP_VFNMLS, 2, 1, 3>}, // VFNMSUB
|
||||
{OPD(2, 0b01, 0xAF), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAScalarImpl, IR::OP_VFNMLSSCALARINSERT, 2, 1, 3>}, // VFNMSUB
|
||||
|
||||
{OPD(2, 0b01, 0xB8), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAImpl, IR::OP_VFMLA, 2, 3, 1>}, // VFMADD
|
||||
{OPD(2, 0b01, 0xB9), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAScalarImpl, IR::OP_VFMLASCALARINSERT, 2, 3, 1>}, // VFMADD
|
||||
{OPD(2, 0b01, 0xBA), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAImpl, IR::OP_VFMLS, 2, 3, 1>}, // VFMSUB
|
||||
{OPD(2, 0b01, 0xBB), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAScalarImpl, IR::OP_VFMLSSCALARINSERT, 2, 3, 1>}, // VFMSUB
|
||||
{OPD(2, 0b01, 0xBC), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAImpl, IR::OP_VFNMLA, 2, 3, 1>}, // VFNMADD
|
||||
{OPD(2, 0b01, 0xBD), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAScalarImpl, IR::OP_VFNMLASCALARINSERT, 2, 3, 1>}, // VFNMADD
|
||||
{OPD(2, 0b01, 0xBE), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAImpl, IR::OP_VFNMLS, 2, 3, 1>}, // VFNMSUB
|
||||
{OPD(2, 0b01, 0xBF), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAScalarImpl, IR::OP_VFNMLSSCALARINSERT, 2, 3, 1>}, // VFNMSUB
|
||||
|
||||
{OPD(2, 0b01, 0xA6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAddSubImpl, true, 2, 1, 3>}, // VFMADDSUB
|
||||
{OPD(2, 0b01, 0xA7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAddSubImpl, false, 2, 1, 3>}, // VFMSUBADD
|
||||
|
||||
{OPD(2, 0b01, 0xB6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAddSubImpl, true, 2, 3, 1>}, // VFMADDSUB
|
||||
{OPD(2, 0b01, 0xB7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VFMAddSubImpl, false, 2, 3, 1>}, // VFMSUBADD
|
||||
|
||||
{OPD(2, 0b01, 0xDB), 1, &OpDispatchBuilder::AVX128_VAESImc},
|
||||
{OPD(2, 0b01, 0xDC), 1, &OpDispatchBuilder::AVX128_VAESEnc},
|
||||
{OPD(2, 0b01, 0xDD), 1, &OpDispatchBuilder::AVX128_VAESEncLast},
|
||||
{OPD(2, 0b01, 0xDE), 1, &OpDispatchBuilder::AVX128_VAESDec},
|
||||
{OPD(2, 0b01, 0xDF), 1, &OpDispatchBuilder::AVX128_VAESDecLast},
|
||||
|
||||
{OPD(3, 0b01, 0x00), 1, &OpDispatchBuilder::AVX128_VPERMQ},
|
||||
{OPD(3, 0b01, 0x01), 1, &OpDispatchBuilder::AVX128_VPERMQ},
|
||||
{OPD(3, 0b01, 0x02), 1, &OpDispatchBuilder::AVX128_VBLEND<OpSize::i32Bit>},
|
||||
{OPD(3, 0b01, 0x04), 1, &OpDispatchBuilder::AVX128_VPERMILImm<OpSize::i32Bit>},
|
||||
{OPD(3, 0b01, 0x05), 1, &OpDispatchBuilder::AVX128_VPERMILImm<OpSize::i64Bit>},
|
||||
{OPD(3, 0b01, 0x06), 1, &OpDispatchBuilder::AVX128_VPERM2},
|
||||
{OPD(3, 0b01, 0x08), 1, &OpDispatchBuilder::AVX128_VectorRound<OpSize::i32Bit>},
|
||||
{OPD(3, 0b01, 0x09), 1, &OpDispatchBuilder::AVX128_VectorRound<OpSize::i64Bit>},
|
||||
{OPD(3, 0b01, 0x0A), 1, &OpDispatchBuilder::AVX128_InsertScalarRound<OpSize::i32Bit>},
|
||||
{OPD(3, 0b01, 0x0B), 1, &OpDispatchBuilder::AVX128_InsertScalarRound<OpSize::i64Bit>},
|
||||
{OPD(3, 0b01, 0x0C), 1, &OpDispatchBuilder::AVX128_VBLEND<OpSize::i32Bit>},
|
||||
{OPD(3, 0b01, 0x0D), 1, &OpDispatchBuilder::AVX128_VBLEND<OpSize::i64Bit>},
|
||||
{OPD(3, 0b01, 0x0E), 1, &OpDispatchBuilder::AVX128_VBLEND<OpSize::i16Bit>},
|
||||
{OPD(3, 0b01, 0x0F), 1, &OpDispatchBuilder::AVX128_VPALIGNR},
|
||||
|
||||
{OPD(3, 0b01, 0x14), 1, &OpDispatchBuilder::AVX128_PExtr<OpSize::i8Bit>},
|
||||
{OPD(3, 0b01, 0x15), 1, &OpDispatchBuilder::AVX128_PExtr<OpSize::i16Bit>},
|
||||
{OPD(3, 0b01, 0x16), 1, &OpDispatchBuilder::AVX128_PExtr<OpSize::i32Bit>},
|
||||
{OPD(3, 0b01, 0x17), 1, &OpDispatchBuilder::AVX128_PExtr<OpSize::i32Bit>},
|
||||
|
||||
{OPD(3, 0b01, 0x18), 1, &OpDispatchBuilder::AVX128_VINSERT},
|
||||
{OPD(3, 0b01, 0x19), 1, &OpDispatchBuilder::AVX128_VEXTRACT128},
|
||||
{OPD(3, 0b01, 0x1D), 1, &OpDispatchBuilder::AVX128_VCVTPS2PH},
|
||||
{OPD(3, 0b01, 0x20), 1, &OpDispatchBuilder::AVX128_VPINSRB},
|
||||
{OPD(3, 0b01, 0x21), 1, &OpDispatchBuilder::AVX128_VINSERTPS},
|
||||
{OPD(3, 0b01, 0x22), 1, &OpDispatchBuilder::AVX128_VPINSRDQ},
|
||||
|
||||
{OPD(3, 0b01, 0x38), 1, &OpDispatchBuilder::AVX128_VINSERT},
|
||||
{OPD(3, 0b01, 0x39), 1, &OpDispatchBuilder::AVX128_VEXTRACT128},
|
||||
|
||||
{OPD(3, 0b01, 0x40), 1, &OpDispatchBuilder::AVX128_VDPP<OpSize::i32Bit>},
|
||||
{OPD(3, 0b01, 0x41), 1, &OpDispatchBuilder::AVX128_VDPP<OpSize::i64Bit>},
|
||||
{OPD(3, 0b01, 0x42), 1, &OpDispatchBuilder::AVX128_VMPSADBW},
|
||||
|
||||
{OPD(3, 0b01, 0x46), 1, &OpDispatchBuilder::AVX128_VPERM2},
|
||||
|
||||
{OPD(3, 0b01, 0x4A), 1, &OpDispatchBuilder::AVX128_VectorVariableBlend<OpSize::i32Bit>},
|
||||
{OPD(3, 0b01, 0x4B), 1, &OpDispatchBuilder::AVX128_VectorVariableBlend<OpSize::i64Bit>},
|
||||
{OPD(3, 0b01, 0x4C), 1, &OpDispatchBuilder::AVX128_VectorVariableBlend<OpSize::i8Bit>},
|
||||
|
||||
{OPD(3, 0b01, 0x60), 1, &OpDispatchBuilder::AVX128_VPCMPESTRM},
|
||||
{OPD(3, 0b01, 0x61), 1, &OpDispatchBuilder::AVX128_VPCMPESTRI},
|
||||
{OPD(3, 0b01, 0x62), 1, &OpDispatchBuilder::AVX128_VPCMPISTRM},
|
||||
{OPD(3, 0b01, 0x63), 1, &OpDispatchBuilder::AVX128_VPCMPISTRI},
|
||||
|
||||
{OPD(3, 0b01, 0xDF), 1, &OpDispatchBuilder::AVX128_VAESKeyGenAssist},
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
#define OPD(group, pp, opcode) (((group - X86Tables::TYPE_VEX_GROUP_12) << 4) | (pp << 3) | (opcode))
|
||||
static constexpr DispatchTableEntry VEX128TableGroupOps[] {
|
||||
// VPSRLI
|
||||
{OPD(X86Tables::TYPE_VEX_GROUP_12, 1, 0b010), 1,
|
||||
&OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorShiftImmImpl, OpSize::i16Bit, IROps::OP_VUSHRI>},
|
||||
// VPSLLI
|
||||
{OPD(X86Tables::TYPE_VEX_GROUP_12, 1, 0b110), 1,
|
||||
&OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorShiftImmImpl, OpSize::i16Bit, IROps::OP_VSHLI>},
|
||||
// VPSRAI
|
||||
{OPD(X86Tables::TYPE_VEX_GROUP_12, 1, 0b100), 1,
|
||||
&OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorShiftImmImpl, OpSize::i16Bit, IROps::OP_VSSHRI>},
|
||||
|
||||
// VPSRLI
|
||||
{OPD(X86Tables::TYPE_VEX_GROUP_13, 1, 0b010), 1,
|
||||
&OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorShiftImmImpl, OpSize::i32Bit, IROps::OP_VUSHRI>},
|
||||
// VPSLLI
|
||||
{OPD(X86Tables::TYPE_VEX_GROUP_13, 1, 0b110), 1,
|
||||
&OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorShiftImmImpl, OpSize::i32Bit, IROps::OP_VSHLI>},
|
||||
// VPSRAI
|
||||
{OPD(X86Tables::TYPE_VEX_GROUP_13, 1, 0b100), 1,
|
||||
&OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorShiftImmImpl, OpSize::i32Bit, IROps::OP_VSSHRI>},
|
||||
|
||||
// VPSRLI
|
||||
{OPD(X86Tables::TYPE_VEX_GROUP_14, 1, 0b010), 1,
|
||||
&OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorShiftImmImpl, OpSize::i64Bit, IROps::OP_VUSHRI>},
|
||||
// VPSRLDQ
|
||||
{OPD(X86Tables::TYPE_VEX_GROUP_14, 1, 0b011), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_ShiftDoubleImm, ShiftDirection::RIGHT>},
|
||||
// VPSLLI
|
||||
{OPD(X86Tables::TYPE_VEX_GROUP_14, 1, 0b110), 1,
|
||||
&OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VectorShiftImmImpl, OpSize::i64Bit, IROps::OP_VSHLI>},
|
||||
// VPSLLDQ
|
||||
{OPD(X86Tables::TYPE_VEX_GROUP_14, 1, 0b111), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_ShiftDoubleImm, ShiftDirection::LEFT>},
|
||||
|
||||
///< Use the regular implementation. It just happens to be in the VEX table.
|
||||
{OPD(X86Tables::TYPE_VEX_GROUP_15, 0, 0b010), 1, &OpDispatchBuilder::LDMXCSR},
|
||||
{OPD(X86Tables::TYPE_VEX_GROUP_15, 0, 0b011), 1, &OpDispatchBuilder::STMXCSR},
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
#define OPD(map_select, pp, opcode) (((map_select - 1) << 10) | (pp << 8) | (opcode))
|
||||
constexpr DispatchTableEntry VEX128_PCLMUL[] = {
|
||||
{OPD(3, 0b01, 0x44), 1, &OpDispatchBuilder::AVX128_VPCLMULQDQ},
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
InstallToTable(FEXCore::X86Tables::VEXTableOps, AVX128Table);
|
||||
InstallToTable(FEXCore::X86Tables::VEXTableGroupOps, VEX128TableGroupOps);
|
||||
if (CTX->HostFeatures.SupportsPMULL_128Bit) {
|
||||
InstallToTable(FEXCore::X86Tables::VEXTableOps, VEX128_PCLMUL);
|
||||
}
|
||||
|
||||
SaveAVXStateFunc = &OpDispatchBuilder::AVX128_SaveAVXState;
|
||||
RestoreAVXStateFunc = &OpDispatchBuilder::AVX128_RestoreAVXState;
|
||||
DefaultAVXStateFunc = &OpDispatchBuilder::AVX128_DefaultAVXState;
|
||||
}
|
||||
|
||||
OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_LoadSource_WithOpSize(
|
||||
const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags, bool NeedsHigh, MemoryAccessType AccessType) {
|
||||
|
||||
@@ -501,7 +41,7 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_LoadSource_WithOpSize(
|
||||
HighA.Offset += 16;
|
||||
|
||||
if (Operand.IsSIB()) {
|
||||
[[maybe_unused]] const bool IsVSIB = (Op->Flags & X86Tables::DecodeFlags::FLAG_VSIB_BYTE) != 0;
|
||||
const bool IsVSIB = (Op->Flags & X86Tables::DecodeFlags::FLAG_VSIB_BYTE) != 0;
|
||||
LOGMAN_THROW_A_FMT(!IsVSIB, "VSIB uses LoadVSIB instead");
|
||||
}
|
||||
|
||||
@@ -515,7 +55,7 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_LoadSource_WithOpSize(
|
||||
|
||||
OpDispatchBuilder::RefVSIB
|
||||
OpDispatchBuilder::AVX128_LoadVSIB(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags, bool NeedsHigh) {
|
||||
[[maybe_unused]] const bool IsVSIB = (Op->Flags & X86Tables::DecodeFlags::FLAG_VSIB_BYTE) != 0;
|
||||
const bool IsVSIB = (Op->Flags & X86Tables::DecodeFlags::FLAG_VSIB_BYTE) != 0;
|
||||
LOGMAN_THROW_A_FMT(Operand.IsSIB() && IsVSIB, "Trying to load VSIB for something that isn't the correct type!");
|
||||
|
||||
// VSIB is a very special case which has a ton of encoded data.
|
||||
@@ -954,8 +494,7 @@ void OpDispatchBuilder::AVX128_VMOVSHDUP(OpcodeArgs) {
|
||||
[this](IR::OpSize ElementSize, Ref Src) { return _VTrn2(OpSize::i128Bit, ElementSize, Src, Src); });
|
||||
}
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void OpDispatchBuilder::AVX128_VBROADCAST(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_VBROADCAST(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto Is128Bit = DstSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
RefPair Src {};
|
||||
@@ -981,14 +520,12 @@ void OpDispatchBuilder::AVX128_VBROADCAST(OpcodeArgs) {
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Src);
|
||||
}
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void OpDispatchBuilder::AVX128_VPUNPCKL(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_VPUNPCKL(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
AVX128_VectorBinaryImpl(Op, OpSizeFromSrc(Op), ElementSize,
|
||||
[this](IR::OpSize _ElementSize, Ref Src1, Ref Src2) { return _VZip(OpSize::i128Bit, _ElementSize, Src1, Src2); });
|
||||
}
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void OpDispatchBuilder::AVX128_VPUNPCKH(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_VPUNPCKH(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
AVX128_VectorBinaryImpl(Op, OpSizeFromSrc(Op), ElementSize,
|
||||
[this](IR::OpSize _ElementSize, Ref Src1, Ref Src2) { return _VZip2(OpSize::i128Bit, _ElementSize, Src1, Src2); });
|
||||
}
|
||||
@@ -1011,8 +548,7 @@ void OpDispatchBuilder::AVX128_MOVVectorUnaligned(OpcodeArgs) {
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Src);
|
||||
}
|
||||
|
||||
template<IR::OpSize DstElementSize>
|
||||
void OpDispatchBuilder::AVX128_InsertCVTGPR_To_FPR(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_InsertCVTGPR_To_FPR(OpcodeArgs, IR::OpSize DstElementSize) {
|
||||
const auto SrcSize = OpSizeFromSrc(Op);
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
|
||||
@@ -1033,20 +569,19 @@ void OpDispatchBuilder::AVX128_InsertCVTGPR_To_FPR(OpcodeArgs) {
|
||||
} else {
|
||||
// In the case of cvtsi2s{s,d} where the source and destination are the same size,
|
||||
// then it is more optimal to load in to the FPR register directly and convert there.
|
||||
auto Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false);
|
||||
auto Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
// Always signed
|
||||
Result.Low = _VSToFVectorInsert(DstSize, DstElementSize, DstElementSize, Src1.Low, Src2.Low, false, false);
|
||||
Result.Low = _VSToFVectorInsert(DstSize, DstElementSize, DstElementSize, Src1.Low, Src2, false, false);
|
||||
}
|
||||
|
||||
[[maybe_unused]] const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "Programming Error: This should never occur!");
|
||||
Result.High = LoadZeroVector(OpSize::i128Bit);
|
||||
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
}
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
|
||||
void OpDispatchBuilder::AVX128_CVTFPR_To_GPR(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_CVTFPR_To_GPR(OpcodeArgs, IR::OpSize SrcElementSize, bool HostRoundingMode) {
|
||||
// If loading a vector, use the full size, so we don't
|
||||
// unnecessarily zero extend the vector. Otherwise, if
|
||||
// memory, then we want to load the element size exactly.
|
||||
@@ -1066,15 +601,13 @@ void OpDispatchBuilder::AVX128_VANDN(OpcodeArgs) {
|
||||
[this](IR::OpSize _ElementSize, Ref Src1, Ref Src2) { return _VAndn(OpSize::i128Bit, _ElementSize, Src2, Src1); });
|
||||
}
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void OpDispatchBuilder::AVX128_VPACKSS(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_VPACKSS(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
AVX128_VectorBinaryImpl(Op, OpSizeFromSrc(Op), ElementSize, [this](IR::OpSize _ElementSize, Ref Src1, Ref Src2) {
|
||||
return _VSQXTNPair(OpSize::i128Bit, _ElementSize, Src1, Src2);
|
||||
});
|
||||
}
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void OpDispatchBuilder::AVX128_VPACKUS(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_VPACKUS(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
AVX128_VectorBinaryImpl(Op, OpSizeFromSrc(Op), ElementSize, [this](IR::OpSize _ElementSize, Ref Src1, Ref Src2) {
|
||||
return _VSQXTUNPair(OpSize::i128Bit, _ElementSize, Src1, Src2);
|
||||
});
|
||||
@@ -1086,14 +619,12 @@ Ref OpDispatchBuilder::AVX128_PSIGNImpl(IR::OpSize ElementSize, Ref Src1, Ref Sr
|
||||
return _VMul(OpSize::i128Bit, ElementSize, Src1, Control);
|
||||
}
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void OpDispatchBuilder::AVX128_VPSIGN(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_VPSIGN(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
AVX128_VectorBinaryImpl(Op, OpSizeFromSrc(Op), ElementSize,
|
||||
[this](IR::OpSize _ElementSize, Ref Src1, Ref Src2) { return AVX128_PSIGNImpl(_ElementSize, Src1, Src2); });
|
||||
}
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void OpDispatchBuilder::AVX128_UCOMISx(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_UCOMISx(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
const auto SrcSize = Op->Src[0].IsGPR() ? GetGuestVectorLength() : ElementSize;
|
||||
|
||||
auto Src1 = AVX128_LoadSource_WithOpSize(Op, Op->Dest, Op->Flags, false);
|
||||
@@ -1131,8 +662,7 @@ void OpDispatchBuilder::AVX128_VectorScalarInsertALU(OpcodeArgs, FEXCore::IR::IR
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, RefPair {.Low = Result_Low, .High = High});
|
||||
}
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void OpDispatchBuilder::AVX128_VFCMP(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_VFCMP(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
const uint8_t CompType = Op->Src[2].Literal();
|
||||
|
||||
struct {
|
||||
@@ -1148,8 +678,7 @@ void OpDispatchBuilder::AVX128_VFCMP(OpcodeArgs) {
|
||||
});
|
||||
}
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void OpDispatchBuilder::AVX128_InsertScalarFCMP(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_InsertScalarFCMP(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
// We load the full vector width when dealing with a source vector,
|
||||
// so that we don't do any unnecessary zero extension to the scalar
|
||||
// element that we're going to operate on.
|
||||
@@ -1206,8 +735,7 @@ void OpDispatchBuilder::AVX128_MOVBetweenGPR_FPR(OpcodeArgs) {
|
||||
}
|
||||
}
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void OpDispatchBuilder::AVX128_PExtr(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_PExtr(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
|
||||
auto Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false);
|
||||
@@ -1218,7 +746,7 @@ void OpDispatchBuilder::AVX128_PExtr(OpcodeArgs) {
|
||||
// is the same except that REX.W or VEX.W is set to 1. Incredibly frustrating.
|
||||
// Use the destination size as the element size in this case.
|
||||
auto OverridenElementSize = ElementSize;
|
||||
if constexpr (ElementSize == OpSize::i32Bit) {
|
||||
if (ElementSize == OpSize::i32Bit) {
|
||||
OverridenElementSize = DstSize;
|
||||
}
|
||||
|
||||
@@ -1291,8 +819,7 @@ void OpDispatchBuilder::AVX128_ExtendVectorElements(OpcodeArgs, IR::OpSize Eleme
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
}
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void OpDispatchBuilder::AVX128_MOVMSK(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_MOVMSK(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
const auto SrcSize = OpSizeFromSrc(Op);
|
||||
const auto Is128Bit = SrcSize == OpSize::i128Bit;
|
||||
|
||||
@@ -1474,8 +1001,7 @@ void OpDispatchBuilder::AVX128_VINSERTPS(OpcodeArgs) {
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, AVX128_Zext(Result));
|
||||
}
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void OpDispatchBuilder::AVX128_VPHSUB(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_VPHSUB(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
AVX128_VectorBinaryImpl(Op, OpSizeFromDst(Op), ElementSize, [this](IR::OpSize _ElementSize, Ref Src1, Ref Src2) {
|
||||
return PHSUBOpImpl(OpSize::i128Bit, Src1, Src2, _ElementSize);
|
||||
});
|
||||
@@ -1486,19 +1012,17 @@ void OpDispatchBuilder::AVX128_VPHSUBSW(OpcodeArgs) {
|
||||
[this](IR::OpSize _ElementSize, Ref Src1, Ref Src2) { return PHSUBSOpImpl(OpSize::i128Bit, Src1, Src2); });
|
||||
}
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void OpDispatchBuilder::AVX128_VADDSUBP(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_VADDSUBP(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
AVX128_VectorBinaryImpl(Op, OpSizeFromDst(Op), ElementSize, [this](IR::OpSize _ElementSize, Ref Src1, Ref Src2) {
|
||||
return ADDSUBPOpImpl(OpSize::i128Bit, _ElementSize, Src1, Src2);
|
||||
});
|
||||
}
|
||||
|
||||
template<IR::OpSize ElementSize, bool Signed>
|
||||
void OpDispatchBuilder::AVX128_VPMULL(OpcodeArgs) {
|
||||
static_assert(ElementSize == OpSize::i32Bit, "Currently only handles 32-bit -> 64-bit");
|
||||
void OpDispatchBuilder::AVX128_VPMULL(OpcodeArgs, IR::OpSize ElementSize, bool Signed) {
|
||||
LOGMAN_THROW_A_FMT(ElementSize == OpSize::i32Bit, "Currently only handles 32-bit -> 64-bit");
|
||||
|
||||
AVX128_VectorBinaryImpl(Op, OpSizeFromDst(Op), ElementSize, [this](IR::OpSize _ElementSize, Ref Src1, Ref Src2) -> Ref {
|
||||
return PMULLOpImpl(OpSize::i128Bit, ElementSize, Signed, Src1, Src2);
|
||||
AVX128_VectorBinaryImpl(Op, OpSizeFromDst(Op), ElementSize, [&](IR::OpSize _ElementSize, Ref Src1, Ref Src2) -> Ref {
|
||||
return PMULLOpImpl(OpSize::i128Bit, _ElementSize, Signed, Src1, Src2);
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1507,9 +1031,8 @@ void OpDispatchBuilder::AVX128_VPMULHRSW(OpcodeArgs) {
|
||||
[this](IR::OpSize _ElementSize, Ref Src1, Ref Src2) -> Ref { return PMULHRSWOpImpl(OpSize::i128Bit, Src1, Src2); });
|
||||
}
|
||||
|
||||
template<bool Signed>
|
||||
void OpDispatchBuilder::AVX128_VPMULHW(OpcodeArgs) {
|
||||
AVX128_VectorBinaryImpl(Op, OpSizeFromDst(Op), OpSize::i16Bit, [this](IR::OpSize _ElementSize, Ref Src1, Ref Src2) -> Ref {
|
||||
void OpDispatchBuilder::AVX128_VPMULHW(OpcodeArgs, bool Signed) {
|
||||
AVX128_VectorBinaryImpl(Op, OpSizeFromDst(Op), OpSize::i16Bit, [&](IR::OpSize _ElementSize, Ref Src1, Ref Src2) -> Ref {
|
||||
if (Signed) {
|
||||
return _VSMulH(OpSize::i128Bit, _ElementSize, Src1, Src2);
|
||||
} else {
|
||||
@@ -1518,12 +1041,11 @@ void OpDispatchBuilder::AVX128_VPMULHW(OpcodeArgs) {
|
||||
});
|
||||
}
|
||||
|
||||
template<IR::OpSize DstElementSize, IR::OpSize SrcElementSize>
|
||||
void OpDispatchBuilder::AVX128_InsertScalar_CVT_Float_To_Float(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_InsertScalar_CVT_Float_To_Float(OpcodeArgs, IR::OpSize DstElementSize, IR::OpSize SrcElementSize) {
|
||||
// Gotta be careful with this operation.
|
||||
// It inserts in to the lowest element, retaining the remainder of the lower 128-bits.
|
||||
// Then zero extends the top 128-bit.
|
||||
const auto SrcSize = OpSizeFromSrc(Op);
|
||||
const auto SrcSize = Op->Src[1].IsGPR() ? OpSize::i128Bit : SrcElementSize;
|
||||
auto Src1 = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false);
|
||||
Ref Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
@@ -1531,8 +1053,7 @@ void OpDispatchBuilder::AVX128_InsertScalar_CVT_Float_To_Float(OpcodeArgs) {
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, AVX128_Zext(Result));
|
||||
}
|
||||
|
||||
template<IR::OpSize DstElementSize, IR::OpSize SrcElementSize>
|
||||
void OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Float(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Float(OpcodeArgs, IR::OpSize DstElementSize, IR::OpSize SrcElementSize) {
|
||||
const auto SrcSize = OpSizeFromSrc(Op);
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
|
||||
@@ -1560,11 +1081,11 @@ void OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Float(OpcodeArgs) {
|
||||
|
||||
RefPair Result {};
|
||||
|
||||
auto TransformLow = [this](Ref Src) -> Ref {
|
||||
auto TransformLow = [&](Ref Src) -> Ref {
|
||||
return _Vector_FToF(OpSize::i128Bit, DstElementSize, Src, SrcElementSize);
|
||||
};
|
||||
|
||||
auto TransformHigh = [this](Ref Src) -> Ref {
|
||||
auto TransformHigh = [&](Ref Src) -> Ref {
|
||||
return _VFCVTL2(OpSize::i128Bit, SrcElementSize, Src);
|
||||
};
|
||||
|
||||
@@ -1595,8 +1116,7 @@ void OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Float(OpcodeArgs) {
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
}
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
|
||||
void OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Int(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Int(OpcodeArgs, IR::OpSize SrcElementSize, bool HostRoundingMode) {
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
|
||||
const auto Is128BitSrc = SrcSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
@@ -1624,8 +1144,7 @@ void OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Int(OpcodeArgs) {
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
}
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool Widen>
|
||||
void OpDispatchBuilder::AVX128_Vector_CVT_Int_To_Float(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_Vector_CVT_Int_To_Float(OpcodeArgs, IR::OpSize SrcElementSize, bool Widen) {
|
||||
const auto Size = OpSizeFromDst(Op);
|
||||
const auto Is128Bit = Size == OpSize::i128Bit;
|
||||
|
||||
@@ -1641,7 +1160,7 @@ void OpDispatchBuilder::AVX128_Vector_CVT_Int_To_Float(OpcodeArgs) {
|
||||
}
|
||||
}();
|
||||
|
||||
auto Convert = [this](Ref Src, IROps Op) -> Ref {
|
||||
auto Convert = [&](Ref Src, IROps Op) -> Ref {
|
||||
auto ElementSize = SrcElementSize;
|
||||
if (Widen) {
|
||||
DeriveOp(Extended, Op, _VSXTL(OpSize::i128Bit, ElementSize, Src));
|
||||
@@ -1766,17 +1285,15 @@ void OpDispatchBuilder::AVX128_PHMINPOSUW(OpcodeArgs) {
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, AVX128_Zext(Result));
|
||||
}
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void OpDispatchBuilder::AVX128_VectorRound(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_VectorRound(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
const auto Mode = Op->Src[1].Literal();
|
||||
|
||||
AVX128_VectorUnaryImpl(Op, Size, ElementSize,
|
||||
[this, Mode](IR::OpSize, Ref Src) { return VectorRoundImpl(OpSize::i128Bit, ElementSize, Src, Mode); });
|
||||
[this, Mode](IR::OpSize ElementSize, Ref Src) { return VectorRoundImpl(OpSize::i128Bit, ElementSize, Src, Mode); });
|
||||
}
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void OpDispatchBuilder::AVX128_InsertScalarRound(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_InsertScalarRound(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
// We load the full vector width when dealing with a source vector,
|
||||
// so that we don't do any unnecessary zero extension to the scalar
|
||||
// element that we're going to operate on.
|
||||
@@ -1797,11 +1314,10 @@ void OpDispatchBuilder::AVX128_InsertScalarRound(OpcodeArgs) {
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, AVX128_Zext(Result));
|
||||
}
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void OpDispatchBuilder::AVX128_VDPP(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_VDPP(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
const uint64_t Literal = Op->Src[2].Literal();
|
||||
|
||||
AVX128_VectorBinaryImpl(Op, OpSizeFromSrc(Op), ElementSize, [this, Literal](IR::OpSize, Ref Src1, Ref Src2) {
|
||||
AVX128_VectorBinaryImpl(Op, OpSizeFromSrc(Op), ElementSize, [this, Literal](IR::OpSize ElementSize, Ref Src1, Ref Src2) {
|
||||
return DPPOpImpl(OpSize::i128Bit, Src1, Src2, Literal, ElementSize);
|
||||
});
|
||||
}
|
||||
@@ -1869,8 +1385,7 @@ void OpDispatchBuilder::AVX128_VPSHUFW(OpcodeArgs, bool Low) {
|
||||
});
|
||||
}
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void OpDispatchBuilder::AVX128_VSHUF(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_VSHUF(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto Is128Bit = SrcSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
auto Shuffle = Op->Src[2].Literal();
|
||||
@@ -1884,14 +1399,13 @@ void OpDispatchBuilder::AVX128_VSHUF(OpcodeArgs) {
|
||||
if (Is128Bit) {
|
||||
Result.High = LoadZeroVector(OpSize::i128Bit);
|
||||
} else {
|
||||
constexpr uint8_t ShiftAmount = ElementSize == OpSize::i32Bit ? 0 : 2;
|
||||
const uint8_t ShiftAmount = ElementSize == OpSize::i32Bit ? 0 : 2;
|
||||
Result.High = SHUFOpImpl(Op, OpSize::i128Bit, ElementSize, Src1.High, Src2.High, Shuffle >> ShiftAmount);
|
||||
}
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
}
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void OpDispatchBuilder::AVX128_VPERMILImm(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_VPERMILImm(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto Is128Bit = SrcSize == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
@@ -1900,7 +1414,7 @@ void OpDispatchBuilder::AVX128_VPERMILImm(OpcodeArgs) {
|
||||
|
||||
RefPair Result = AVX128_Zext(LoadZeroVector(OpSize::i128Bit));
|
||||
|
||||
if constexpr (ElementSize == OpSize::i64Bit) {
|
||||
if (ElementSize == OpSize::i64Bit) {
|
||||
auto DoSwizzle64 = [this](Ref Src, uint8_t Selector) -> Ref {
|
||||
switch (Selector) {
|
||||
case 0b00:
|
||||
@@ -1928,9 +1442,8 @@ void OpDispatchBuilder::AVX128_VPERMILImm(OpcodeArgs) {
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
}
|
||||
|
||||
template<IROps IROp, IR::OpSize ElementSize>
|
||||
void OpDispatchBuilder::AVX128_VHADDP(OpcodeArgs) {
|
||||
AVX128_VectorBinaryImpl(Op, OpSizeFromSrc(Op), ElementSize, [this](IR::OpSize, Ref Src1, Ref Src2) {
|
||||
void OpDispatchBuilder::AVX128_VHADDP(OpcodeArgs, IROps IROp, IR::OpSize ElementSize) {
|
||||
AVX128_VectorBinaryImpl(Op, OpSizeFromSrc(Op), ElementSize, [&](IR::OpSize ElementSize, Ref Src1, Ref Src2) {
|
||||
DeriveOp(Res, IROp, _VFAddP(OpSize::i128Bit, ElementSize, Src1, Src2));
|
||||
return Res;
|
||||
});
|
||||
@@ -1951,8 +1464,7 @@ void OpDispatchBuilder::AVX128_VPMADDWD(OpcodeArgs) {
|
||||
[this](IR::OpSize _ElementSize, Ref Src1, Ref Src2) { return PMADDWDOpImpl(OpSize::i128Bit, Src1, Src2); });
|
||||
}
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void OpDispatchBuilder::AVX128_VBLEND(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_VBLEND(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
const auto SrcSize = OpSizeFromSrc(Op);
|
||||
const auto Is128Bit = SrcSize == OpSize::i128Bit;
|
||||
const uint64_t Selector = Op->Src[2].Literal();
|
||||
@@ -1961,7 +1473,7 @@ void OpDispatchBuilder::AVX128_VBLEND(OpcodeArgs) {
|
||||
/// i16Bit: Reuses same bits, no shift
|
||||
/// i32Bit: Shift by 4
|
||||
/// i64Bit: Shift by 2
|
||||
constexpr uint64_t SelectorShift = ElementSize == OpSize::i64Bit ? 2 : ElementSize == OpSize::i32Bit ? 4 : 0;
|
||||
const uint64_t SelectorShift = ElementSize == OpSize::i64Bit ? 2 : ElementSize == OpSize::i32Bit ? 4 : 0;
|
||||
|
||||
auto Src1 = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, !Is128Bit);
|
||||
auto Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, !Is128Bit);
|
||||
@@ -1978,10 +1490,9 @@ void OpDispatchBuilder::AVX128_VBLEND(OpcodeArgs) {
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
}
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void OpDispatchBuilder::AVX128_VHSUBP(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_VHSUBP(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
AVX128_VectorBinaryImpl(Op, OpSizeFromDst(Op), ElementSize,
|
||||
[this](IR::OpSize, Ref Src1, Ref Src2) { return HSUBPOpImpl(OpSize::i128Bit, ElementSize, Src1, Src2); });
|
||||
[&](IR::OpSize, Ref Src1, Ref Src2) { return HSUBPOpImpl(OpSize::i128Bit, ElementSize, Src1, Src2); });
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VPSHUFB(OpcodeArgs) {
|
||||
@@ -2081,13 +1592,11 @@ void OpDispatchBuilder::AVX128_VMASKMOVImpl(OpcodeArgs, IR::OpSize ElementSize,
|
||||
}
|
||||
}
|
||||
|
||||
template<bool IsStore>
|
||||
void OpDispatchBuilder::AVX128_VPMASKMOV(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_VPMASKMOV(OpcodeArgs, bool IsStore) {
|
||||
AVX128_VMASKMOVImpl(Op, OpSizeFromSrc(Op), OpSizeFromDst(Op), IsStore, Op->Src[0], Op->Src[1]);
|
||||
}
|
||||
|
||||
template<IR::OpSize ElementSize, bool IsStore>
|
||||
void OpDispatchBuilder::AVX128_VMASKMOV(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_VMASKMOV(OpcodeArgs, IR::OpSize ElementSize, bool IsStore) {
|
||||
AVX128_VMASKMOVImpl(Op, ElementSize, OpSizeFromDst(Op), IsStore, Op->Src[0], Op->Src[1]);
|
||||
}
|
||||
|
||||
@@ -2114,8 +1623,7 @@ void OpDispatchBuilder::AVX128_MASKMOV(OpcodeArgs) {
|
||||
_StoreMem(FPRClass, Size, MemDest, XMMReg, OpSize::i8Bit);
|
||||
}
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void OpDispatchBuilder::AVX128_VectorVariableBlend(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_VectorVariableBlend(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
const auto Is128Bit = Size == OpSize::i128Bit;
|
||||
const auto Src3Selector = Op->Src[2].Literal();
|
||||
@@ -2130,7 +1638,7 @@ void OpDispatchBuilder::AVX128_VectorVariableBlend(OpcodeArgs) {
|
||||
Mask.High = AVX128_LoadXMMRegister(MaskRegister, true);
|
||||
}
|
||||
|
||||
auto Convert = [this](Ref Src1, Ref Src2, Ref Mask) {
|
||||
auto Convert = [&](Ref Src1, Ref Src2, Ref Mask) {
|
||||
const auto ElementSizeBits = IR::OpSizeAsBits(ElementSize);
|
||||
Ref Shifted = _VSShrI(OpSize::i128Bit, ElementSize, Mask, ElementSizeBits - 1);
|
||||
return _VBSL(OpSize::i128Bit, Shifted, Src2, Src1);
|
||||
@@ -2195,8 +1703,7 @@ void OpDispatchBuilder::AVX128_VPERM2(OpcodeArgs) {
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
}
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void OpDispatchBuilder::AVX128_VTESTP(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_VTESTP(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
const auto Size = GetSrcSize(Op);
|
||||
const auto Is128Bit = Size == Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
|
||||
@@ -2212,8 +1719,6 @@ void OpDispatchBuilder::AVX128_VTESTP(OpcodeArgs) {
|
||||
// For 256-bit, we need to split up the operation. This is nontrivial.
|
||||
// Let's go the simple route here.
|
||||
Ref ZF, CFInv;
|
||||
Ref ZeroConst = Constant(0);
|
||||
Ref OneConst = Constant(1);
|
||||
|
||||
const auto ElementSizeInBits = IR::OpSizeAsBits(ElementSize);
|
||||
|
||||
@@ -2255,7 +1760,7 @@ void OpDispatchBuilder::AVX128_VTESTP(OpcodeArgs) {
|
||||
|
||||
// ExtGPR will either be [0, 8] or [0, 16] If 0 then set Flag.
|
||||
auto ExtGPR = _VExtractToGPR(OpSize::i128Bit, ElementSize, AddWide, 0);
|
||||
CFInv = _Select(IR::COND_NEQ, ExtGPR, ZeroConst, OneConst, ZeroConst);
|
||||
CFInv = To01(OpSize::i64Bit, ExtGPR);
|
||||
}
|
||||
|
||||
// As in PTest, this sets Z appropriately while zeroing the rest of NZCV.
|
||||
@@ -2294,10 +1799,7 @@ void OpDispatchBuilder::AVX128_PTest(OpcodeArgs) {
|
||||
Test1 = _VExtractToGPR(OpSize::i128Bit, OpSize::i16Bit, Test1, 0);
|
||||
Test2 = _VExtractToGPR(OpSize::i128Bit, OpSize::i16Bit, Test2, 0);
|
||||
|
||||
auto ZeroConst = Constant(0);
|
||||
auto OneConst = Constant(1);
|
||||
|
||||
Test2 = _Select(FEXCore::IR::COND_NEQ, Test2, ZeroConst, OneConst, ZeroConst);
|
||||
Test2 = To01(OpSize::i64Bit, Test2);
|
||||
|
||||
// Careful, these flags are different between {V,}PTEST and VTESTP{S,D}
|
||||
// Set ZF according to Test1. SF will be zeroed since we do a 32-bit test on
|
||||
@@ -2308,10 +1810,9 @@ void OpDispatchBuilder::AVX128_PTest(OpcodeArgs) {
|
||||
ZeroPF_AF();
|
||||
}
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void OpDispatchBuilder::AVX128_VPERMILReg(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_VPERMILReg(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
AVX128_VectorBinaryImpl(Op, OpSizeFromSrc(Op), ElementSize, [this](IR::OpSize _ElementSize, Ref Src, Ref Indices) {
|
||||
return VPERMILRegOpImpl(OpSize::i128Bit, ElementSize, Src, Indices);
|
||||
return VPERMILRegOpImpl(OpSize::i128Bit, _ElementSize, Src, Indices);
|
||||
});
|
||||
}
|
||||
|
||||
@@ -2338,6 +1839,11 @@ void OpDispatchBuilder::AVX128_VPERMD(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VPCLMULQDQ(OpcodeArgs) {
|
||||
if (!CTX->HostFeatures.SupportsPMULL_128Bit) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[2].Literal());
|
||||
|
||||
AVX128_VectorBinaryImpl(Op, OpSizeFromSrc(Op), OpSize::iInvalid, [this, Selector](IR::OpSize, Ref Src1, Ref Src2) {
|
||||
@@ -2518,7 +2024,7 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherImpl(OpSize Size, O
|
||||
///< BaseAddr doesn't need to exist, calculate that here.
|
||||
Ref BaseAddr = VSIB.BaseAddr;
|
||||
if (BaseAddr && VSIB.Displacement) {
|
||||
BaseAddr = _Add(OpSize::i64Bit, BaseAddr, Constant(VSIB.Displacement));
|
||||
BaseAddr = Add(OpSize::i64Bit, BaseAddr, VSIB.Displacement);
|
||||
} else if (VSIB.Displacement) {
|
||||
BaseAddr = Constant(VSIB.Displacement);
|
||||
} else if (!BaseAddr) {
|
||||
@@ -2613,7 +2119,7 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherQPSImpl(Ref Dest, R
|
||||
///< BaseAddr doesn't need to exist, calculate that here.
|
||||
Ref BaseAddr = VSIB.BaseAddr;
|
||||
if (BaseAddr && VSIB.Displacement) {
|
||||
BaseAddr = _Add(OpSize::i64Bit, BaseAddr, Constant(VSIB.Displacement));
|
||||
BaseAddr = Add(OpSize::i64Bit, BaseAddr, VSIB.Displacement);
|
||||
} else if (VSIB.Displacement) {
|
||||
BaseAddr = Constant(VSIB.Displacement);
|
||||
} else if (!BaseAddr) {
|
||||
@@ -2648,8 +2154,7 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherQPSImpl(Ref Dest, R
|
||||
return Result;
|
||||
}
|
||||
|
||||
template<OpSize AddrElementSize>
|
||||
void OpDispatchBuilder::AVX128_VPGATHER(OpcodeArgs) {
|
||||
void OpDispatchBuilder::AVX128_VPGATHER(OpcodeArgs, OpSize AddrElementSize) {
|
||||
|
||||
const auto Size = OpSizeFromDst(Op);
|
||||
const auto Is128Bit = Size == OpSize::i128Bit;
|
||||
|
||||
@@ -53,10 +53,11 @@ constexpr inline DispatchTableEntry OpDispatch_BaseOpTable[] = {
|
||||
{0xAA, 2, &OpDispatchBuilder::STOSOp},
|
||||
{0xAC, 2, &OpDispatchBuilder::LODSOp},
|
||||
{0xAE, 2, &OpDispatchBuilder::SCASOp},
|
||||
{0xB0, 16, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVGPROp, 0>},
|
||||
{0xB0, 16, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVGPRImmediate>},
|
||||
{0xC2, 2, &OpDispatchBuilder::RETOp},
|
||||
{0xC8, 1, &OpDispatchBuilder::EnterOp},
|
||||
{0xC9, 1, &OpDispatchBuilder::LEAVEOp},
|
||||
{0xCA, 2, &OpDispatchBuilder::RETFARIndirectOp},
|
||||
{0xCC, 2, &OpDispatchBuilder::INTOp},
|
||||
{0xCF, 1, &OpDispatchBuilder::IRETOp},
|
||||
{0xD7, 2, &OpDispatchBuilder::XLATOp},
|
||||
@@ -75,33 +76,4 @@ constexpr inline DispatchTableEntry OpDispatch_BaseOpTable[] = {
|
||||
{0xFA, 2, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{0xFC, 2, &OpDispatchBuilder::FLAGControlOp},
|
||||
};
|
||||
|
||||
constexpr inline DispatchTableEntry OpDispatch_BaseOpTable_64[] = {
|
||||
{0x63, 1, &OpDispatchBuilder::MOVSXDOp},
|
||||
{0xA0, 4, &OpDispatchBuilder::MOVOffsetOp},
|
||||
};
|
||||
|
||||
constexpr inline DispatchTableEntry OpDispatch_BaseOpTable_32[] = {
|
||||
{0x06, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX>},
|
||||
{0x07, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX>},
|
||||
{0x0E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX>},
|
||||
{0x16, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX>},
|
||||
{0x17, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX>},
|
||||
{0x1E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX>},
|
||||
{0x1F, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX>},
|
||||
{0x27, 1, &OpDispatchBuilder::DAAOp},
|
||||
{0x2F, 1, &OpDispatchBuilder::DASOp},
|
||||
{0x37, 1, &OpDispatchBuilder::AAAOp},
|
||||
{0x3F, 1, &OpDispatchBuilder::AASOp},
|
||||
{0x40, 8, &OpDispatchBuilder::INCOp},
|
||||
{0x48, 8, &OpDispatchBuilder::DECOp},
|
||||
|
||||
{0x60, 1, &OpDispatchBuilder::PUSHAOp},
|
||||
{0x61, 1, &OpDispatchBuilder::POPAOp},
|
||||
{0xA0, 4, &OpDispatchBuilder::MOVOffsetOp},
|
||||
{0xCE, 1, &OpDispatchBuilder::INTOp},
|
||||
{0xD4, 1, &OpDispatchBuilder::AAMOp},
|
||||
{0xD5, 1, &OpDispatchBuilder::AADOp},
|
||||
{0xD6, 1, &OpDispatchBuilder::SALCOp},
|
||||
};
|
||||
} // namespace FEXCore::IR
|
||||
@@ -19,6 +19,10 @@ class OrderedNode;
|
||||
#define OpcodeArgs [[maybe_unused]] FEXCore::X86Tables::DecodedOp Op
|
||||
|
||||
void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
|
||||
if (!CTX->HostFeatures.SupportsSHA) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
@@ -36,6 +40,10 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
|
||||
if (!CTX->HostFeatures.SupportsSHA) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
@@ -48,6 +56,10 @@ void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
if (!CTX->HostFeatures.SupportsSHA) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
@@ -62,6 +74,10 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
if (!CTX->HostFeatures.SupportsSHA) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
const uint64_t Imm8 = Op->Src[1].Literal() & 0b11;
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
@@ -100,6 +116,10 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
|
||||
if (!CTX->HostFeatures.SupportsSHA) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
@@ -109,6 +129,10 @@ void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
if (!CTX->HostFeatures.SupportsSHA) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
@@ -121,17 +145,11 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::BitwiseAtLeastTwo(Ref A, Ref B, Ref C) {
|
||||
// Returns whether at least 2/3 of A/B/C is true.
|
||||
// Expressed as (A & (B | C)) | (B & C)
|
||||
//
|
||||
// Equivalent to expression in SHA calculations: (A & B) ^ (A & C) ^ (B & C)
|
||||
auto And = _And(OpSize::i32Bit, B, C);
|
||||
auto Or = _Or(OpSize::i32Bit, B, C);
|
||||
return _Or(OpSize::i32Bit, _And(OpSize::i32Bit, A, Or), And);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
if (!CTX->HostFeatures.SupportsSHA) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
// Hardcoded to XMM0
|
||||
@@ -163,12 +181,20 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
|
||||
if (!CTX->HostFeatures.SupportsAES) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESImc(Src);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
|
||||
if (!CTX->HostFeatures.SupportsAES) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESEnc(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
@@ -177,7 +203,7 @@ void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
[[maybe_unused]] const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
|
||||
// TODO: Handle 256-bit VAESENC.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENC unimplemented");
|
||||
@@ -190,6 +216,10 @@ void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
|
||||
if (!CTX->HostFeatures.SupportsAES) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESEncLast(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
@@ -198,7 +228,7 @@ void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
[[maybe_unused]] const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
|
||||
// TODO: Handle 256-bit VAESENCLAST.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENCLAST unimplemented");
|
||||
@@ -211,6 +241,10 @@ void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
|
||||
if (!CTX->HostFeatures.SupportsAES) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESDec(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
@@ -219,7 +253,7 @@ void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
[[maybe_unused]] const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
|
||||
// TODO: Handle 256-bit VAESDEC.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDEC unimplemented");
|
||||
@@ -232,6 +266,10 @@ void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
|
||||
if (!CTX->HostFeatures.SupportsAES) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESDecLast(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
@@ -240,7 +278,7 @@ void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
[[maybe_unused]] const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
const auto Is128Bit = DstSize == OpSize::i128Bit;
|
||||
|
||||
// TODO: Handle 256-bit VAESDECLAST.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDECLAST unimplemented");
|
||||
@@ -261,11 +299,20 @@ Ref OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) {
|
||||
if (!CTX->HostFeatures.SupportsAES) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
|
||||
Ref Result = AESKeyGenAssistImpl(Op);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
|
||||
if (!CTX->HostFeatures.SupportsPMULL_128Bit) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[1].Literal());
|
||||
@@ -275,6 +322,10 @@ void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) {
|
||||
if (!CTX->HostFeatures.SupportsPMULL_128Bit) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
|
||||
Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
@@ -201,7 +201,7 @@ void OpDispatchBuilder::FixupAF() {
|
||||
auto PFRaw = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
|
||||
auto AFRaw = GetRFLAG(FEXCore::X86State::RFLAG_AF_RAW_LOC);
|
||||
|
||||
// Again 64-bit as masking is more expensive given our ConstProp design.
|
||||
// Again 64-bit as masking is more expensive.
|
||||
Ref XorRes = _Xor(OpSize::i64Bit, AFRaw, PFRaw);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
|
||||
}
|
||||
@@ -267,8 +267,6 @@ Ref OpDispatchBuilder::IncrementByCarry(OpSize OpSize, Ref Src) {
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::CalculateFlags_ADC(IR::OpSize SrcSize, Ref Src1, Ref Src2) {
|
||||
auto Zero = _InlineConstant(0);
|
||||
auto One = _InlineConstant(1);
|
||||
auto OpSize = SrcSize == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
Ref Res;
|
||||
|
||||
@@ -288,11 +286,11 @@ Ref OpDispatchBuilder::CalculateFlags_ADC(IR::OpSize SrcSize, Ref Src1, Ref Src2
|
||||
Ref Src2PlusCF = IncrementByCarry(OpSize, Src2);
|
||||
|
||||
// Need to zero-extend for the comparison.
|
||||
Res = _Add(OpSize, Src1, Src2PlusCF);
|
||||
Res = Add(OpSize, Src1, Src2PlusCF);
|
||||
Res = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Res);
|
||||
|
||||
// TODO: We can fold that second Bfe in (cmp uxth).
|
||||
auto SelectCFInv = _Select(FEXCore::IR::COND_UGE, Res, Src2PlusCF, One, Zero);
|
||||
auto SelectCFInv = Select01(OpSize, CondClassType {COND_UGE}, Res, Src2PlusCF);
|
||||
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
SetCFInverted(SelectCFInv);
|
||||
@@ -304,8 +302,6 @@ Ref OpDispatchBuilder::CalculateFlags_ADC(IR::OpSize SrcSize, Ref Src1, Ref Src2
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::CalculateFlags_SBB(IR::OpSize SrcSize, Ref Src1, Ref Src2) {
|
||||
auto Zero = _InlineConstant(0);
|
||||
auto One = _InlineConstant(1);
|
||||
auto OpSize = SrcSize == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
|
||||
CalculateAF(Src1, Src2);
|
||||
@@ -325,10 +321,10 @@ Ref OpDispatchBuilder::CalculateFlags_SBB(IR::OpSize SrcSize, Ref Src1, Ref Src2
|
||||
|
||||
auto Src2PlusCF = IncrementByCarry(OpSize, Src2);
|
||||
|
||||
Res = _Sub(OpSize, Src1, Src2PlusCF);
|
||||
Res = Sub(OpSize, Src1, Src2PlusCF);
|
||||
Res = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Res);
|
||||
|
||||
auto SelectCFInv = _Select(FEXCore::IR::COND_UGE, Src1, Src2PlusCF, One, Zero);
|
||||
auto SelectCFInv = Select01(OpSize, CondClassType {COND_UGE}, Src1, Src2PlusCF);
|
||||
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
SetCFInverted(SelectCFInv);
|
||||
@@ -349,10 +345,10 @@ Ref OpDispatchBuilder::CalculateFlags_SUB(IR::OpSize SrcSize, Ref Src1, Ref Src2
|
||||
|
||||
Ref Res;
|
||||
if (SrcSize >= OpSize::i32Bit) {
|
||||
Res = _SubWithFlags(SrcSize, Src1, Src2);
|
||||
Res = SubWithFlags(SrcSize, Src1, Src2);
|
||||
} else {
|
||||
_SubNZCV(SrcSize, Src1, Src2);
|
||||
Res = _Sub(OpSize::i32Bit, Src1, Src2);
|
||||
Res = Sub(OpSize::i32Bit, Src1, Src2);
|
||||
}
|
||||
|
||||
CalculatePF(Res);
|
||||
@@ -379,10 +375,10 @@ Ref OpDispatchBuilder::CalculateFlags_ADD(IR::OpSize SrcSize, Ref Src1, Ref Src2
|
||||
|
||||
Ref Res;
|
||||
if (SrcSize >= OpSize::i32Bit) {
|
||||
Res = _AddWithFlags(SrcSize, Src1, Src2);
|
||||
Res = AddWithFlags(SrcSize, Src1, Src2);
|
||||
} else {
|
||||
_AddNZCV(SrcSize, Src1, Src2);
|
||||
Res = _Add(OpSize::i32Bit, Src1, Src2);
|
||||
Res = Add(OpSize::i32Bit, Src1, Src2);
|
||||
}
|
||||
|
||||
CalculatePF(Res);
|
||||
|
||||
@@ -6,6 +6,7 @@ namespace FEXCore::IR {
|
||||
#define OPD(prefix, opcode) (((prefix) << 8) | opcode)
|
||||
constexpr uint16_t PF_38_NONE = 0;
|
||||
constexpr uint16_t PF_38_66 = (1U << 0);
|
||||
constexpr uint16_t PF_38_F2 = (1U << 1);
|
||||
constexpr uint16_t PF_38_F3 = (1U << 2);
|
||||
|
||||
constexpr DispatchTableEntry OpDispatch_H0F38Table[] = {
|
||||
@@ -71,9 +72,28 @@ constexpr DispatchTableEntry OpDispatch_H0F38Table[] = {
|
||||
{OPD(PF_38_66, 0x40), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VMUL, OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x41), 1, &OpDispatchBuilder::PHMINPOSUWOp},
|
||||
|
||||
{OPD(PF_38_NONE, 0xC8), 1, &OpDispatchBuilder::SHA1NEXTEOp},
|
||||
{OPD(PF_38_NONE, 0xC9), 1, &OpDispatchBuilder::SHA1MSG1Op},
|
||||
{OPD(PF_38_NONE, 0xCA), 1, &OpDispatchBuilder::SHA1MSG2Op},
|
||||
{OPD(PF_38_NONE, 0xCB), 1, &OpDispatchBuilder::SHA256RNDS2Op},
|
||||
{OPD(PF_38_NONE, 0xCC), 1, &OpDispatchBuilder::SHA256MSG1Op},
|
||||
{OPD(PF_38_NONE, 0xCD), 1, &OpDispatchBuilder::SHA256MSG2Op},
|
||||
|
||||
{OPD(PF_38_66, 0xDB), 1, &OpDispatchBuilder::AESImcOp},
|
||||
{OPD(PF_38_66, 0xDC), 1, &OpDispatchBuilder::AESEncOp},
|
||||
{OPD(PF_38_66, 0xDD), 1, &OpDispatchBuilder::AESEncLastOp},
|
||||
{OPD(PF_38_66, 0xDE), 1, &OpDispatchBuilder::AESDecOp},
|
||||
{OPD(PF_38_66, 0xDF), 1, &OpDispatchBuilder::AESDecLastOp},
|
||||
|
||||
{OPD(PF_38_NONE, 0xF0), 2, &OpDispatchBuilder::MOVBEOp},
|
||||
{OPD(PF_38_66, 0xF0), 2, &OpDispatchBuilder::MOVBEOp},
|
||||
|
||||
{OPD(PF_38_F2, 0xF0), 1, &OpDispatchBuilder::CRC32},
|
||||
{OPD(PF_38_F2, 0xF1), 1, &OpDispatchBuilder::CRC32},
|
||||
|
||||
{OPD(PF_38_66 | PF_38_F2, 0xF0), 1, &OpDispatchBuilder::CRC32},
|
||||
{OPD(PF_38_66 | PF_38_F2, 0xF1), 1, &OpDispatchBuilder::CRC32},
|
||||
|
||||
{OPD(PF_38_66, 0xF6), 1, &OpDispatchBuilder::ADXOp},
|
||||
{OPD(PF_38_F3, 0xF6), 1, &OpDispatchBuilder::ADXOp},
|
||||
};
|
||||
|
||||
@@ -29,6 +29,7 @@ constexpr auto OpDispatchTableGenH0F3A = []() consteval {
|
||||
{OPD(REX, PF_3A_66, 0x40), 1, &OpDispatchBuilder::DPPOp<OpSize::i32Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x41), 1, &OpDispatchBuilder::DPPOp<OpSize::i64Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x42), 1, &OpDispatchBuilder::MPSADBWOp},
|
||||
{OPD(REX, PF_3A_66, 0x44), 1, &OpDispatchBuilder::PCLMULQDQOp},
|
||||
|
||||
{OPD(REX, PF_3A_66, 0x60), 1, &OpDispatchBuilder::VPCMPESTRMOp},
|
||||
{OPD(REX, PF_3A_66, 0x61), 1, &OpDispatchBuilder::VPCMPESTRIOp},
|
||||
@@ -36,6 +37,8 @@ constexpr auto OpDispatchTableGenH0F3A = []() consteval {
|
||||
{OPD(REX, PF_3A_66, 0x63), 1, &OpDispatchBuilder::VPCMPISTRIOp},
|
||||
|
||||
{OPD(REX, PF_3A_NONE, 0xCC), 1, &OpDispatchBuilder::SHA1RNDS4Op},
|
||||
{OPD(REX, PF_3A_66, 0xDF), 1, &OpDispatchBuilder::AESKeyGenAssist},
|
||||
|
||||
};
|
||||
return std::to_array(Table);
|
||||
};
|
||||
@@ -65,11 +68,6 @@ constexpr DispatchTableEntry OpDispatch_H0F3ATableNeedsREX0[] = {
|
||||
{OPD(0, PF_3A_66, 0x22), 1, &OpDispatchBuilder::PINSROp<OpSize::i32Bit>},
|
||||
};
|
||||
|
||||
constexpr DispatchTableEntry OpDispatch_H0F3ATable_64[] = {
|
||||
{OPD(1, PF_3A_66, 0x16), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i64Bit>},
|
||||
{OPD(1, PF_3A_66, 0x22), 1, &OpDispatchBuilder::PINSROp<OpSize::i64Bit>},
|
||||
};
|
||||
|
||||
#undef PF_3A_NONE
|
||||
#undef PF_3A_66
|
||||
|
||||
|
||||
@@ -117,7 +117,9 @@ constexpr DispatchTableEntry OpDispatch_PrimaryGroupTables[] = {
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 0), 1, &OpDispatchBuilder::INCOp}, // INC
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 1), 1, &OpDispatchBuilder::DECOp}, // DEC
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 2), 1, &OpDispatchBuilder::CALLAbsoluteOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 3), 1, &OpDispatchBuilder::CALLFARIndirectOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 4), 1, &OpDispatchBuilder::JUMPAbsoluteOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 5), 1, &OpDispatchBuilder::JUMPFARIndirectOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 6), 1, &OpDispatchBuilder::PUSHOp},
|
||||
|
||||
// GROUP 11
|
||||
|
||||
@@ -69,10 +69,16 @@ constexpr DispatchTableEntry OpDispatch_SecondaryGroupTables[] = {
|
||||
|
||||
// GROUP 9
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_NONE, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_F3, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_NONE, 6), 1, &OpDispatchBuilder::RDRANDOp<false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_NONE, 7), 1, &OpDispatchBuilder::RDRANDOp<true>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_66, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_66, 6), 1, &OpDispatchBuilder::RDRANDOp<false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_66, 7), 1, &OpDispatchBuilder::RDRANDOp<true>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_F2, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_F3, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_F3, 7), 1, &OpDispatchBuilder::RDPIDOp},
|
||||
|
||||
// GROUP 12
|
||||
@@ -156,18 +162,6 @@ constexpr DispatchTableEntry OpDispatch_SecondaryGroupTables[] = {
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_F2, 0), 8, &OpDispatchBuilder::NOPOp},
|
||||
};
|
||||
|
||||
constexpr DispatchTableEntry OpDispatch_SecondaryGroupTables_64[] = {
|
||||
// GROUP 15
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 0), 1,
|
||||
&OpDispatchBuilder::Bind<&OpDispatchBuilder::ReadSegmentReg, OpDispatchBuilder::Segment::FS>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 1), 1,
|
||||
&OpDispatchBuilder::Bind<&OpDispatchBuilder::ReadSegmentReg, OpDispatchBuilder::Segment::GS>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 2), 1,
|
||||
&OpDispatchBuilder::Bind<&OpDispatchBuilder::WriteSegmentReg, OpDispatchBuilder::Segment::FS>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_F3, 3), 1,
|
||||
&OpDispatchBuilder::Bind<&OpDispatchBuilder::WriteSegmentReg, OpDispatchBuilder::Segment::GS>},
|
||||
};
|
||||
|
||||
#undef OPD
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -17,6 +17,7 @@ constexpr DispatchTableEntry OpDispatch_SecondaryModRMTables[] = {
|
||||
// REG /7
|
||||
{((3 << 3) | 0), 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{((3 << 3) | 1), 1, &OpDispatchBuilder::RDTSCPOp},
|
||||
{((3 << 3) | 4), 1, &OpDispatchBuilder::CLZeroOp},
|
||||
};
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -313,20 +313,4 @@ constexpr DispatchTableEntry OpDispatch_SecondaryOpSizeModTables[] = {
|
||||
{0xFD, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, OpSize::i16Bit>},
|
||||
{0xFE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VADD, OpSize::i32Bit>},
|
||||
};
|
||||
|
||||
constexpr DispatchTableEntry OpDispatch_TwoByteOpTable_64[] = {
|
||||
{0x05, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SyscallOp, true>},
|
||||
{0xA0, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX>},
|
||||
{0xA1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX>},
|
||||
{0xA8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX>},
|
||||
{0xA9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX>},
|
||||
};
|
||||
|
||||
constexpr DispatchTableEntry OpDispatch_TwoByteOpTable_32[] = {
|
||||
{0x05, 1, &OpDispatchBuilder::NOPOp},
|
||||
{0xA0, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX>},
|
||||
{0xA1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX>},
|
||||
{0xA8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX>},
|
||||
{0xA9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX>},
|
||||
};
|
||||
} // namespace FEXCore::IR
|
||||
@@ -477,7 +477,7 @@ Ref OpDispatchBuilder::InsertScalar_CVT_Float_To_FloatImpl(OpcodeArgs, IR::OpSiz
|
||||
// We load the full vector width when dealing with a source vector,
|
||||
// so that we don't do any unnecessary zero extension to the scalar
|
||||
// element that we're going to operate on.
|
||||
const auto SrcSize = OpSizeFromSrc(Op);
|
||||
const auto SrcSize = Src2Op.IsGPR() ? OpSize::i128Bit : SrcElementSize;
|
||||
|
||||
Ref Src1 = LoadSource_WithOpSize(FPRClass, Op, Src1Op, DstSize, Op->Flags);
|
||||
Ref Src2 = LoadSource_WithOpSize(FPRClass, Op, Src2Op, SrcSize, Op->Flags, {.AllowUpperGarbage = true});
|
||||
@@ -2557,6 +2557,8 @@ void OpDispatchBuilder::XSaveOpImpl(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SaveX87State(OpcodeArgs, Ref MemBase) {
|
||||
_SyncStackToSlow();
|
||||
|
||||
// Saves 512bytes to the memory location provided
|
||||
// Header changes depending on if REX.W is set or not
|
||||
if (Op->Flags & X86Tables::DecodeFlags::FLAG_REX_WIDENING) {
|
||||
@@ -2580,7 +2582,8 @@ void OpDispatchBuilder::SaveX87State(OpcodeArgs, Ref MemBase) {
|
||||
|
||||
{
|
||||
// Abridged FTW
|
||||
_StoreMem(GPRClass, OpSize::i8Bit, LoadContext(AbridgedFTWIndex), MemBase, Constant(4), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
auto FTW = _LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
_StoreMem(GPRClass, OpSize::i8Bit, FTW, MemBase, Constant(4), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
// BYTE | 0 1 | 2 3 | 4 | 5 | 6 7 | 8 9 | a b | c d | e f |
|
||||
@@ -2627,9 +2630,19 @@ void OpDispatchBuilder::SaveX87State(OpcodeArgs, Ref MemBase) {
|
||||
// MXCSR_MASK: Mask for writes to the MXCSR register
|
||||
// If OSFXSR bit in CR4 is not set than FXSAVE /may/ not save the XMM registers
|
||||
// This is implementation dependent
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; i += 2) {
|
||||
RefPair MMRegs = LoadContextPair(OpSize::i128Bit, MM0Index + i);
|
||||
_StoreMemPair(FPRClass, OpSize::i128Bit, MMRegs.Low, MMRegs.High, MemBase, i * 16 + 32);
|
||||
//
|
||||
// x87 registers are stored rotated depending on the current TOP.
|
||||
Ref Top = GetX87Top();
|
||||
auto SevenConst = Constant(7);
|
||||
const auto LoadSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
Ref data = _LoadContextIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
|
||||
if (ReducedPrecisionMode) {
|
||||
data = _F80CVTTo(data, OpSize::i64Bit);
|
||||
}
|
||||
_StoreMem(FPRClass, OpSize::i128Bit, data, MemBase, Constant(16 * i + 32), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
Top = _And(OpSize::i32Bit, Add(OpSize::i32Bit, Top, 1), SevenConst);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2740,6 +2753,8 @@ void OpDispatchBuilder::XRstorOpImpl(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RestoreX87State(Ref MemBase) {
|
||||
_StackForceSlow();
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, MemBase, OpSize::i16Bit);
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
@@ -2750,14 +2765,14 @@ void OpDispatchBuilder::RestoreX87State(Ref MemBase) {
|
||||
|
||||
{
|
||||
// Abridged FTW
|
||||
StoreContext(AbridgedFTWIndex, _LoadMem(GPRClass, OpSize::i8Bit, MemBase, Constant(4), OpSize::i8Bit, MEM_OFFSET_SXTX, 1));
|
||||
auto NewFTW = _LoadMem(GPRClass, OpSize::i8Bit, MemBase, Constant(4), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
_StoreContext(OpSize::i8Bit, GPRClass, NewFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
}
|
||||
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; i += 2) {
|
||||
auto MMRegs = LoadMemPair(FPRClass, OpSize::i128Bit, MemBase, i * 16 + 32);
|
||||
|
||||
StoreContext(MM0Index + i, MMRegs.Low);
|
||||
StoreContext(MM0Index + i + 1, MMRegs.High);
|
||||
_StoreContext(OpSize::i128Bit, FPRClass, MMRegs.Low, MMBaseOffset() + i * 16);
|
||||
_StoreContext(OpSize::i128Bit, FPRClass, MMRegs.High, MMBaseOffset() + (i + 1) * 16);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2803,7 +2818,7 @@ void OpDispatchBuilder::DefaultX87State(OpcodeArgs) {
|
||||
// all of the ST0-7/MM0-7 registers to zero.
|
||||
Ref ZeroVector = LoadZeroVector(OpSize::i64Bit);
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
StoreContext(MM0Index + i, ZeroVector);
|
||||
_StoreContext(OpSize::i128Bit, FPRClass, ZeroVector, MMBaseOffset() + i * 16);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3950,10 +3965,7 @@ void OpDispatchBuilder::PTestOpImpl(OpSize Size, Ref Dest, Ref Src) {
|
||||
Test1 = _VExtractToGPR(Size, OpSize::i16Bit, Test1, 0);
|
||||
Test2 = _VExtractToGPR(Size, OpSize::i16Bit, Test2, 0);
|
||||
|
||||
auto ZeroConst = Constant(0);
|
||||
auto OneConst = Constant(1);
|
||||
|
||||
Test2 = _Select(FEXCore::IR::COND_NEQ, Test2, ZeroConst, OneConst, ZeroConst);
|
||||
Test2 = To01(OpSize::i64Bit, Test2);
|
||||
|
||||
// Careful, these flags are different between {V,}PTEST and VTESTP{S,D}
|
||||
// Set ZF according to Test1. SF will be zeroed since we do a 32-bit test on
|
||||
@@ -3990,10 +4002,7 @@ void OpDispatchBuilder::VTESTOpImpl(OpSize SrcSize, IR::OpSize ElementSize, Ref
|
||||
Ref AndGPR = _VExtractToGPR(SrcSize, OpSize::i16Bit, MaxAnd, 0);
|
||||
Ref AndNotGPR = _VExtractToGPR(SrcSize, OpSize::i16Bit, MaxAndNot, 0);
|
||||
|
||||
Ref ZeroConst = Constant(0);
|
||||
Ref OneConst = Constant(1);
|
||||
|
||||
Ref CFInv = _Select(IR::COND_NEQ, AndNotGPR, ZeroConst, OneConst, ZeroConst);
|
||||
Ref CFInv = To01(OpSize::i64Bit, AndNotGPR);
|
||||
|
||||
// As in PTest, this sets Z appropriately while zeroing the rest of NZCV.
|
||||
SetNZ_ZeroCV(OpSize::i32Bit, AndGPR);
|
||||
@@ -5007,7 +5016,7 @@ void OpDispatchBuilder::VFMAddSubImpl(OpcodeArgs, bool AddSub, uint8_t Src1Idx,
|
||||
}
|
||||
|
||||
OpDispatchBuilder::RefVSIB OpDispatchBuilder::LoadVSIB(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags) {
|
||||
[[maybe_unused]] const bool IsVSIB = (Op->Flags & X86Tables::DecodeFlags::FLAG_VSIB_BYTE) != 0;
|
||||
const bool IsVSIB = (Op->Flags & X86Tables::DecodeFlags::FLAG_VSIB_BYTE) != 0;
|
||||
LOGMAN_THROW_A_FMT(Operand.IsSIB() && IsVSIB, "Trying to load VSIB for something that isn't the correct type!");
|
||||
|
||||
// VSIB is a very special case which has a ton of encoded data.
|
||||
@@ -5084,7 +5093,7 @@ void OpDispatchBuilder::VPGATHER(OpcodeArgs) {
|
||||
///< BaseAddr doesn't need to exist, calculate that here.
|
||||
Ref BaseAddr = VSIB.BaseAddr;
|
||||
if (BaseAddr && VSIB.Displacement) {
|
||||
BaseAddr = _Add(OpSize::i64Bit, BaseAddr, Constant(VSIB.Displacement));
|
||||
BaseAddr = Add(OpSize::i64Bit, BaseAddr, VSIB.Displacement);
|
||||
} else if (VSIB.Displacement) {
|
||||
BaseAddr = Constant(VSIB.Displacement);
|
||||
} else if (!BaseAddr) {
|
||||
|
||||
@@ -32,21 +32,27 @@ Ref OpDispatchBuilder::GetX87Top() {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SetX87FTW(Ref FTW) {
|
||||
Ref X87Empty = Constant(static_cast<uint8_t>(FPState::X87Tag::Empty));
|
||||
Ref NewAbridgedFTW {};
|
||||
_StackForceSlow(); // Invalidate x87 FTW register cache
|
||||
|
||||
for (int i = 0; i < 8; i++) {
|
||||
Ref RegTag = _Bfe(OpSize::i32Bit, 2, i * 2, FTW);
|
||||
Ref RegValid = _Select(FEXCore::IR::COND_NEQ, RegTag, X87Empty, Constant(1), Constant(0));
|
||||
// For the output, we want a 1-bit for each pair not equal to 11 (Empty).
|
||||
static_assert(static_cast<uint8_t>(FPState::X87Tag::Empty) == 0b11);
|
||||
|
||||
if (i) {
|
||||
NewAbridgedFTW = _Orlshl(OpSize::i32Bit, NewAbridgedFTW, RegValid, i);
|
||||
} else {
|
||||
NewAbridgedFTW = RegValid;
|
||||
}
|
||||
}
|
||||
// Make even bits 1 if the pair is equal to 11, and 0 otherwise.
|
||||
FTW = _AndShift(OpSize::i32Bit, FTW, FTW, ShiftType::LSR, 1);
|
||||
|
||||
StoreContext(AbridgedFTWIndex, NewAbridgedFTW);
|
||||
// Invert FTW and clear the odd bits. Even bits are 1 if the pair
|
||||
// is not equal to 11, and odd bits are 0.
|
||||
FTW = _Andn(OpSize::i32Bit, Constant(0x55555555), FTW);
|
||||
|
||||
// All that's left is to compact away the odd bits. That is a Morton
|
||||
// deinterleave operation, which has a standard solution. See
|
||||
// https://stackoverflow.com/questions/3137266/how-to-de-interleave-bits-unmortonizing
|
||||
FTW = _And(OpSize::i32Bit, _Orlshr(OpSize::i32Bit, FTW, FTW, 1), Constant(0x33333333));
|
||||
FTW = _And(OpSize::i32Bit, _Orlshr(OpSize::i32Bit, FTW, FTW, 2), Constant(0x0f0f0f0f));
|
||||
FTW = _Orlshr(OpSize::i32Bit, FTW, FTW, 4);
|
||||
|
||||
// ...and that's it. StoreContext implicitly does the final masking.
|
||||
_StoreContext(OpSize::i8Bit, GPRClass, FTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SetX87Top(Ref Value) {
|
||||
@@ -110,10 +116,10 @@ void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
auto absolute = _Neg(OpSize::i64Bit, Data, CondClassType {COND_MI});
|
||||
|
||||
// left justify the absolute integer
|
||||
auto shift = _Sub(OpSize::i64Bit, Constant(63), _FindMSB(IR::OpSize::i64Bit, absolute));
|
||||
auto shift = Sub(OpSize::i64Bit, Constant(63), _FindMSB(IR::OpSize::i64Bit, absolute));
|
||||
auto shifted = _Lshl(OpSize::i64Bit, absolute, shift);
|
||||
|
||||
auto adjusted_exponent = _Sub(OpSize::i64Bit, Constant(0x3fff + 63), shift);
|
||||
auto adjusted_exponent = Sub(OpSize::i64Bit, Constant(0x3fff + 63), shift);
|
||||
auto zeroed_exponent = _Select(COND_EQ, absolute, zero, zero, adjusted_exponent);
|
||||
auto upper = _Or(OpSize::i64Bit, sign, zeroed_exponent);
|
||||
|
||||
@@ -160,12 +166,12 @@ void OpDispatchBuilder::FIST(OpcodeArgs, bool Truncate) {
|
||||
// Check for NaN/Infinity: exponent = 0x7fff
|
||||
SaveNZCV();
|
||||
_TestNZ(OpSize::i64Bit, Exponent, Constant(0x7fff));
|
||||
Ref IsSpecial = _NZCVSelect(OpSize::i64Bit, {COND_EQ}, Constant(1), Constant(0));
|
||||
Ref IsSpecial = _NZCVSelect01({COND_EQ});
|
||||
|
||||
// For overflow detection, check if exponent indicates a value >= 2^15
|
||||
// Biased exponent for 2^15 is 0x3fff + 15 = 0x400e
|
||||
_SubWithFlags(OpSize::i64Bit, Exponent, Constant(0x400e));
|
||||
Ref IsOverflow = _NZCVSelect(OpSize::i64Bit, {COND_UGE}, Constant(1), Constant(0));
|
||||
SubWithFlags(OpSize::i64Bit, Exponent, 0x400e);
|
||||
Ref IsOverflow = _NZCVSelect01({COND_UGE});
|
||||
|
||||
// Set Invalid Operation flag if overflow or special value
|
||||
Ref InvalidFlag = _Or(OpSize::i64Bit, IsSpecial, IsOverflow);
|
||||
@@ -334,7 +340,7 @@ Ref OpDispatchBuilder::GetX87FTW_Helper() {
|
||||
// bytes, we use the well-known bit twiddling algorithm:
|
||||
//
|
||||
// https://graphics.stanford.edu/~seander/bithacks.html#InterleaveBMN
|
||||
Ref X = LoadContext(AbridgedFTWIndex);
|
||||
Ref X = _LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
X = _Orlshl(OpSize::i32Bit, X, X, 4);
|
||||
X = _And(OpSize::i32Bit, X, Constant(0x0f0f0f0f));
|
||||
X = _Orlshl(OpSize::i32Bit, X, X, 2);
|
||||
@@ -421,11 +427,13 @@ Ref OpDispatchBuilder::ReconstructX87StateFromFSW_Helper(Ref FSW) {
|
||||
auto C1 = _Bfe(OpSize::i32Bit, 1, 9, FSW);
|
||||
auto C2 = _Bfe(OpSize::i32Bit, 1, 10, FSW);
|
||||
auto C3 = _Bfe(OpSize::i32Bit, 1, 14, FSW);
|
||||
auto IE = _Bfe(OpSize::i32Bit, 1, 0, FSW);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(C0);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(C1);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(C2);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(C3);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_IE_LOC>(IE);
|
||||
return Top;
|
||||
}
|
||||
|
||||
@@ -439,13 +447,13 @@ void OpDispatchBuilder::X87LDENV(OpcodeArgs) {
|
||||
auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, Mem, OpSize::i16Bit);
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
Ref MemLocation = _Add(OpSize::i64Bit, Mem, Constant(IR::OpSizeToSize(Size) * 1));
|
||||
Ref MemLocation = Add(OpSize::i64Bit, Mem, IR::OpSizeToSize(Size) * 1);
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, MemLocation, Size);
|
||||
ReconstructX87StateFromFSW_Helper(NewFSW);
|
||||
|
||||
{
|
||||
// FTW
|
||||
Ref MemLocation = _Add(OpSize::i64Bit, Mem, Constant(IR::OpSizeToSize(Size) * 2));
|
||||
Ref MemLocation = Add(OpSize::i64Bit, Mem, IR::OpSizeToSize(Size) * 2);
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, MemLocation, Size));
|
||||
}
|
||||
}
|
||||
@@ -508,7 +516,6 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
auto OneConst = Constant(1);
|
||||
auto SevenConst = Constant(7);
|
||||
const auto LoadSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
for (int i = 0; i < 7; ++i) {
|
||||
@@ -517,7 +524,7 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
data = _F80CVTTo(data, OpSize::i64Bit);
|
||||
}
|
||||
_StoreMem(FPRClass, OpSize::i128Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
Top = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, Top, OneConst), SevenConst);
|
||||
Top = _And(OpSize::i32Bit, Add(OpSize::i32Bit, Top, 1), SevenConst);
|
||||
}
|
||||
|
||||
// The final st(7) needs a bit of special handling here
|
||||
@@ -561,9 +568,7 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MEM_OFFSET_SXTX, 1));
|
||||
}
|
||||
|
||||
auto OneConst = Constant(1);
|
||||
auto SevenConst = Constant(7);
|
||||
|
||||
auto low = Constant(~0ULL);
|
||||
auto high = Constant(0xFFFF);
|
||||
Ref Mask = _VLoadTwoGPRs(low, high);
|
||||
@@ -578,7 +583,7 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
}
|
||||
_StoreContextIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
|
||||
|
||||
Top = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, Top, OneConst), SevenConst);
|
||||
Top = _And(OpSize::i32Bit, Add(OpSize::i32Bit, Top, 1), SevenConst);
|
||||
}
|
||||
|
||||
// The final st(7) needs a bit of special handling here
|
||||
@@ -586,8 +591,7 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
// Lower 64bits [63:0]
|
||||
// upper 16 bits [79:64]
|
||||
Ref Reg = _LoadMem(FPRClass, OpSize::i64Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
Ref RegHigh =
|
||||
_LoadMem(FPRClass, OpSize::i16Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7) + 8), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
Ref RegHigh = _LoadMem(FPRClass, OpSize::i16Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7) + 8), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
Reg = _VInsElement(OpSize::i128Bit, OpSize::i16Bit, 4, 0, Reg, RegHigh);
|
||||
if (ReducedPrecisionMode) {
|
||||
Reg = _F80CVT(OpSize::i64Bit, Reg); // Convert to double precision
|
||||
@@ -677,6 +681,9 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs, IR::OpSize Width, bool Integer, OpDisp
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(PF);
|
||||
}
|
||||
|
||||
// Set Invalid Operation flag when unordered (NaN comparison)
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_IE_LOC>(HostFlag_Unordered);
|
||||
|
||||
if (PopTwice) {
|
||||
_PopStackDestroy();
|
||||
_PopStackDestroy();
|
||||
@@ -698,6 +705,9 @@ void OpDispatchBuilder::FTST(OpcodeArgs) {
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(HostFlag_Unordered);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(HostFlag_ZF);
|
||||
|
||||
// Set Invalid Operation flag when unordered (NaN comparison)
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_IE_LOC>(HostFlag_Unordered);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87OpHelper(OpcodeArgs, FEXCore::IR::IROps IROp, bool ZeroC2) {
|
||||
@@ -759,7 +769,14 @@ void OpDispatchBuilder::X87FNSTSW(OpcodeArgs) {
|
||||
StoreResult(GPRClass, Op, StatusWord, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FNCLEX(OpcodeArgs) {
|
||||
// Clear the exception flag bit
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_IE_LOC>(_Constant(0));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FNINIT(OpcodeArgs) {
|
||||
_SyncStackToSlow(); // Invalidate x87 register caches
|
||||
|
||||
auto Zero = Constant(0);
|
||||
|
||||
if (ReducedPrecisionMode) {
|
||||
@@ -773,7 +790,7 @@ void OpDispatchBuilder::FNINIT(OpcodeArgs) {
|
||||
// Set top to zero
|
||||
SetX87Top(Zero);
|
||||
// Tags all get marked as invalid
|
||||
StoreContext(AbridgedFTWIndex, Zero);
|
||||
_StoreContext(OpSize::i8Bit, GPRClass, Zero, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
|
||||
// Reinits the simulated stack
|
||||
_InitStack();
|
||||
@@ -782,6 +799,7 @@ void OpDispatchBuilder::FNINIT(OpcodeArgs) {
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_IE_LOC>(Zero);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FFREE(OpcodeArgs) {
|
||||
@@ -827,11 +845,7 @@ void OpDispatchBuilder::X87FCMOV(OpcodeArgs) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled FCMOV op: 0x{:x}", Opcode); break;
|
||||
}
|
||||
|
||||
auto ZeroConst = Constant(0);
|
||||
auto AllOneConst = Constant(0xffff'ffff'ffff'ffffull);
|
||||
|
||||
Ref SrcCond = SelectCC(CC, OpSize::i64Bit, AllOneConst, ZeroConst);
|
||||
Ref VecCond = _VDupFromGPR(OpSize::i128Bit, OpSize::i64Bit, SrcCond);
|
||||
Ref VecCond = _VDupFromGPR(OpSize::i128Bit, OpSize::i64Bit, SelectCC0All1(CC));
|
||||
_F80VBSLStack(OpSize::i128Bit, VecCond, Op->OP & 7, 0);
|
||||
}
|
||||
|
||||
@@ -847,11 +861,9 @@ void OpDispatchBuilder::X87FXAM(OpcodeArgs) {
|
||||
// Claim this is a normal number
|
||||
// We don't support anything else
|
||||
auto TopValid = _StackValidTag(0);
|
||||
auto ZeroConst = Constant(0);
|
||||
auto OneConst = Constant(1);
|
||||
|
||||
// In the case of top being invalid then C3:C2:C0 is 0b101
|
||||
auto C3 = _Select(FEXCore::IR::COND_NEQ, TopValid, OneConst, OneConst, ZeroConst);
|
||||
auto C3 = Select01(OpSize::i32Bit, CondClassType {COND_NEQ}, TopValid, Constant(1));
|
||||
|
||||
auto C2 = TopValid;
|
||||
auto C0 = C3; // Mirror C3 until something other than zero is supported
|
||||
|
||||
@@ -60,7 +60,7 @@ void OpDispatchBuilder::X87FLDCWF64(OpcodeArgs) {
|
||||
// Float load op with memory operand
|
||||
void OpDispatchBuilder::FLDF64(OpcodeArgs, IR::OpSize Width) {
|
||||
const auto ReadWidth = (Width == OpSize::f80Bit) ? OpSize::i128Bit : Width;
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], ReadWidth, Op->Flags);
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], Width, Op->Flags);
|
||||
// Convert to 64bit float
|
||||
Ref ConvertedData = Data;
|
||||
if (Width == OpSize::i32Bit) {
|
||||
@@ -73,7 +73,7 @@ void OpDispatchBuilder::FLDF64(OpcodeArgs, IR::OpSize Width) {
|
||||
|
||||
void OpDispatchBuilder::FBLDF64(OpcodeArgs) {
|
||||
// Read from memory
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::i128Bit, Op->Flags);
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::f80Bit, Op->Flags);
|
||||
Ref ConvertedData = _F80BCDLoad(Data);
|
||||
ConvertedData = _F80CVT(OpSize::i64Bit, ConvertedData);
|
||||
_PushStack(ConvertedData, Data, OpSize::i64Bit, true);
|
||||
@@ -382,7 +382,7 @@ void OpDispatchBuilder::X87FXTRACTF64(OpcodeArgs) {
|
||||
|
||||
// non zero case
|
||||
Ref ExpNZ = _Bfe(OpSize::i64Bit, 11, 52, Gpr);
|
||||
ExpNZ = _Sub(OpSize::i64Bit, ExpNZ, Constant(1023));
|
||||
ExpNZ = Sub(OpSize::i64Bit, ExpNZ, Constant(1023));
|
||||
Ref ExpNZV = _Float_FromGPR_S(OpSize::i64Bit, OpSize::i64Bit, ExpNZ);
|
||||
|
||||
Ref SigNZ = _And(OpSize::i64Bit, Gpr, Constant(0x800f'ffff'ffff'ffffLL));
|
||||
|
||||
@@ -33,7 +33,7 @@ X86GeneratedCode::X86GeneratedCode() {
|
||||
|
||||
CallbackReturn = reinterpret_cast<uint64_t>(CodePtr);
|
||||
|
||||
memcpy(reinterpret_cast<void*>(CallbackReturn), &SignalReturnCode.at(0), SignalReturnCode.size());
|
||||
memcpy(reinterpret_cast<void*>(CallbackReturn), SignalReturnCode.data(), SignalReturnCode.size());
|
||||
|
||||
mprotect(CodePtr, CODE_SIZE, PROT_READ);
|
||||
#endif
|
||||
|
||||
@@ -1,29 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
meta: frontend|x86-tables ~ Metadata that drives the frontend x86/64 decoding
|
||||
tags: frontend|x86-tables
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <FEXCore/Core/Context.h>
|
||||
|
||||
namespace FEXCore::X86Tables {
|
||||
|
||||
void InitializeBaseTables(Context::OperatingMode Mode);
|
||||
void InitializeSecondaryTables(Context::OperatingMode Mode);
|
||||
void InitializeSecondaryGroupTables(Context::OperatingMode Mode);
|
||||
void InitializePrimaryGroupTables(Context::OperatingMode Mode);
|
||||
void InitializeH0F3ATables(Context::OperatingMode Mode);
|
||||
|
||||
void InitializeInfoTables(Context::OperatingMode Mode) {
|
||||
InitializeBaseTables(Mode);
|
||||
InitializeSecondaryTables(Mode);
|
||||
InitializeSecondaryGroupTables(Mode);
|
||||
InitializePrimaryGroupTables(Mode);
|
||||
InitializeH0F3ATables(Mode);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::X86Tables
|
||||
@@ -15,300 +15,425 @@ $end_info$
|
||||
namespace FEXCore::X86Tables {
|
||||
using namespace InstFlags;
|
||||
|
||||
std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
enum Primary_LUT {
|
||||
ENTRY_06,
|
||||
ENTRY_07,
|
||||
ENTRY_0E,
|
||||
ENTRY_16,
|
||||
ENTRY_17,
|
||||
ENTRY_1E,
|
||||
ENTRY_1F,
|
||||
ENTRY_27,
|
||||
ENTRY_2F,
|
||||
ENTRY_37,
|
||||
ENTRY_3F,
|
||||
ENTRY_40,
|
||||
ENTRY_48,
|
||||
ENTRY_60,
|
||||
ENTRY_61,
|
||||
ENTRY_63,
|
||||
ENTRY_9A,
|
||||
ENTRY_A0,
|
||||
ENTRY_A1,
|
||||
ENTRY_A2,
|
||||
ENTRY_A3,
|
||||
ENTRY_CE,
|
||||
ENTRY_D4,
|
||||
ENTRY_D5,
|
||||
ENTRY_D6,
|
||||
ENTRY_EA,
|
||||
ENTRY_MAX,
|
||||
};
|
||||
|
||||
constexpr std::array<X86InstInfo[2], ENTRY_MAX> Primary_ArchSelect_LUT = {{
|
||||
// ENTRY_06
|
||||
{
|
||||
{"PUSH ES", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_DEBUG_MEM_ACCESS, 0, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX> } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_07
|
||||
{
|
||||
{"POP ES", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_DEF) | FLAGS_DEBUG_MEM_ACCESS, 0, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX> } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_0E
|
||||
{
|
||||
{"PUSH CS", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_DEBUG_MEM_ACCESS, 0, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX> } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_16
|
||||
{
|
||||
{"PUSH SS", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_DEBUG_MEM_ACCESS, 0, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX> } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_17
|
||||
{
|
||||
{"POP SS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_DEF) | FLAGS_DEBUG_MEM_ACCESS, 0, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX> } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_1E
|
||||
{
|
||||
{"PUSH DS", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_DEBUG_MEM_ACCESS, 0, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::PUSHSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX> } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_1F
|
||||
{
|
||||
{"POP DS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_DEF) | FLAGS_DEBUG_MEM_ACCESS, 0, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::POPSegmentOp, FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX> } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_27
|
||||
{
|
||||
{"DAA", TYPE_INST, GenFlagsDstSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 0, { .OpDispatch = &IR::OpDispatchBuilder::DAAOp } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_2F
|
||||
{
|
||||
{"DAS", TYPE_INST, GenFlagsDstSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 0, { .OpDispatch = &IR::OpDispatchBuilder::DASOp } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_37
|
||||
{
|
||||
{"AAA", TYPE_INST, GenFlagsDstSize(SIZE_16BIT) | FLAGS_SF_DST_RAX, 0, { .OpDispatch = &IR::OpDispatchBuilder::AAAOp } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_3F
|
||||
{
|
||||
{"AAS", TYPE_INST, GenFlagsDstSize(SIZE_16BIT) | FLAGS_SF_DST_RAX, 0, { .OpDispatch = &IR::OpDispatchBuilder::AASOp } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_40
|
||||
{
|
||||
{"INC", TYPE_INST, FLAGS_SF_REX_IN_BYTE, 0, { .OpDispatch = &IR::OpDispatchBuilder::INCOp } },
|
||||
// REX
|
||||
{"", TYPE_REX_PREFIX, FLAGS_NONE, 0},
|
||||
},
|
||||
// ENTRY_48
|
||||
{
|
||||
{"DEC", TYPE_INST, FLAGS_SF_REX_IN_BYTE, 0, { .OpDispatch = &IR::OpDispatchBuilder::DECOp } },
|
||||
{"", TYPE_REX_PREFIX, FLAGS_NONE, 0},
|
||||
},
|
||||
// ENTRY_60
|
||||
{
|
||||
{"PUSHA", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0, { .OpDispatch = &IR::OpDispatchBuilder::PUSHAOp } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_61
|
||||
{
|
||||
{"POPA", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0, { .OpDispatch = &IR::OpDispatchBuilder::POPAOp } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_63
|
||||
{
|
||||
{"ARPL", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
{"MOVSXD", TYPE_INST, GenFlagsDstSize(SIZE_64BIT) | FLAGS_MODRM, 0, { .OpDispatch = &IR::OpDispatchBuilder::MOVSXDOp } },
|
||||
},
|
||||
// ENTRY_9A
|
||||
{
|
||||
{"CALLF", TYPE_INST, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_A0
|
||||
{
|
||||
{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_MEM_OFFSET, 4, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_MEM_OFFSET, 8, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
},
|
||||
// ENTRY_A1
|
||||
{
|
||||
{"MOV", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_MEM_OFFSET, 4, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
{"MOV", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_MEM_OFFSET, 8, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
},
|
||||
// ENTRY_A2
|
||||
{
|
||||
{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_SRC_RAX | FLAGS_MEM_OFFSET, 4, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_SRC_RAX | FLAGS_MEM_OFFSET, 8, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
},
|
||||
// ENTRY_A3
|
||||
{
|
||||
{"MOV", TYPE_INST, FLAGS_SF_SRC_RAX | FLAGS_MEM_OFFSET, 4, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
{"MOV", TYPE_INST, FLAGS_SF_SRC_RAX | FLAGS_MEM_OFFSET, 8, { .OpDispatch = &IR::OpDispatchBuilder::MOVOffsetOp } },
|
||||
},
|
||||
// ENTRY_CE
|
||||
{
|
||||
{"INTO", TYPE_INST, FLAGS_NONE, 0, { .OpDispatch = &IR::OpDispatchBuilder::INTOp } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_D4
|
||||
{
|
||||
{"AAM", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 1, { .OpDispatch = &IR::OpDispatchBuilder::AAMOp } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_D5
|
||||
{
|
||||
{"AAD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 1, { .OpDispatch = &IR::OpDispatchBuilder::AADOp } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_D6
|
||||
{
|
||||
{"SALC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_SF_SRC_RAX, 0, { .OpDispatch = &IR::OpDispatchBuilder::SALCOp } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
// ENTRY_EA
|
||||
{
|
||||
{"JMPF", TYPE_INST, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
}};
|
||||
|
||||
const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> Table{};
|
||||
|
||||
constexpr U8U8InfoStruct BaseOpTable[] = {
|
||||
// Prefixes
|
||||
// Operand size overide
|
||||
{0x66, 1, X86InstInfo{"", TYPE_PREFIX, FLAGS_NONE, 0, nullptr}},
|
||||
{0x66, 1, X86InstInfo{"", TYPE_PREFIX, FLAGS_NONE, 0}},
|
||||
// Address size override
|
||||
{0x67, 1, X86InstInfo{"", TYPE_PREFIX, FLAGS_NONE, 0, nullptr}},
|
||||
{0x26, 1, X86InstInfo{"ES", TYPE_LEGACY_PREFIX, FLAGS_NONE, 0, nullptr}},
|
||||
{0x2E, 1, X86InstInfo{"CS", TYPE_LEGACY_PREFIX, FLAGS_NONE, 0, nullptr}},
|
||||
{0x36, 1, X86InstInfo{"SS", TYPE_LEGACY_PREFIX, FLAGS_NONE, 0, nullptr}},
|
||||
{0x3E, 1, X86InstInfo{"DS", TYPE_LEGACY_PREFIX, FLAGS_NONE, 0, nullptr}},
|
||||
{0x67, 1, X86InstInfo{"", TYPE_PREFIX, FLAGS_NONE, 0}},
|
||||
{0x26, 1, X86InstInfo{"ES", TYPE_LEGACY_PREFIX, FLAGS_NONE, 0}},
|
||||
{0x2E, 1, X86InstInfo{"CS", TYPE_LEGACY_PREFIX, FLAGS_NONE, 0}},
|
||||
{0x36, 1, X86InstInfo{"SS", TYPE_LEGACY_PREFIX, FLAGS_NONE, 0}},
|
||||
{0x3E, 1, X86InstInfo{"DS", TYPE_LEGACY_PREFIX, FLAGS_NONE, 0}},
|
||||
// These are still invalid on 64bit
|
||||
{0x64, 1, X86InstInfo{"FS", TYPE_PREFIX, FLAGS_NONE, 0, nullptr}},
|
||||
{0x65, 1, X86InstInfo{"GS", TYPE_PREFIX, FLAGS_NONE, 0, nullptr}},
|
||||
{0xF0, 1, X86InstInfo{"LOCK", TYPE_PREFIX, FLAGS_NONE, 0, nullptr}},
|
||||
{0xF2, 1, X86InstInfo{"REPNE", TYPE_PREFIX, FLAGS_NONE, 0, nullptr}},
|
||||
{0xF3, 1, X86InstInfo{"REP", TYPE_PREFIX, FLAGS_NONE, 0, nullptr}},
|
||||
{0x64, 1, X86InstInfo{"FS", TYPE_PREFIX, FLAGS_NONE, 0}},
|
||||
{0x65, 1, X86InstInfo{"GS", TYPE_PREFIX, FLAGS_NONE, 0}},
|
||||
{0xF0, 1, X86InstInfo{"LOCK", TYPE_PREFIX, FLAGS_NONE, 0}},
|
||||
{0xF2, 1, X86InstInfo{"REPNE", TYPE_PREFIX, FLAGS_NONE, 0}},
|
||||
{0xF3, 1, X86InstInfo{"REP", TYPE_PREFIX, FLAGS_NONE, 0}},
|
||||
|
||||
// Instructions
|
||||
{0x00, 1, X86InstInfo{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{0x01, 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_DISPLACE_SIZE_DIV_2, 0, nullptr}},
|
||||
{0x02, 1, X86InstInfo{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0, nullptr}},
|
||||
{0x03, 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{0x04, 1, X86InstInfo{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1, nullptr}},
|
||||
{0x05, 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4, nullptr}},
|
||||
{0x00, 1, X86InstInfo{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x01, 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_DISPLACE_SIZE_DIV_2, 0}},
|
||||
{0x02, 1, X86InstInfo{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x03, 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x04, 1, X86InstInfo{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0x05, 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
|
||||
{0x08, 1, X86InstInfo{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{0x09, 1, X86InstInfo{"OR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{0x0A, 1, X86InstInfo{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0, nullptr}},
|
||||
{0x0B, 1, X86InstInfo{"OR", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{0x0C, 1, X86InstInfo{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1, nullptr}},
|
||||
{0x0D, 1, X86InstInfo{"OR", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4, nullptr}},
|
||||
{0x06, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_06] }}},
|
||||
{0x07, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_07] }}},
|
||||
|
||||
{0x10, 1, X86InstInfo{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{0x11, 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_DISPLACE_SIZE_DIV_2, 0, nullptr}},
|
||||
{0x12, 1, X86InstInfo{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0, nullptr}},
|
||||
{0x13, 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{0x14, 1, X86InstInfo{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1, nullptr}},
|
||||
{0x15, 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4, nullptr}},
|
||||
{0x08, 1, X86InstInfo{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x09, 1, X86InstInfo{"OR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x0A, 1, X86InstInfo{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x0B, 1, X86InstInfo{"OR", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x0C, 1, X86InstInfo{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0x0D, 1, X86InstInfo{"OR", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{0x0E, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_0E] }}},
|
||||
|
||||
{0x18, 1, X86InstInfo{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{0x19, 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_DISPLACE_SIZE_DIV_2, 0, nullptr}},
|
||||
{0x1A, 1, X86InstInfo{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0, nullptr}},
|
||||
{0x1B, 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{0x1C, 1, X86InstInfo{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1, nullptr}},
|
||||
{0x1D, 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4, nullptr}},
|
||||
{0x10, 1, X86InstInfo{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x11, 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_DISPLACE_SIZE_DIV_2, 0}},
|
||||
{0x12, 1, X86InstInfo{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x13, 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x14, 1, X86InstInfo{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0x15, 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{0x16, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_16] }}},
|
||||
{0x17, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_17] }}},
|
||||
|
||||
{0x20, 1, X86InstInfo{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{0x21, 1, X86InstInfo{"AND", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{0x22, 1, X86InstInfo{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0, nullptr}},
|
||||
{0x23, 1, X86InstInfo{"AND", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{0x24, 1, X86InstInfo{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1, nullptr}},
|
||||
{0x25, 1, X86InstInfo{"AND", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4, nullptr}},
|
||||
{0x18, 1, X86InstInfo{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x19, 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_DISPLACE_SIZE_DIV_2, 0}},
|
||||
{0x1A, 1, X86InstInfo{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x1B, 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x1C, 1, X86InstInfo{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0x1D, 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{0x1E, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_1E] }}},
|
||||
{0x1F, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_1F] }}},
|
||||
|
||||
{0x28, 1, X86InstInfo{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{0x29, 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{0x2A, 1, X86InstInfo{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0, nullptr}},
|
||||
{0x2B, 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{0x2C, 1, X86InstInfo{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1, nullptr}},
|
||||
{0x2D, 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4, nullptr}},
|
||||
{0x20, 1, X86InstInfo{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x21, 1, X86InstInfo{"AND", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x22, 1, X86InstInfo{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x23, 1, X86InstInfo{"AND", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x24, 1, X86InstInfo{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0x25, 1, X86InstInfo{"AND", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
|
||||
{0x30, 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{0x31, 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{0x32, 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0, nullptr}},
|
||||
{0x33, 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{0x34, 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1, nullptr}},
|
||||
{0x35, 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4, nullptr}},
|
||||
{0x27, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_27] }}},
|
||||
{0x28, 1, X86InstInfo{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x29, 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x2A, 1, X86InstInfo{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x2B, 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x2C, 1, X86InstInfo{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0x2D, 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{0x2F, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_2F] }}},
|
||||
|
||||
{0x38, 1, X86InstInfo{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{0x39, 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{0x3A, 1, X86InstInfo{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0, nullptr}},
|
||||
{0x3B, 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{0x3C, 1, X86InstInfo{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1, nullptr}},
|
||||
{0x3D, 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4, nullptr}},
|
||||
{0x30, 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x31, 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x32, 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x33, 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x34, 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0x35, 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
|
||||
{0x50, 8, X86InstInfo{"PUSH", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SF_REX_IN_BYTE | FLAGS_DEBUG_MEM_ACCESS , 0, nullptr}},
|
||||
{0x58, 8, X86InstInfo{"POP", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SF_REX_IN_BYTE | FLAGS_DEBUG_MEM_ACCESS , 0, nullptr}},
|
||||
{0x37, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_37] }}},
|
||||
{0x38, 1, X86InstInfo{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x39, 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x3A, 1, X86InstInfo{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x3B, 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x3C, 1, X86InstInfo{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0x3D, 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{0x3F, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_3F] }}},
|
||||
|
||||
{0x62, 1, X86InstInfo{"", TYPE_GROUP_EVEX, FLAGS_NONE, 0, nullptr}},
|
||||
{0x40, 8, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_40] }}},
|
||||
{0x48, 8, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_48] }}},
|
||||
|
||||
{0x68, 1, X86InstInfo{"PUSH", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SRC_SEXT, 4, nullptr}},
|
||||
{0x69, 1, X86InstInfo{"IMUL", TYPE_INST, FLAGS_MODRM | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4, nullptr}},
|
||||
{0x6A, 1, X86InstInfo{"PUSH", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
{0x6B, 1, X86InstInfo{"IMUL", TYPE_INST, FLAGS_MODRM | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
{0x50, 8, X86InstInfo{"PUSH", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SF_REX_IN_BYTE | FLAGS_DEBUG_MEM_ACCESS , 0}},
|
||||
{0x58, 8, X86InstInfo{"POP", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SF_REX_IN_BYTE | FLAGS_DEBUG_MEM_ACCESS , 0}},
|
||||
|
||||
|
||||
{0x60, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_60] }}},
|
||||
{0x61, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_61] }}},
|
||||
{0x62, 1, X86InstInfo{"", TYPE_GROUP_EVEX, FLAGS_NONE, 0}},
|
||||
{0x63, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_63] }}},
|
||||
|
||||
{0x68, 1, X86InstInfo{"PUSH", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SRC_SEXT, 4}},
|
||||
{0x69, 1, X86InstInfo{"IMUL", TYPE_INST, FLAGS_MODRM | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{0x6A, 1, X86InstInfo{"PUSH", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SRC_SEXT , 1}},
|
||||
{0x6B, 1, X86InstInfo{"IMUL", TYPE_INST, FLAGS_MODRM | FLAGS_SRC_SEXT , 1}},
|
||||
|
||||
// This should just throw a GP
|
||||
{0x6C, 1, X86InstInfo{"INSB", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0x6D, 1, X86InstInfo{"INSW", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0x6E, 1, X86InstInfo{"OUTS", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0x6F, 1, X86InstInfo{"OUTS", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0x6C, 1, X86InstInfo{"INSB", TYPE_INST, FLAGS_BLOCK_END, 0}},
|
||||
{0x6D, 1, X86InstInfo{"INSW", TYPE_INST, FLAGS_BLOCK_END, 0}},
|
||||
{0x6E, 1, X86InstInfo{"OUTS", TYPE_INST, FLAGS_BLOCK_END, 0}},
|
||||
{0x6F, 1, X86InstInfo{"OUTS", TYPE_INST, FLAGS_BLOCK_END, 0}},
|
||||
|
||||
{0x70, 1, X86InstInfo{"JO", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
{0x71, 1, X86InstInfo{"JNO", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
{0x72, 1, X86InstInfo{"JB", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
{0x73, 1, X86InstInfo{"JNB", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
{0x74, 1, X86InstInfo{"JZ", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
{0x75, 1, X86InstInfo{"JNZ", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
{0x76, 1, X86InstInfo{"JBE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
{0x77, 1, X86InstInfo{"JNBE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
{0x78, 1, X86InstInfo{"JS", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
{0x79, 1, X86InstInfo{"JNS", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
{0x7A, 1, X86InstInfo{"JP", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
{0x7B, 1, X86InstInfo{"JNP", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
{0x7C, 1, X86InstInfo{"JL", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
{0x7D, 1, X86InstInfo{"JNL", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
{0x7E, 1, X86InstInfo{"JLE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
{0x7F, 1, X86InstInfo{"JNLE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
{0x70, 1, X86InstInfo{"JO", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1}},
|
||||
{0x71, 1, X86InstInfo{"JNO", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1}},
|
||||
{0x72, 1, X86InstInfo{"JB", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1}},
|
||||
{0x73, 1, X86InstInfo{"JNB", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1}},
|
||||
{0x74, 1, X86InstInfo{"JZ", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1}},
|
||||
{0x75, 1, X86InstInfo{"JNZ", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1}},
|
||||
{0x76, 1, X86InstInfo{"JBE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1}},
|
||||
{0x77, 1, X86InstInfo{"JNBE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1}},
|
||||
{0x78, 1, X86InstInfo{"JS", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1}},
|
||||
{0x79, 1, X86InstInfo{"JNS", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1}},
|
||||
{0x7A, 1, X86InstInfo{"JP", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1}},
|
||||
{0x7B, 1, X86InstInfo{"JNP", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1}},
|
||||
{0x7C, 1, X86InstInfo{"JL", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1}},
|
||||
{0x7D, 1, X86InstInfo{"JNL", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1}},
|
||||
{0x7E, 1, X86InstInfo{"JLE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1}},
|
||||
{0x7F, 1, X86InstInfo{"JNLE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1}},
|
||||
|
||||
{0x84, 1, X86InstInfo{"TEST", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{0x85, 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{0x86, 1, X86InstInfo{"XCHG", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{0x87, 1, X86InstInfo{"XCHG", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{0x84, 1, X86InstInfo{"TEST", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x85, 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x86, 1, X86InstInfo{"XCHG", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x87, 1, X86InstInfo{"XCHG", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
|
||||
{0x88, 1, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{0x89, 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{0x8A, 1, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0, nullptr}},
|
||||
{0x8B, 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{0x8C, 1, X86InstInfo{"MOV", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{0x8D, 1, X86InstInfo{"LEA", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_MODRM, 0, nullptr}},
|
||||
{0x8E, 1, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_16BIT) | FLAGS_MODRM, 0, nullptr}},
|
||||
{0x8F, 1, X86InstInfo{"POP", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_ZERO_REG | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0x90, 8, X86InstInfo{"XCHG", TYPE_INST, FLAGS_SF_REX_IN_BYTE | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0x98, 1, X86InstInfo{"CDQE", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0x99, 1, X86InstInfo{"CQO", TYPE_INST, FLAGS_SF_DST_RDX | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0x88, 1, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x89, 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x8A, 1, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x8B, 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x8C, 1, X86InstInfo{"MOV", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x8D, 1, X86InstInfo{"LEA", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_MODRM, 0}},
|
||||
{0x8E, 1, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_16BIT) | FLAGS_MODRM, 0}},
|
||||
{0x8F, 1, X86InstInfo{"POP", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_ZERO_REG | FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{0x90, 8, X86InstInfo{"XCHG", TYPE_INST, FLAGS_SF_REX_IN_BYTE | FLAGS_SF_SRC_RAX, 0}},
|
||||
{0x98, 1, X86InstInfo{"CDQE", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SF_SRC_RAX, 0}},
|
||||
{0x99, 1, X86InstInfo{"CQO", TYPE_INST, FLAGS_SF_DST_RDX | FLAGS_SF_SRC_RAX, 0}},
|
||||
|
||||
// These three are all X87 instructions
|
||||
{0x9B, 1, X86InstInfo{"FWAIT", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x9C, 1, X86InstInfo{"PUSHF", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF), 0, nullptr}},
|
||||
{0x9D, 1, X86InstInfo{"POPF", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0x9B, 1, X86InstInfo{"FWAIT", TYPE_INST, FLAGS_NONE, 0}},
|
||||
{0x9C, 1, X86InstInfo{"PUSHF", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF), 0}},
|
||||
{0x9D, 1, X86InstInfo{"POPF", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_BLOCK_END, 0}},
|
||||
|
||||
{0x9E, 1, X86InstInfo{"SAHF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x9F, 1, X86InstInfo{"LAHF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x9E, 1, X86InstInfo{"SAHF", TYPE_INST, FLAGS_NONE, 0}},
|
||||
{0x9F, 1, X86InstInfo{"LAHF", TYPE_INST, FLAGS_NONE, 0}},
|
||||
|
||||
{0xA4, 1, X86InstInfo{"MOVSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0xA5, 1, X86InstInfo{"MOVS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0xA6, 1, X86InstInfo{"CMPSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0xA7, 1, X86InstInfo{"CMPS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0xA0, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_A0] }}},
|
||||
{0xA1, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_A1] }}},
|
||||
{0xA2, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_A2] }}},
|
||||
{0xA3, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_A3] }}},
|
||||
|
||||
{0xA8, 1, X86InstInfo{"TEST", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1, nullptr}},
|
||||
{0xA9, 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4, nullptr}},
|
||||
{0xAA, 1, X86InstInfo{"STOS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0xAB, 1, X86InstInfo{"STOS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0xAC, 1, X86InstInfo{"LODS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0xAD, 1, X86InstInfo{"LODS", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0xAE, 1, X86InstInfo{"SCAS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0xAF, 1, X86InstInfo{"SCAS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0xA4, 1, X86InstInfo{"MOVSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{0xA5, 1, X86InstInfo{"MOVS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{0xA6, 1, X86InstInfo{"CMPSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{0xA7, 1, X86InstInfo{"CMPS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
|
||||
{0xB0, 8, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_REX_IN_BYTE , 1, nullptr}},
|
||||
{0xB8, 8, X86InstInfo{"MOV", TYPE_INST, FLAGS_SF_REX_IN_BYTE | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_DISPLACE_SIZE_MUL_2, 4, nullptr}},
|
||||
{0xA8, 1, X86InstInfo{"TEST", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0xA9, 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{0xAA, 1, X86InstInfo{"STOS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0}},
|
||||
{0xAB, 1, X86InstInfo{"STOS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0}},
|
||||
{0xAC, 1, X86InstInfo{"LODS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{0xAD, 1, X86InstInfo{"LODS", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
{0xAE, 1, X86InstInfo{"SCAS", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0}},
|
||||
{0xAF, 1, X86InstInfo{"SCAS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SF_SRC_RAX, 0}},
|
||||
|
||||
{0xC2, 1, X86InstInfo{"RET", TYPE_INST, FLAGS_SETS_RIP | FLAGS_BLOCK_END, 2, nullptr}},
|
||||
{0xC3, 1, X86InstInfo{"RET", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_BLOCK_END , 0, nullptr}},
|
||||
{0xC8, 1, X86InstInfo{"ENTER", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_DEBUG_MEM_ACCESS , 3, nullptr}},
|
||||
{0xC9, 1, X86InstInfo{"LEAVE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_DEBUG_MEM_ACCESS , 0, nullptr}},
|
||||
{0xCA, 2, X86InstInfo{"RETF", TYPE_PRIV, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0xCC, 1, X86InstInfo{"INT3", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0xCD, 1, X86InstInfo{"INT", TYPE_INST, DEFAULT_SYSCALL_FLAGS, 1, nullptr}},
|
||||
{0xCF, 1, X86InstInfo{"IRET", TYPE_INST, FLAGS_SETS_RIP | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0xB0, 8, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_REX_IN_BYTE , 1}},
|
||||
{0xB8, 8, X86InstInfo{"MOV", TYPE_INST, FLAGS_SF_REX_IN_BYTE | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_DISPLACE_SIZE_MUL_2, 4}},
|
||||
|
||||
{0xD7, 1, X86InstInfo{"XLAT", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0xC2, 1, X86InstInfo{"RET", TYPE_INST, FLAGS_SETS_RIP | FLAGS_BLOCK_END, 2}},
|
||||
{0xC3, 1, X86InstInfo{"RET", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_BLOCK_END , 0}},
|
||||
{0xC8, 1, X86InstInfo{"ENTER", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_DEBUG_MEM_ACCESS , 3}},
|
||||
{0xC9, 1, X86InstInfo{"LEAVE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_DEBUG_MEM_ACCESS , 0}},
|
||||
{0xCA, 1, X86InstInfo{"RETF", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_BLOCK_END, 2}},
|
||||
{0xCB, 1, X86InstInfo{"RETF", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_BLOCK_END, 0}},
|
||||
{0xCC, 1, X86InstInfo{"INT3", TYPE_INST, FLAGS_BLOCK_END, 0}},
|
||||
{0xCD, 1, X86InstInfo{"INT", TYPE_INST, DEFAULT_SYSCALL_FLAGS, 1}},
|
||||
{0xCE, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_CE] }}},
|
||||
{0xCF, 1, X86InstInfo{"IRET", TYPE_INST, FLAGS_SETS_RIP | FLAGS_BLOCK_END, 0}},
|
||||
|
||||
{0xE0, 1, X86InstInfo{"LOOPNE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT | FLAGS_SF_SRC_RCX, 1, nullptr}},
|
||||
{0xE1, 1, X86InstInfo{"LOOPE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT | FLAGS_SF_SRC_RCX, 1, nullptr}},
|
||||
{0xE2, 1, X86InstInfo{"LOOP", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT | FLAGS_SF_SRC_RCX, 1, nullptr}},
|
||||
{0xE3, 1, X86InstInfo{"JrCXZ", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1, nullptr}},
|
||||
{0xD4, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_D4] }}},
|
||||
{0xD5, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_D5] }}},
|
||||
{0xD6, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_D6] }}},
|
||||
{0xD7, 1, X86InstInfo{"XLAT", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0}},
|
||||
|
||||
{0xE0, 1, X86InstInfo{"LOOPNE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT | FLAGS_SF_SRC_RCX, 1}},
|
||||
{0xE1, 1, X86InstInfo{"LOOPE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT | FLAGS_SF_SRC_RCX, 1}},
|
||||
{0xE2, 1, X86InstInfo{"LOOP", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT | FLAGS_SF_SRC_RCX, 1}},
|
||||
{0xE3, 1, X86InstInfo{"JrCXZ", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT , 1}},
|
||||
|
||||
// Should just throw GP
|
||||
{0xE4, 2, X86InstInfo{"IN", TYPE_INST, FLAGS_BLOCK_END, 1, nullptr}},
|
||||
{0xE6, 2, X86InstInfo{"OUT", TYPE_INST, FLAGS_BLOCK_END, 1, nullptr}},
|
||||
{0xE4, 2, X86InstInfo{"IN", TYPE_INST, FLAGS_BLOCK_END, 1}},
|
||||
{0xE6, 2, X86InstInfo{"OUT", TYPE_INST, FLAGS_BLOCK_END, 1}},
|
||||
|
||||
{0xE8, 1, X86InstInfo{"CALL", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_BLOCK_END | FLAGS_CALL , 4, nullptr}},
|
||||
{0xE9, 1, X86InstInfo{"JMP", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_BLOCK_END , 4, nullptr}},
|
||||
{0xEB, 1, X86InstInfo{"JMP", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT | FLAGS_BLOCK_END , 1, nullptr}},
|
||||
{0xE8, 1, X86InstInfo{"CALL", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_BLOCK_END | FLAGS_CALL , 4}},
|
||||
{0xE9, 1, X86InstInfo{"JMP", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_BLOCK_END , 4}},
|
||||
{0xEB, 1, X86InstInfo{"JMP", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_SRC_SEXT | FLAGS_BLOCK_END , 1}},
|
||||
|
||||
// Should just throw GP
|
||||
{0xEC, 2, X86InstInfo{"IN", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0xEE, 2, X86InstInfo{"OUT", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0xEC, 2, X86InstInfo{"IN", TYPE_INST, FLAGS_BLOCK_END, 0}},
|
||||
{0xEE, 2, X86InstInfo{"OUT", TYPE_INST, FLAGS_BLOCK_END, 0}},
|
||||
|
||||
{0xF1, 1, X86InstInfo{"INT1", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0xF4, 1, X86InstInfo{"HLT", TYPE_INST, FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0xF5, 1, X86InstInfo{"CMC", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xF8, 1, X86InstInfo{"CLC", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xF9, 1, X86InstInfo{"STC", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xFA, 1, X86InstInfo{"CLI", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xFB, 1, X86InstInfo{"STI", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xFC, 1, X86InstInfo{"CLD", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xFD, 1, X86InstInfo{"STD", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xF1, 1, X86InstInfo{"INT1", TYPE_INST, FLAGS_BLOCK_END, 0}},
|
||||
{0xF4, 1, X86InstInfo{"HLT", TYPE_INST, FLAGS_BLOCK_END, 0}},
|
||||
{0xF5, 1, X86InstInfo{"CMC", TYPE_INST, FLAGS_NONE, 0}},
|
||||
{0xF8, 1, X86InstInfo{"CLC", TYPE_INST, FLAGS_NONE, 0}},
|
||||
{0xF9, 1, X86InstInfo{"STC", TYPE_INST, FLAGS_NONE, 0}},
|
||||
{0xFA, 1, X86InstInfo{"CLI", TYPE_INST, FLAGS_NONE, 0}},
|
||||
{0xFB, 1, X86InstInfo{"STI", TYPE_INST, FLAGS_NONE, 0}},
|
||||
{0xFC, 1, X86InstInfo{"CLD", TYPE_INST, FLAGS_NONE, 0}},
|
||||
{0xFD, 1, X86InstInfo{"STD", TYPE_INST, FLAGS_NONE, 0}},
|
||||
|
||||
// Two Byte table
|
||||
{0x0F, 1, X86InstInfo{"", TYPE_SECONDARY_TABLE_PREFIX, FLAGS_NONE, 0, nullptr}},
|
||||
{0x0F, 1, X86InstInfo{"", TYPE_SECONDARY_TABLE_PREFIX, FLAGS_NONE, 0}},
|
||||
|
||||
// x87 table
|
||||
{0xD8, 8, X86InstInfo{"", TYPE_X87_TABLE_PREFIX, FLAGS_MODRM, 0, nullptr}},
|
||||
{0xD8, 8, X86InstInfo{"", TYPE_X87_TABLE_PREFIX, FLAGS_MODRM, 0}},
|
||||
|
||||
// ModRM table
|
||||
// MoreBytes field repurposed for valid bits mask
|
||||
{0x80, 1, X86InstInfo{"", TYPE_GROUP_1, FLAGS_MODRM, 0, nullptr}},
|
||||
{0x81, 1, X86InstInfo{"", TYPE_GROUP_1, FLAGS_MODRM, 1, nullptr}},
|
||||
{0x82, 1, X86InstInfo{"", TYPE_GROUP_1, FLAGS_MODRM, 2, nullptr}},
|
||||
{0x83, 1, X86InstInfo{"", TYPE_GROUP_1, FLAGS_MODRM, 3, nullptr}},
|
||||
{0xC0, 1, X86InstInfo{"", TYPE_GROUP_2, FLAGS_MODRM, 0, nullptr}},
|
||||
{0xC1, 1, X86InstInfo{"", TYPE_GROUP_2, FLAGS_MODRM, 1, nullptr}},
|
||||
{0xD0, 1, X86InstInfo{"", TYPE_GROUP_2, FLAGS_MODRM, 2, nullptr}},
|
||||
{0xD1, 1, X86InstInfo{"", TYPE_GROUP_2, FLAGS_MODRM, 3, nullptr}},
|
||||
{0xD2, 1, X86InstInfo{"", TYPE_GROUP_2, FLAGS_MODRM, 4, nullptr}},
|
||||
{0xD3, 1, X86InstInfo{"", TYPE_GROUP_2, FLAGS_MODRM, 5, nullptr}},
|
||||
{0xF6, 1, X86InstInfo{"", TYPE_GROUP_3, FLAGS_MODRM, 0, nullptr}},
|
||||
{0xF7, 1, X86InstInfo{"", TYPE_GROUP_3, FLAGS_MODRM, 1, nullptr}},
|
||||
{0xFE, 1, X86InstInfo{"", TYPE_GROUP_4, FLAGS_MODRM, 0, nullptr}},
|
||||
{0xFF, 1, X86InstInfo{"", TYPE_GROUP_5, FLAGS_MODRM, 0, nullptr}},
|
||||
{0x80, 1, X86InstInfo{"", TYPE_GROUP_1, FLAGS_MODRM, 0}},
|
||||
{0x81, 1, X86InstInfo{"", TYPE_GROUP_1, FLAGS_MODRM, 1}},
|
||||
{0x82, 1, X86InstInfo{"", TYPE_GROUP_1, FLAGS_MODRM, 2}},
|
||||
{0x83, 1, X86InstInfo{"", TYPE_GROUP_1, FLAGS_MODRM, 3}},
|
||||
{0xC0, 1, X86InstInfo{"", TYPE_GROUP_2, FLAGS_MODRM, 0}},
|
||||
{0xC1, 1, X86InstInfo{"", TYPE_GROUP_2, FLAGS_MODRM, 1}},
|
||||
{0xD0, 1, X86InstInfo{"", TYPE_GROUP_2, FLAGS_MODRM, 2}},
|
||||
{0xD1, 1, X86InstInfo{"", TYPE_GROUP_2, FLAGS_MODRM, 3}},
|
||||
{0xD2, 1, X86InstInfo{"", TYPE_GROUP_2, FLAGS_MODRM, 4}},
|
||||
{0xD3, 1, X86InstInfo{"", TYPE_GROUP_2, FLAGS_MODRM, 5}},
|
||||
{0xF6, 1, X86InstInfo{"", TYPE_GROUP_3, FLAGS_MODRM, 0}},
|
||||
{0xF7, 1, X86InstInfo{"", TYPE_GROUP_3, FLAGS_MODRM, 1}},
|
||||
{0xFE, 1, X86InstInfo{"", TYPE_GROUP_4, FLAGS_MODRM, 0}},
|
||||
{0xFF, 1, X86InstInfo{"", TYPE_GROUP_5, FLAGS_MODRM, 0}},
|
||||
|
||||
// Group 11
|
||||
{0xC6, 1, X86InstInfo{"", TYPE_GROUP_11, FLAGS_MODRM, 0, nullptr}},
|
||||
{0xC7, 1, X86InstInfo{"", TYPE_GROUP_11, FLAGS_MODRM, 1, nullptr}},
|
||||
{0xC6, 1, X86InstInfo{"", TYPE_GROUP_11, FLAGS_MODRM, 0}},
|
||||
{0xC7, 1, X86InstInfo{"", TYPE_GROUP_11, FLAGS_MODRM, 1}},
|
||||
|
||||
// VEX table
|
||||
{0xC4, 2, X86InstInfo{"", TYPE_VEX_TABLE_PREFIX, FLAGS_NONE, 0, nullptr}},
|
||||
{0xC4, 2, X86InstInfo{"", TYPE_VEX_TABLE_PREFIX, FLAGS_NONE, 0}},
|
||||
};
|
||||
|
||||
GenerateTable(&Table.at(0), BaseOpTable, std::size(BaseOpTable));
|
||||
GenerateTable(Table.data(), BaseOpTable, std::size(BaseOpTable));
|
||||
IR::InstallToTable(Table, IR::OpDispatch_BaseOpTable);
|
||||
|
||||
return Table;
|
||||
}();
|
||||
|
||||
void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
static constexpr U8U8InfoStruct BaseOpTable_64[] = {
|
||||
{0x06, 2, X86InstInfo{"[INV]", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x0E, 1, X86InstInfo{"[INV]", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x16, 2, X86InstInfo{"[INV]", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x1E, 2, X86InstInfo{"[INV]", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x27, 1, X86InstInfo{"[INV]", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x2F, 1, X86InstInfo{"[INV]", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x37, 1, X86InstInfo{"[INV]", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x3F, 1, X86InstInfo{"[INV]", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
// REX
|
||||
{0x40, 16, X86InstInfo{"", TYPE_REX_PREFIX, FLAGS_NONE, 0, nullptr}},
|
||||
{0x60, 2, X86InstInfo{"[INV]", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0x63, 1, X86InstInfo{"MOVSXD", TYPE_INST, GenFlagsDstSize(SIZE_64BIT) | FLAGS_MODRM, 0, nullptr}},
|
||||
{0x9A, 1, X86InstInfo{"[INV]", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0xA0, 1, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_MEM_OFFSET, 8, nullptr}},
|
||||
{0xA2, 1, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_SRC_RAX | FLAGS_MEM_OFFSET, 8, nullptr}},
|
||||
{0xA1, 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_MEM_OFFSET, 8, nullptr}},
|
||||
{0xA3, 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_SF_SRC_RAX | FLAGS_MEM_OFFSET, 8, nullptr}},
|
||||
{0xCE, 1, X86InstInfo{"[INV]", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0xD4, 2, X86InstInfo{"[INV]", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
// `L1OM` Larrabee instructions used this as an escape byte.
|
||||
// FEX will never support this.
|
||||
{0xD6, 1, X86InstInfo{"[INV]", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{0xEA, 1, X86InstInfo{"[INV]", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
};
|
||||
|
||||
static constexpr U8U8InfoStruct BaseOpTable_32[] = {
|
||||
{0x06, 1, X86InstInfo{"PUSH ES", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0x07, 1, X86InstInfo{"POP ES", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_DEF) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0x0E, 1, X86InstInfo{"PUSH CS", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0x16, 1, X86InstInfo{"PUSH SS", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0x17, 1, X86InstInfo{"POP SS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_DEF) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0x1E, 1, X86InstInfo{"PUSH DS", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0x1F, 1, X86InstInfo{"POP DS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_DEF) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0x27, 1, X86InstInfo{"DAA", TYPE_INST, GenFlagsDstSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 0, nullptr}},
|
||||
{0x2F, 1, X86InstInfo{"DAS", TYPE_INST, GenFlagsDstSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 0, nullptr}},
|
||||
{0x37, 1, X86InstInfo{"AAA", TYPE_INST, GenFlagsDstSize(SIZE_16BIT) | FLAGS_SF_DST_RAX, 0, nullptr}},
|
||||
{0x3F, 1, X86InstInfo{"AAS", TYPE_INST, GenFlagsDstSize(SIZE_16BIT) | FLAGS_SF_DST_RAX, 0, nullptr}},
|
||||
|
||||
{0x40, 8, X86InstInfo{"INC", TYPE_INST, FLAGS_SF_REX_IN_BYTE, 0, nullptr}},
|
||||
{0x48, 8, X86InstInfo{"DEC", TYPE_INST, FLAGS_SF_REX_IN_BYTE, 0, nullptr}},
|
||||
{0x60, 1, X86InstInfo{"PUSHA", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0x61, 1, X86InstInfo{"POPA", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0x63, 1, X86InstInfo{"ARPL", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0x9A, 1, X86InstInfo{"CALLF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xA0, 1, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_MEM_OFFSET, 4, nullptr}},
|
||||
{0xA2, 1, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_SRC_RAX | FLAGS_MEM_OFFSET, 4, nullptr}},
|
||||
{0xA1, 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_MEM_OFFSET, 4, nullptr}},
|
||||
{0xA3, 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_SF_SRC_RAX | FLAGS_MEM_OFFSET, 4, nullptr}},
|
||||
{0xCE, 1, X86InstInfo{"INTO", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xD4, 1, X86InstInfo{"AAM", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 1, nullptr}},
|
||||
{0xD5, 1, X86InstInfo{"AAD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 1, nullptr}},
|
||||
{0xD6, 1, X86InstInfo{"SALC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX | FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{0xEA, 1, X86InstInfo{"JMPF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
};
|
||||
|
||||
if (Mode == Context::MODE_64BIT) {
|
||||
GenerateTable(&BaseOps.at(0), BaseOpTable_64, std::size(BaseOpTable_64));
|
||||
IR::InstallToTable(BaseOps, IR::OpDispatch_BaseOpTable_64);
|
||||
}
|
||||
else {
|
||||
GenerateTable(&BaseOps.at(0), BaseOpTable_32, std::size(BaseOpTable_32));
|
||||
IR::InstallToTable(BaseOps, IR::OpDispatch_BaseOpTable_32);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -13,48 +13,48 @@ $end_info$
|
||||
namespace FEXCore::X86Tables {
|
||||
using namespace InstFlags;
|
||||
|
||||
std::array<X86InstInfo, MAX_3DNOW_TABLE_SIZE> DDDNowOps = []() consteval {
|
||||
constexpr std::array<X86InstInfo, MAX_3DNOW_TABLE_SIZE> DDDNowOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_3DNOW_TABLE_SIZE> Table{};
|
||||
constexpr U8U8InfoStruct DDDNowOpTable[] = {
|
||||
{0x0C, 1, X86InstInfo{"PI2FW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0x0D, 1, X86InstInfo{"PI2FD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0x1C, 1, X86InstInfo{"PF2IW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0x1D, 1, X86InstInfo{"PF2ID", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0x0C, 1, X86InstInfo{"PI2FW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{0x0D, 1, X86InstInfo{"PI2FD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{0x1C, 1, X86InstInfo{"PF2IW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{0x1D, 1, X86InstInfo{"PF2ID", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
|
||||
// Inverse 3DNow! These two instructions are Geode product line specific
|
||||
// No CPUID for these, you're expected to read ID_CONFIG_MSR (1250h) bit 1
|
||||
{0x86, 1, X86InstInfo{"PFRCPV", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0x87, 1, X86InstInfo{"PFRSQRTV", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0x86, 1, X86InstInfo{"PFRCPV", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{0x87, 1, X86InstInfo{"PFRSQRTV", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
|
||||
{0x8A, 1, X86InstInfo{"PFNACC", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0x8E, 1, X86InstInfo{"PFPNACC", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0x8A, 1, X86InstInfo{"PFNACC", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{0x8E, 1, X86InstInfo{"PFPNACC", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
|
||||
{0x90, 1, X86InstInfo{"PFCMPGE", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0x94, 1, X86InstInfo{"PFMIN", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0x96, 1, X86InstInfo{"PFRCP", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0x97, 1, X86InstInfo{"PFRSQRT", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0x90, 1, X86InstInfo{"PFCMPGE", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{0x94, 1, X86InstInfo{"PFMIN", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{0x96, 1, X86InstInfo{"PFRCP", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{0x97, 1, X86InstInfo{"PFRSQRT", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
|
||||
{0x9A, 1, X86InstInfo{"PFSUB", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0x9E, 1, X86InstInfo{"PFADD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0x9A, 1, X86InstInfo{"PFSUB", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{0x9E, 1, X86InstInfo{"PFADD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
|
||||
{0xA0, 1, X86InstInfo{"PFCMPGT", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0xA4, 1, X86InstInfo{"PFMAX", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0xA6, 1, X86InstInfo{"PFRCPIT1", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0xA7, 1, X86InstInfo{"PFRSQIT1", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0xA0, 1, X86InstInfo{"PFCMPGT", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{0xA4, 1, X86InstInfo{"PFMAX", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{0xA6, 1, X86InstInfo{"PFRCPIT1", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{0xA7, 1, X86InstInfo{"PFRSQIT1", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
|
||||
{0xAA, 1, X86InstInfo{"PFSUBR", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0xAE, 1, X86InstInfo{"PFACC", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0xAA, 1, X86InstInfo{"PFSUBR", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{0xAE, 1, X86InstInfo{"PFACC", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
|
||||
{0xB0, 1, X86InstInfo{"PFCMPEQ", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0xB4, 1, X86InstInfo{"PFMUL", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0xB6, 1, X86InstInfo{"PFRCPIT2", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0xB7, 1, X86InstInfo{"PMULHRW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0xB0, 1, X86InstInfo{"PFCMPEQ", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{0xB4, 1, X86InstInfo{"PFMUL", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{0xB6, 1, X86InstInfo{"PFRCPIT2", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{0xB7, 1, X86InstInfo{"PMULHRW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
|
||||
{0xBB, 1, X86InstInfo{"PSWAPD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0xBF, 1, X86InstInfo{"PAVGUSB", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{0xBB, 1, X86InstInfo{"PSWAPD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{0xBF, 1, X86InstInfo{"PAVGUSB", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
};
|
||||
|
||||
GenerateTable(&Table.at(0), DDDNowOpTable, std::size(DDDNowOpTable));
|
||||
GenerateTable(Table.data(), DDDNowOpTable, std::size(DDDNowOpTable));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_DDDTable);
|
||||
return Table;
|
||||
|
||||
@@ -13,7 +13,7 @@ $end_info$
|
||||
|
||||
namespace FEXCore::X86Tables {
|
||||
using namespace InstFlags;
|
||||
std::array<X86InstInfo, MAX_0F_38_TABLE_SIZE> H0F38TableOps = []() consteval {
|
||||
constexpr std::array<X86InstInfo, MAX_0F_38_TABLE_SIZE> H0F38TableOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_0F_38_TABLE_SIZE> Table{};
|
||||
|
||||
#define OPD(prefix, opcode) (((prefix) << 8) | opcode)
|
||||
@@ -23,103 +23,103 @@ std::array<X86InstInfo, MAX_0F_38_TABLE_SIZE> H0F38TableOps = []() consteval {
|
||||
constexpr uint16_t PF_38_F3 = (1U << 2);
|
||||
|
||||
constexpr U16U8InfoStruct H0F38Table[] = {
|
||||
{OPD(PF_38_NONE, 0x00), 1, X86InstInfo{"PSHUFB", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x00), 1, X86InstInfo{"PSHUFB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0x01), 1, X86InstInfo{"PHADDW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x01), 1, X86InstInfo{"PHADDW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0x02), 1, X86InstInfo{"PHADDD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x02), 1, X86InstInfo{"PHADDD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0x03), 1, X86InstInfo{"PHADDSW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x03), 1, X86InstInfo{"PHADDSW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0x04), 1, X86InstInfo{"PMADDUBSW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x04), 1, X86InstInfo{"PMADDUBSW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0x05), 1, X86InstInfo{"PHSUBW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x05), 1, X86InstInfo{"PHSUBW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0x06), 1, X86InstInfo{"PHSUBD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x06), 1, X86InstInfo{"PHSUBD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0x07), 1, X86InstInfo{"PHSUBSW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x07), 1, X86InstInfo{"PHSUBSW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0x08), 1, X86InstInfo{"PSIGNB", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x08), 1, X86InstInfo{"PSIGNB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0x09), 1, X86InstInfo{"PSIGNW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x09), 1, X86InstInfo{"PSIGNW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0x0A), 1, X86InstInfo{"PSIGND", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x0A), 1, X86InstInfo{"PSIGND", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0x0B), 1, X86InstInfo{"PMULHRSW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x0B), 1, X86InstInfo{"PMULHRSW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0x00), 1, X86InstInfo{"PSHUFB", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{OPD(PF_38_66, 0x00), 1, X86InstInfo{"PSHUFB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_NONE, 0x01), 1, X86InstInfo{"PHADDW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{OPD(PF_38_66, 0x01), 1, X86InstInfo{"PHADDW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_NONE, 0x02), 1, X86InstInfo{"PHADDD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{OPD(PF_38_66, 0x02), 1, X86InstInfo{"PHADDD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_NONE, 0x03), 1, X86InstInfo{"PHADDSW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{OPD(PF_38_66, 0x03), 1, X86InstInfo{"PHADDSW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_NONE, 0x04), 1, X86InstInfo{"PMADDUBSW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{OPD(PF_38_66, 0x04), 1, X86InstInfo{"PMADDUBSW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_NONE, 0x05), 1, X86InstInfo{"PHSUBW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{OPD(PF_38_66, 0x05), 1, X86InstInfo{"PHSUBW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_NONE, 0x06), 1, X86InstInfo{"PHSUBD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{OPD(PF_38_66, 0x06), 1, X86InstInfo{"PHSUBD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_NONE, 0x07), 1, X86InstInfo{"PHSUBSW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{OPD(PF_38_66, 0x07), 1, X86InstInfo{"PHSUBSW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_NONE, 0x08), 1, X86InstInfo{"PSIGNB", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{OPD(PF_38_66, 0x08), 1, X86InstInfo{"PSIGNB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_NONE, 0x09), 1, X86InstInfo{"PSIGNW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{OPD(PF_38_66, 0x09), 1, X86InstInfo{"PSIGNW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_NONE, 0x0A), 1, X86InstInfo{"PSIGND", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{OPD(PF_38_66, 0x0A), 1, X86InstInfo{"PSIGND", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_NONE, 0x0B), 1, X86InstInfo{"PMULHRSW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{OPD(PF_38_66, 0x0B), 1, X86InstInfo{"PMULHRSW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
|
||||
{OPD(PF_38_66, 0x10), 1, X86InstInfo{"PBLENDVB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x14), 1, X86InstInfo{"BLENDVPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x15), 1, X86InstInfo{"BLENDVPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x17), 1, X86InstInfo{"PTEST", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0x1C), 1, X86InstInfo{"PABSB", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x1C), 1, X86InstInfo{"PABSB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0x1D), 1, X86InstInfo{"PABSW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x1D), 1, X86InstInfo{"PABSW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0x1E), 1, X86InstInfo{"PABSD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x1E), 1, X86InstInfo{"PABSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x10), 1, X86InstInfo{"PBLENDVB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0x14), 1, X86InstInfo{"BLENDVPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0x15), 1, X86InstInfo{"BLENDVPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0x17), 1, X86InstInfo{"PTEST", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_NONE, 0x1C), 1, X86InstInfo{"PABSB", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{OPD(PF_38_66, 0x1C), 1, X86InstInfo{"PABSB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_NONE, 0x1D), 1, X86InstInfo{"PABSW", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{OPD(PF_38_66, 0x1D), 1, X86InstInfo{"PABSW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_NONE, 0x1E), 1, X86InstInfo{"PABSD", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 0}},
|
||||
{OPD(PF_38_66, 0x1E), 1, X86InstInfo{"PABSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
|
||||
{OPD(PF_38_66, 0x20), 1, X86InstInfo{"PMOVSXBW", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x21), 1, X86InstInfo{"PMOVSXBD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x22), 1, X86InstInfo{"PMOVSXBQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_16BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x23), 1, X86InstInfo{"PMOVSXWD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x24), 1, X86InstInfo{"PMOVSXWQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x25), 1, X86InstInfo{"PMOVSXDQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x28), 1, X86InstInfo{"PMULDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x29), 1, X86InstInfo{"PCMPEQQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x2A), 1, X86InstInfo{"MOVNTDQA", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x2B), 1, X86InstInfo{"PACKUSDW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x20), 1, X86InstInfo{"PMOVSXBW", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0x21), 1, X86InstInfo{"PMOVSXBD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0x22), 1, X86InstInfo{"PMOVSXBQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_16BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0x23), 1, X86InstInfo{"PMOVSXWD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0x24), 1, X86InstInfo{"PMOVSXWQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0x25), 1, X86InstInfo{"PMOVSXDQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0x28), 1, X86InstInfo{"PMULDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0x29), 1, X86InstInfo{"PCMPEQQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0x2A), 1, X86InstInfo{"MOVNTDQA", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0x2B), 1, X86InstInfo{"PACKUSDW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
|
||||
{OPD(PF_38_66, 0x30), 1, X86InstInfo{"PMOVZXBW", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x31), 1, X86InstInfo{"PMOVZXBD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x32), 1, X86InstInfo{"PMOVZXBQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_16BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x33), 1, X86InstInfo{"PMOVZXWD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x34), 1, X86InstInfo{"PMOVZXWQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x35), 1, X86InstInfo{"PMOVZXDQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x37), 1, X86InstInfo{"PCMPGTQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x38), 1, X86InstInfo{"PMINSB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x39), 1, X86InstInfo{"PMINSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x3A), 1, X86InstInfo{"PMINUW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x3B), 1, X86InstInfo{"PMINUD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x3C), 1, X86InstInfo{"PMAXSB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x3D), 1, X86InstInfo{"PMAXSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x3E), 1, X86InstInfo{"PMAXUW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x3F), 1, X86InstInfo{"PMAXUD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x30), 1, X86InstInfo{"PMOVZXBW", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0x31), 1, X86InstInfo{"PMOVZXBD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0x32), 1, X86InstInfo{"PMOVZXBQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_16BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0x33), 1, X86InstInfo{"PMOVZXWD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0x34), 1, X86InstInfo{"PMOVZXWQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0x35), 1, X86InstInfo{"PMOVZXDQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0x37), 1, X86InstInfo{"PCMPGTQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0x38), 1, X86InstInfo{"PMINSB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0x39), 1, X86InstInfo{"PMINSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0x3A), 1, X86InstInfo{"PMINUW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0x3B), 1, X86InstInfo{"PMINUD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0x3C), 1, X86InstInfo{"PMAXSB", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0x3D), 1, X86InstInfo{"PMAXSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0x3E), 1, X86InstInfo{"PMAXUW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0x3F), 1, X86InstInfo{"PMAXUD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
|
||||
{OPD(PF_38_66, 0x40), 1, X86InstInfo{"PMULLD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x41), 1, X86InstInfo{"PHMINPOSUW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0x40), 1, X86InstInfo{"PMULLD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0x41), 1, X86InstInfo{"PHMINPOSUW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
|
||||
{OPD(PF_38_NONE, 0xC8), 1, X86InstInfo{"SHA1NEXTE", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0xC9), 1, X86InstInfo{"SHA1MSG1", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0xCA), 1, X86InstInfo{"SHA1MSG2", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0xC8), 1, X86InstInfo{"SHA1NEXTE", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_NONE, 0xC9), 1, X86InstInfo{"SHA1MSG1", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_NONE, 0xCA), 1, X86InstInfo{"SHA1MSG2", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
|
||||
{OPD(PF_38_NONE, 0xCB), 1, X86InstInfo{"SHA256RNDS2", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0xCC), 1, X86InstInfo{"SHA256MSG1", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0xCD), 1, X86InstInfo{"SHA256MSG2", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0xCB), 1, X86InstInfo{"SHA256RNDS2", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_NONE, 0xCC), 1, X86InstInfo{"SHA256MSG1", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_NONE, 0xCD), 1, X86InstInfo{"SHA256MSG2", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
|
||||
{OPD(PF_38_66, 0xDB), 1, X86InstInfo{"AESIMC", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0xDC), 1, X86InstInfo{"AESENC", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0xDD), 1, X86InstInfo{"AESENCLAST", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0xDE), 1, X86InstInfo{"AESDEC", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0xDF), 1, X86InstInfo{"AESDECLAST", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0xDB), 1, X86InstInfo{"AESIMC", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0xDC), 1, X86InstInfo{"AESENC", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0xDD), 1, X86InstInfo{"AESENCLAST", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0xDE), 1, X86InstInfo{"AESDEC", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(PF_38_66, 0xDF), 1, X86InstInfo{"AESDECLAST", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
|
||||
{OPD(PF_38_NONE, 0xF0), 1, X86InstInfo{"MOVBE", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0xF1), 1, X86InstInfo{"MOVBE", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0, nullptr}},
|
||||
{OPD(PF_38_NONE, 0xF0), 1, X86InstInfo{"MOVBE", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY, 0}},
|
||||
{OPD(PF_38_NONE, 0xF1), 1, X86InstInfo{"MOVBE", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0}},
|
||||
|
||||
{OPD(PF_38_66, 0xF0), 1, X86InstInfo{"MOVBE", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0xF1), 1, X86InstInfo{"MOVBE", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0xF0), 1, X86InstInfo{"MOVBE", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY, 0}},
|
||||
{OPD(PF_38_66, 0xF1), 1, X86InstInfo{"MOVBE", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0}},
|
||||
|
||||
{OPD(PF_38_F2, 0xF0), 1, X86InstInfo{"CRC32", TYPE_INST, GenFlagsSizes(SIZE_DEF, SIZE_8BIT) | FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(PF_38_F2, 0xF1), 1, X86InstInfo{"CRC32", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(PF_38_66 | PF_38_F2, 0xF0), 1, X86InstInfo{"CRC32", TYPE_INST, GenFlagsSizes(SIZE_DEF, SIZE_8BIT) | FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(PF_38_66 | PF_38_F2, 0xF1), 1, X86InstInfo{"CRC32", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(PF_38_F2, 0xF0), 1, X86InstInfo{"CRC32", TYPE_INST, GenFlagsSizes(SIZE_DEF, SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{OPD(PF_38_F2, 0xF1), 1, X86InstInfo{"CRC32", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{OPD(PF_38_66 | PF_38_F2, 0xF0), 1, X86InstInfo{"CRC32", TYPE_INST, GenFlagsSizes(SIZE_DEF, SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{OPD(PF_38_66 | PF_38_F2, 0xF1), 1, X86InstInfo{"CRC32", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
|
||||
{OPD(PF_38_66, 0xF6), 1, X86InstInfo{"ADCX", TYPE_INST, FLAGS_MODRM | FLAGS_NO_OVERLAY66, 0, nullptr}},
|
||||
{OPD(PF_38_F3, 0xF6), 1, X86InstInfo{"ADOX", TYPE_INST, FLAGS_MODRM, 0, nullptr}},
|
||||
{OPD(PF_38_66, 0xF6), 1, X86InstInfo{"ADCX", TYPE_INST, FLAGS_MODRM | FLAGS_NO_OVERLAY66, 0}},
|
||||
{OPD(PF_38_F3, 0xF6), 1, X86InstInfo{"ADOX", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
};
|
||||
#undef OPD
|
||||
|
||||
GenerateTable(&Table.at(0), H0F38Table, std::size(H0F38Table));
|
||||
GenerateTable(Table.data(), H0F38Table, std::size(H0F38Table));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_H0F38Table);
|
||||
return Table;
|
||||
|
||||
@@ -19,71 +19,82 @@ using namespace InstFlags;
|
||||
constexpr uint16_t PF_3A_NONE = 0;
|
||||
constexpr uint16_t PF_3A_66 = 1;
|
||||
|
||||
std::array<X86InstInfo, MAX_0F_3A_TABLE_SIZE> H0F3ATableOps = []() consteval {
|
||||
enum H0F3A_LUT {
|
||||
ENTRY_1_3A_66_16,
|
||||
ENTRY_1_3A_66_22,
|
||||
ENTRY_MAX,
|
||||
};
|
||||
|
||||
constexpr std::array<X86InstInfo[2], ENTRY_MAX> H0F3A_ArchSelect_LUT = {{
|
||||
// ENTRY_1_3A_66_16
|
||||
{
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
{"PEXTRQ", TYPE_INST, GenFlagsSizes(SIZE_64BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::PExtrOp, IR::OpSize::i64Bit> }},
|
||||
},
|
||||
// ENTRY_1_3A_66_22
|
||||
{
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
{"PINSRQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1, { .OpDispatch = &IR::OpDispatchBuilder::PINSROp<IR::OpSize::i64Bit> }},
|
||||
},
|
||||
}};
|
||||
|
||||
constexpr std::array<X86InstInfo, MAX_0F_3A_TABLE_SIZE> H0F3ATableOps = []() consteval {
|
||||
std::array<X86InstInfo, MAX_0F_3A_TABLE_SIZE> Table{};
|
||||
auto TableGen = []<uint16_t REX>() consteval {
|
||||
constexpr U16U8InfoStruct Table[] = {
|
||||
{OPD(REX, PF_3A_NONE, 0x0F), 1, X86InstInfo{"PALIGNR", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x08), 1, X86InstInfo{"ROUNDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x09), 1, X86InstInfo{"ROUNDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x0A), 1, X86InstInfo{"ROUNDSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x0B), 1, X86InstInfo{"ROUNDSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x0C), 1, X86InstInfo{"BLENDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x0D), 1, X86InstInfo{"BLENDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x0E), 1, X86InstInfo{"PBLENDW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x0F), 1, X86InstInfo{"PALIGNR", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_NONE, 0x0F), 1, X86InstInfo{"PALIGNR", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX, 1}},
|
||||
{OPD(REX, PF_3A_66, 0x08), 1, X86InstInfo{"ROUNDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1}},
|
||||
{OPD(REX, PF_3A_66, 0x09), 1, X86InstInfo{"ROUNDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1}},
|
||||
{OPD(REX, PF_3A_66, 0x0A), 1, X86InstInfo{"ROUNDSS", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1}},
|
||||
{OPD(REX, PF_3A_66, 0x0B), 1, X86InstInfo{"ROUNDSD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1}},
|
||||
{OPD(REX, PF_3A_66, 0x0C), 1, X86InstInfo{"BLENDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1}},
|
||||
{OPD(REX, PF_3A_66, 0x0D), 1, X86InstInfo{"BLENDPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1}},
|
||||
{OPD(REX, PF_3A_66, 0x0E), 1, X86InstInfo{"PBLENDW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1}},
|
||||
{OPD(REX, PF_3A_66, 0x0F), 1, X86InstInfo{"PALIGNR", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1}},
|
||||
|
||||
{OPD(REX, PF_3A_66, 0x14), 1, X86InstInfo{"PEXTRB", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x15), 1, X86InstInfo{"PEXTRW", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x17), 1, X86InstInfo{"EXTRACTPS", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x14), 1, X86InstInfo{"PEXTRB", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1}},
|
||||
{OPD(REX, PF_3A_66, 0x15), 1, X86InstInfo{"PEXTRW", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1}},
|
||||
{OPD(REX, PF_3A_66, 0x17), 1, X86InstInfo{"EXTRACTPS", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1}},
|
||||
|
||||
{OPD(REX, PF_3A_66, 0x20), 1, X86InstInfo{"PINSRB", TYPE_INST, GenFlagsDstSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x21), 1, X86InstInfo{"INSERTPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x40), 1, X86InstInfo{"DPPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x41), 1, X86InstInfo{"DPPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x42), 1, X86InstInfo{"MPSADBW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x44), 1, X86InstInfo{"PCLMULQDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x20), 1, X86InstInfo{"PINSRB", TYPE_INST, GenFlagsDstSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1}},
|
||||
{OPD(REX, PF_3A_66, 0x21), 1, X86InstInfo{"INSERTPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1}},
|
||||
{OPD(REX, PF_3A_66, 0x40), 1, X86InstInfo{"DPPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1}},
|
||||
{OPD(REX, PF_3A_66, 0x41), 1, X86InstInfo{"DPPD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1}},
|
||||
{OPD(REX, PF_3A_66, 0x42), 1, X86InstInfo{"MPSADBW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1}},
|
||||
{OPD(REX, PF_3A_66, 0x44), 1, X86InstInfo{"PCLMULQDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1}},
|
||||
|
||||
{OPD(REX, PF_3A_66, 0x60), 1, X86InstInfo{"PCMPESTRM", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x61), 1, X86InstInfo{"PCMPESTRI", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x62), 1, X86InstInfo{"PCMPISTRM", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x63), 1, X86InstInfo{"PCMPISTRI", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0x60), 1, X86InstInfo{"PCMPESTRM", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1}},
|
||||
{OPD(REX, PF_3A_66, 0x61), 1, X86InstInfo{"PCMPESTRI", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1}},
|
||||
{OPD(REX, PF_3A_66, 0x62), 1, X86InstInfo{"PCMPISTRM", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1}},
|
||||
{OPD(REX, PF_3A_66, 0x63), 1, X86InstInfo{"PCMPISTRI", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1}},
|
||||
|
||||
{OPD(REX, PF_3A_NONE, 0xCC), 1, X86InstInfo{"SHA1RNDS4", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_NONE, 0xCC), 1, X86InstInfo{"SHA1RNDS4", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1}},
|
||||
|
||||
{OPD(REX, PF_3A_66, 0xDF), 1, X86InstInfo{"AESKEYGENASSIST", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(REX, PF_3A_66, 0xDF), 1, X86InstInfo{"AESKEYGENASSIST", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1}},
|
||||
};
|
||||
return std::to_array(Table);
|
||||
};
|
||||
constexpr auto H0F3ATable_IgnoresREX0 = TableGen.template operator()<0>();
|
||||
constexpr auto H0F3ATable_IgnoresREX1 = TableGen.template operator()<1>();
|
||||
|
||||
GenerateTable(&Table.at(0), &H0F3ATable_IgnoresREX0.at(0), H0F3ATable_IgnoresREX0.size());
|
||||
GenerateTable(&Table.at(0), &H0F3ATable_IgnoresREX1.at(0), H0F3ATable_IgnoresREX1.size());
|
||||
GenerateTable(Table.data(), H0F3ATable_IgnoresREX0.data(), H0F3ATable_IgnoresREX0.size());
|
||||
GenerateTable(Table.data(), H0F3ATable_IgnoresREX1.data(), H0F3ATable_IgnoresREX1.size());
|
||||
|
||||
constexpr U16U8InfoStruct TableNeedsREX[] = {
|
||||
{OPD(0, PF_3A_66, 0x16), 1, X86InstInfo{"PEXTRD", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x22), 1, X86InstInfo{"PINSRD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1, nullptr}},
|
||||
constexpr U16U8InfoStruct TableNeedsREX0[] = {
|
||||
{OPD(0, PF_3A_66, 0x16), 1, X86InstInfo{"PEXTRD", TYPE_INST, GenFlagsSizes(SIZE_32BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1}},
|
||||
{OPD(0, PF_3A_66, 0x22), 1, X86InstInfo{"PINSRD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1}},
|
||||
};
|
||||
GenerateTable(&Table.at(0), TableNeedsREX, std::size(TableNeedsREX));
|
||||
GenerateTable(Table.data(), TableNeedsREX0, std::size(TableNeedsREX0));
|
||||
|
||||
constexpr U16U8InfoStruct TableNeedsREX1[] = {
|
||||
{OPD(1, PF_3A_66, 0x16), 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = H0F3A_ArchSelect_LUT[ENTRY_1_3A_66_16] }}},
|
||||
{OPD(1, PF_3A_66, 0x22), 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = H0F3A_ArchSelect_LUT[ENTRY_1_3A_66_22] }}},
|
||||
};
|
||||
GenerateTable(Table.data(), TableNeedsREX1, std::size(TableNeedsREX1));
|
||||
|
||||
IR::InstallToTable(Table, IR::OpDispatch_H0F3ATableIgnoreREX);
|
||||
IR::InstallToTable(Table, IR::OpDispatch_H0F3ATableNeedsREX0);
|
||||
|
||||
return Table;
|
||||
}();
|
||||
|
||||
void InitializeH0F3ATables(Context::OperatingMode Mode) {
|
||||
static constexpr U16U8InfoStruct H0F3ATable_64[] = {
|
||||
{OPD(1, PF_3A_66, 0x16), 1, X86InstInfo{"PEXTRQ", TYPE_INST, GenFlagsSizes(SIZE_64BIT, SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_DST_GPR | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(1, PF_3A_66, 0x22), 1, X86InstInfo{"PINSRQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1, nullptr}},
|
||||
};
|
||||
|
||||
#undef OPD
|
||||
|
||||
if (Mode == Context::MODE_64BIT) {
|
||||
GenerateTable(&H0F3ATableOps.at(0), H0F3ATable_64, std::size(H0F3ATable_64));
|
||||
IR::InstallToTable(H0F3ATableOps, IR::OpDispatch_H0F3ATable_64);
|
||||
}
|
||||
}
|
||||
}
|
||||
Loaded 100 of 565 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user