mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 23:00:17 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
1cc4b93e7a | ||
|
|
b4e2f5118a | ||
|
|
812b6398e5 | ||
|
|
1db45e2a70 | ||
|
|
6bcadde658 | ||
|
|
5f1c8efe0e | ||
|
|
b9aeccf13b | ||
|
|
16f90b33f3 | ||
|
|
5d8d052a77 | ||
|
|
7fa4d78269 | ||
|
|
6bf0db7df6 | ||
|
|
110313e7de | ||
|
|
44e24c9e6b | ||
|
|
6c58fef220 | ||
|
|
3d593ce87d | ||
|
|
36e1b5107e | ||
|
|
f89123f489 | ||
|
|
b7280a765d | ||
|
|
37b010795e | ||
|
|
a4f89b79a3 | ||
|
|
5bde4d875a | ||
|
|
a79c471c31 | ||
|
|
c1e29f9013 | ||
|
|
4c27dfd5eb | ||
|
|
201216ba54 | ||
|
|
d3a85e14d9 | ||
|
|
417bd8604c | ||
|
|
3d289f4489 | ||
|
|
d138c854f3 | ||
|
|
e26a792b70 | ||
|
|
7e2d3b07c0 | ||
|
|
394a6f28db | ||
|
|
72e01274c9 | ||
|
|
9f2e982944 | ||
|
|
c6b0f360fe | ||
|
|
ef19242be3 | ||
|
|
2cb4f8b6f5 | ||
|
|
59919a0b0c | ||
|
|
bf1857ecc7 | ||
|
|
b40920596a | ||
|
|
3b14c322e8 | ||
|
|
3e60aa5738 | ||
|
|
6ca2d27c82 | ||
|
|
d2f26d1969 | ||
|
|
fc25443827 | ||
|
|
986800885a | ||
|
|
52c6ab1cec | ||
|
|
83a989c6dc | ||
|
|
32b96c259b | ||
|
|
954581c750 | ||
|
|
24da43f823 | ||
|
|
600e2ddecf | ||
|
|
9740f488cf | ||
|
|
f81467fd30 | ||
|
|
2ec2c39cf1 | ||
|
|
c85e426cc5 | ||
|
|
41fc57f46c | ||
|
|
a215bb9709 | ||
|
|
b23fa30099 | ||
|
|
848c4b2d68 | ||
|
|
4f995cbc1c | ||
|
|
b21c49352e | ||
|
|
3470dd1e7b | ||
|
|
2ba35286ef | ||
|
|
bbd8212ced | ||
|
|
4564325bc7 | ||
|
|
b9052ed7f0 | ||
|
|
4160a92621 | ||
|
|
201bb73980 | ||
|
|
78832cc0d0 | ||
|
|
a5ecb71993 | ||
|
|
6d86dca20b | ||
|
|
e24f84504f | ||
|
|
ee2fb57f4e | ||
|
|
11fe95d8ea | ||
|
|
70fe9a4405 | ||
|
|
c09225f868 | ||
|
|
126bcd365d | ||
|
|
fe4d2bc6c5 | ||
|
|
dbaf22372c | ||
|
|
43bd243457 | ||
|
|
c0251dc8be | ||
|
|
26266c6a94 | ||
|
|
64392b2d45 | ||
|
|
d555ee8bcc | ||
|
|
0c1a35f297 | ||
|
|
9ac608ca43 | ||
|
|
7ae55d73c1 | ||
|
|
8102a0974a | ||
|
|
f8491794d3 | ||
|
|
d5be15c90e | ||
|
|
e91efc6694 | ||
|
|
3d66be9e5a | ||
|
|
ec1b24d05b | ||
|
|
d93997c1cb | ||
|
|
e19aa975c8 | ||
|
|
3f02dd0a36 | ||
|
|
ab4e0f653a | ||
|
|
fda023e7dc | ||
|
|
ad618be979 | ||
|
|
5881266256 | ||
|
|
e03187852b | ||
|
|
b846cb6c2c | ||
|
|
62ecdd650f | ||
|
|
9f195ff377 | ||
|
|
27a5f09185 | ||
|
|
01b0b4e653 | ||
|
|
ff5dfff5bb | ||
|
|
e3e9777ee6 | ||
|
|
7dc2dc8749 | ||
|
|
4eb5694872 | ||
|
|
681c5e8097 | ||
|
|
5ad06b0255 | ||
|
|
a2889a09e3 | ||
|
|
a4cd5f7584 | ||
|
|
cf098a0de6 | ||
|
|
1619374252 | ||
|
|
c13064e201 | ||
|
|
20647f2287 | ||
|
|
37e32fbcb9 | ||
|
|
27acbba52e | ||
|
|
6f33d2b4c4 | ||
|
|
a2e4209f4f | ||
|
|
73d6716828 | ||
|
|
1fb2419be2 | ||
|
|
3394808c06 | ||
|
|
c252b58a15 | ||
|
|
39ae8c3ea0 | ||
|
|
eb8c2d964c | ||
|
|
bd9cf9ca11 | ||
|
|
280568df2f | ||
|
|
b87ff1e2dc | ||
|
|
55c90cfc38 | ||
|
|
500d2374a5 | ||
|
|
d78963c021 | ||
|
|
9ab0920f01 | ||
|
|
a6e7fba433 | ||
|
|
8905e39439 | ||
|
|
df41b85827 | ||
|
|
9fa3b9345e | ||
|
|
42af6c8508 | ||
|
|
b7df1bc259 | ||
|
|
b40f9db735 | ||
|
|
938841c658 | ||
|
|
34344e5769 | ||
|
|
538a9624ec | ||
|
|
c4c69ca8de | ||
|
|
90330ab3e4 | ||
|
|
c5880e7618 | ||
|
|
9d0c05d9cc | ||
|
|
f6a68cb7fd | ||
|
|
462c785418 | ||
|
|
a1f90dd8d3 | ||
|
|
2a67261eac | ||
|
|
1d3403fdc2 | ||
|
|
53301b0f56 | ||
|
|
8989ce1766 | ||
|
|
faa121e9ef | ||
|
|
f48759e83a | ||
|
|
2eca733603 | ||
|
|
14b65cec43 | ||
|
|
f5477039fa | ||
|
|
9fa3221687 | ||
|
|
c8c63faf15 | ||
|
|
ee4794c99e | ||
|
|
3a23bb4b73 | ||
|
|
46ffb25f84 | ||
|
|
069a4025c5 | ||
|
|
edd044752d | ||
|
|
3929d25dcc | ||
|
|
99baa4f3d9 | ||
|
|
f64d4c571b | ||
|
|
9372fa169a | ||
|
|
f98ac7f268 | ||
|
|
daaa6ec129 | ||
|
|
88afc22d5b | ||
|
|
23099100b6 | ||
|
|
b7ea9e30df | ||
|
|
844c3bb197 | ||
|
|
208c6d3eac | ||
|
|
886a2e74ac | ||
|
|
adad3c27dd | ||
|
|
470aeab215 | ||
|
|
db1d90ec9d | ||
|
|
124ce8420a | ||
|
|
4433eaf242 | ||
|
|
5ba070f600 | ||
|
|
2d1a42aa00 | ||
|
|
74d9f5a3e2 | ||
|
|
634fbb5a73 | ||
|
|
e16948bf80 | ||
|
|
76c9833ee5 | ||
|
|
99662b70ff | ||
|
|
fcde9eabbf | ||
|
|
4ef15951a8 | ||
|
|
1bc51c2290 | ||
|
|
be1025901a | ||
|
|
1a606de29f | ||
|
|
454c0b31cb | ||
|
|
223e0f4e53 | ||
|
|
3a84091945 | ||
|
|
a0e8f1097f | ||
|
|
08ed4fb983 | ||
|
|
0b1f336e03 | ||
|
|
0258fcb116 | ||
|
|
3272aa3f08 | ||
|
|
d9ea6651f8 | ||
|
|
32b11603d8 | ||
|
|
538fd2672d | ||
|
|
3e37724e3e | ||
|
|
1b249ba76b | ||
|
|
df1295fbd0 | ||
|
|
dcd71fe126 | ||
|
|
610ee5db76 | ||
|
|
9d5494d9f0 | ||
|
|
dd44bc8d00 | ||
|
|
110c7cb62b | ||
|
|
09aa5abbdf | ||
|
|
1d8b6df630 | ||
|
|
195058752e | ||
|
|
c62805e86d | ||
|
|
d15b175c33 | ||
|
|
6cb73adfd5 | ||
|
|
6d1cd67900 | ||
|
|
e12bd27106 | ||
|
|
cb018257cf | ||
|
|
e02953dc17 | ||
|
|
ba56f8e0c5 | ||
|
|
ac12dd55c3 | ||
|
|
24647820d7 | ||
|
|
7e8aa711ef | ||
|
|
329e561eff | ||
|
|
d848cbbc0f | ||
|
|
cd46e43c20 | ||
|
|
98674c1cc8 | ||
|
|
8bfae631b1 | ||
|
|
d00c7cf3a3 | ||
|
|
b45665fee4 | ||
|
|
1b58664541 | ||
|
|
ed6a178ae3 | ||
|
|
e925ca509d | ||
|
|
ca310cf815 | ||
|
|
6fa27aac42 | ||
|
|
a5c3fc4751 | ||
|
|
65b05fa8c1 | ||
|
|
3ee556d858 | ||
|
|
fd3546a999 | ||
|
|
f5fafa5b96 | ||
|
|
7ff2069e60 | ||
|
|
154ff43d7f | ||
|
|
b754fe4810 | ||
|
|
a1071ec01a | ||
|
|
92dce9a2ea | ||
|
|
5fd917ec2f | ||
|
|
d7cda23b25 | ||
|
|
cae5da5777 | ||
|
|
97f1f47fa5 | ||
|
|
53c269ee25 | ||
|
|
1420d3cc10 | ||
|
|
ce401b5ca1 | ||
|
|
cd412bd0f5 | ||
|
|
f33f88072e | ||
|
|
1240a00fa5 | ||
|
|
83f325de0d | ||
|
|
0b871bf54e | ||
|
|
07f7aa3c8f | ||
|
|
7208bc6cdd | ||
|
|
fef5a98602 | ||
|
|
5cce65cdfa | ||
|
|
268081e5d0 | ||
|
|
f5f179117e | ||
|
|
8c85096f98 | ||
|
|
c5e7675c4b | ||
|
|
03009912ac | ||
|
|
e4a1138291 | ||
|
|
a5bc54d2d5 | ||
|
|
df73e84725 | ||
|
|
5bf07c2e77 | ||
|
|
bb0d142a65 | ||
|
|
c98cef0da1 | ||
|
|
1d9c52be02 | ||
|
|
a6c9df1a64 | ||
|
|
f66368b191 | ||
|
|
e4daea406e | ||
|
|
6216f22cb9 | ||
|
|
d4c80d9094 | ||
|
|
27324ded87 | ||
|
|
7d1c625e32 | ||
|
|
b4fe65f2c0 | ||
|
|
f5efdac2e5 | ||
|
|
23de875516 | ||
|
|
82030b8286 | ||
|
|
af4da43bb8 | ||
|
|
33f3b8659c | ||
|
|
ed724a61a7 | ||
|
|
053bd74aa3 | ||
|
|
3c0410e59f | ||
|
|
e621f6c753 | ||
|
|
b05f000f42 | ||
|
|
e60bfc6d23 | ||
|
|
9a3d3201f9 | ||
|
|
abf9724424 | ||
|
|
ab9a8c62ab | ||
|
|
e2fe936152 | ||
|
|
1d71650379 | ||
|
|
a040740974 | ||
|
|
5be0dc9fc5 | ||
|
|
52ad434d24 | ||
|
|
d69d111bb6 | ||
|
|
50f4494875 | ||
|
|
8a4982383a | ||
|
|
0d72890482 | ||
|
|
9be7d6d112 | ||
|
|
3c4121ba07 | ||
|
|
b6e44b04d6 | ||
|
|
02c11afbc6 | ||
|
|
2b8f5b57eb | ||
|
|
0519c9467c | ||
|
|
84d968c7d2 | ||
|
|
9a1d06c6ab | ||
|
|
739e85032b |
No files matched your search
@@ -33,3 +33,17 @@ runs:
|
||||
- name: Install
|
||||
shell: bash
|
||||
run: DESTDIR="$PWD"/install cmake --build build_${{ inputs.target }} -t install
|
||||
|
||||
- name: Configure UnixLib
|
||||
shell: bash
|
||||
run: |
|
||||
cmake -S Source/Windows/UnixLib -B build_unixlib_${{ inputs.target }} -DCMAKE_BUILD_TYPE=$BUILD_TYPE \
|
||||
-G Ninja -DCMAKE_INSTALL_LIBDIR=/usr/lib/wine/aarch64-unix -DCMAKE_INSTALL_PREFIX=/usr
|
||||
|
||||
- name: Build UnixLib
|
||||
shell: bash
|
||||
run: cmake --build build_unixlib_${{ inputs.target }}
|
||||
|
||||
- name: Install UnixLib
|
||||
shell: bash
|
||||
run: DESTDIR="$PWD"/install cmake --build build_unixlib_${{ inputs.target }} -t install
|
||||
@@ -50,6 +50,8 @@ jobs:
|
||||
with:
|
||||
overwrite: true
|
||||
name: wine_dll_artifacts
|
||||
path: ${{ github.workspace }}/install/usr/lib/wine/aarch64-windows/lib*.dll
|
||||
path: |
|
||||
${{ github.workspace }}/install/usr/lib/wine/aarch64-windows/lib*.dll
|
||||
${{ github.workspace }}/install/usr/lib/wine/aarch64-unix/lib*.so
|
||||
retention-days: 60
|
||||
compression-level: 9
|
||||
@@ -0,0 +1 @@
|
||||
AI must not be used to generate code for contributions to this project.
|
||||
@@ -0,0 +1 @@
|
||||
AI must not be used to generate code for contributions to this project.
|
||||
+9
-2
@@ -253,8 +253,15 @@ endif()
|
||||
if (ENABLE_CCACHE)
|
||||
find_program(CCACHE_PROGRAM ccache)
|
||||
if(CCACHE_PROGRAM)
|
||||
message(STATUS "CCache enabled")
|
||||
set_property(GLOBAL PROPERTY RULE_LAUNCH_COMPILE "${CCACHE_PROGRAM}")
|
||||
execute_process(COMMAND "${CCACHE_PROGRAM}" --print-version
|
||||
OUTPUT_VARIABLE CCACHE_VERSION OUTPUT_STRIP_TRAILING_WHITESPACE)
|
||||
message(STATUS "Enabling ccache ${CCACHE_VERSION}")
|
||||
if (CCACHE_VERSION VERSION_GREATER_EQUAL "4.8")
|
||||
# Set sloppiness to enable caching even for files that use __DATE__/__TIME__ macros
|
||||
set_property(GLOBAL PROPERTY RULE_LAUNCH_COMPILE "${CCACHE_PROGRAM} sloppiness=time_macros")
|
||||
else()
|
||||
set_property(GLOBAL PROPERTY RULE_LAUNCH_COMPILE "${CCACHE_PROGRAM}")
|
||||
endif()
|
||||
endif()
|
||||
endif()
|
||||
|
||||
|
||||
@@ -1,132 +0,0 @@
|
||||
{
|
||||
"environments": [
|
||||
{
|
||||
"BuildPath": "${projectDir}\\out\\build\\${name}",
|
||||
"InstallPath": "${projectDir}\\out\\install\\${name}",
|
||||
"clangcl": "clang-cl.exe",
|
||||
"cc": "clang",
|
||||
"cxx": "clang++"
|
||||
}
|
||||
],
|
||||
"configurations": [
|
||||
{
|
||||
"name": "WSL-Clang-Debug",
|
||||
"generator": "Ninja",
|
||||
"configurationType": "Debug",
|
||||
"buildRoot": "${env.BuildPath}",
|
||||
"installRoot": "${env.InstallPath}",
|
||||
"cmakeExecutable": "/usr/bin/cmake",
|
||||
"cmakeCommandArgs": "",
|
||||
"buildCommandArgs": "-v",
|
||||
"ctestCommandArgs": "",
|
||||
"wslPath": "${defaultWSLPath}",
|
||||
"inheritEnvironments": [ "linux_clang_x64" ],
|
||||
"addressSanitizerRuntimeFlags": "detect_leaks=0",
|
||||
"variables": [
|
||||
{
|
||||
"name": "WSL",
|
||||
"value": "TRUE",
|
||||
"type": "BOOL"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "WSL-Clang-Release",
|
||||
"generator": "Ninja",
|
||||
"configurationType": "RelWithDebInfo",
|
||||
"buildRoot": "${env.BuildPath}",
|
||||
"installRoot": "${env.InstallPath}",
|
||||
"cmakeExecutable": "/usr/bin/cmake",
|
||||
"cmakeCommandArgs": "",
|
||||
"buildCommandArgs": "-v",
|
||||
"ctestCommandArgs": "",
|
||||
"wslPath": "${defaultWSLPath}",
|
||||
"inheritEnvironments": [ "linux_clang_x64" ],
|
||||
"addressSanitizerRuntimeFlags": "detect_leaks=0",
|
||||
"variables": [
|
||||
{
|
||||
"name": "WSL",
|
||||
"value": "TRUE",
|
||||
"type": "BOOL"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "x86-Clang-Cross-Debug",
|
||||
"generator": "Ninja",
|
||||
"configurationType": "Debug",
|
||||
"buildRoot": "${env.BuildPath}",
|
||||
"installRoot": "${env.InstallPath}",
|
||||
"cmakeCommandArgs": "",
|
||||
"buildCommandArgs": "-v",
|
||||
"ctestCommandArgs": "",
|
||||
"inheritEnvironments": [ "clang_cl_x86" ],
|
||||
"variables": [
|
||||
{
|
||||
"name": "CMAKE_C_COMPILER",
|
||||
"value": "${env.cc}",
|
||||
"type": "STRING"
|
||||
},
|
||||
{
|
||||
"name": "CMAKE_CXX_COMPILER",
|
||||
"value": "${env.cxx}",
|
||||
"type": "STRING"
|
||||
},
|
||||
{
|
||||
"name": "CMAKE_SYSROOT",
|
||||
"value": "${env.fexsysroot}",
|
||||
"type": "STRING"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "x64-Clang-Cross-Release",
|
||||
"generator": "Ninja",
|
||||
"configurationType": "RelWithDebInfo",
|
||||
"buildRoot": "${env.BuildPath}",
|
||||
"installRoot": "${env.InstallPath}",
|
||||
"cmakeCommandArgs": "",
|
||||
"buildCommandArgs": "-v",
|
||||
"ctestCommandArgs": "",
|
||||
"inheritEnvironments": [ "clang_cl_x86" ],
|
||||
"variables": [
|
||||
{
|
||||
"name": "CMAKE_C_COMPILER",
|
||||
"value": "${env.cc}",
|
||||
"type": "STRING"
|
||||
},
|
||||
{
|
||||
"name": "CMAKE_CXX_COMPILER",
|
||||
"value": "${env.cxx}",
|
||||
"type": "STRING"
|
||||
},
|
||||
{
|
||||
"name": "CMAKE_SYSROOT",
|
||||
"value": "${env.fexsysroot}",
|
||||
"type": "STRING"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "Linux-Clang-Remote-Debug",
|
||||
"generator": "Ninja",
|
||||
"configurationType": "Debug",
|
||||
"cmakeExecutable": "/usr/bin/cmake",
|
||||
"remoteCopySourcesExclusionList": [ ".vs", ".vscode", ".git", ".github", "build", "out", "bin" ],
|
||||
"cmakeCommandArgs": "",
|
||||
"buildCommandArgs": "-v",
|
||||
"ctestCommandArgs": "",
|
||||
"inheritEnvironments": [ "linux_clang_x64" ],
|
||||
"remoteMachineName": "${env.fexremote}",
|
||||
"remoteCMakeListsRoot": "$HOME/projects/.vs/${projectDirName}/src",
|
||||
"remoteBuildRoot": "$HOME/projects/.vs/${projectDirName}/build/${name}",
|
||||
"remoteInstallRoot": "$HOME/projects/.vs/${projectDirName}/install/${name}",
|
||||
"remoteCopySources": true,
|
||||
"rsyncCommandArgs": "-t --delete --delete-excluded",
|
||||
"remoteCopyBuildOutput": false,
|
||||
"remoteCopySourcesMethod": "rsync",
|
||||
"addressSanitizerRuntimeFlags": "detect_leaks=0",
|
||||
"variables": []
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1 @@
|
||||
No AI/ML/LLM/etc code contributions.
|
||||
@@ -46,6 +46,13 @@
|
||||
"@PREFIX_LIB@/libwayland-client.so.0",
|
||||
"@PREFIX_LIB@/libwayland-client.so.0.20.0"
|
||||
]
|
||||
},
|
||||
"cuda": {
|
||||
"Library" : "libcuda-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/libcuda.so",
|
||||
"@PREFIX_LIB@/libcuda.so.1"
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
+60
-60
@@ -1,5 +1,5 @@
|
||||
#
|
||||
# This file is autogenerated by pip-compile with Python 3.13
|
||||
# This file is autogenerated by pip-compile with Python 3.14
|
||||
# by the following command:
|
||||
#
|
||||
# pip-compile --generate-hashes --output-file=requirements_formatting.txt --strip-extras requirements_formatting.txt.in
|
||||
@@ -210,56 +210,56 @@ click==8.1.7 \
|
||||
--hash=sha256:ae74fb96c20a0277a1d615f1e4d73c8414f5a98db8b799a7931d1582f3390c28 \
|
||||
--hash=sha256:ca9853ad459e787e2192211578cc907e7594e294c7ccc834310722b41b9ca6de
|
||||
# via black
|
||||
cryptography==46.0.5 \
|
||||
--hash=sha256:02f547fce831f5096c9a567fd41bc12ca8f11df260959ecc7c3202555cc47a72 \
|
||||
--hash=sha256:039917b0dc418bb9f6edce8a906572d69e74bd330b0b3fea4f79dab7f8ddd235 \
|
||||
--hash=sha256:1abfdb89b41c3be0365328a410baa9df3ff8a9110fb75e7b52e66803ddabc9a9 \
|
||||
--hash=sha256:2ae6971afd6246710480e3f15824ed3029a60fc16991db250034efd0b9fb4356 \
|
||||
--hash=sha256:2b7a67c9cd56372f3249b39699f2ad479f6991e62ea15800973b956f4b73e257 \
|
||||
--hash=sha256:351695ada9ea9618b3500b490ad54c739860883df6c1f555e088eaf25b1bbaad \
|
||||
--hash=sha256:38946c54b16c885c72c4f59846be9743d699eee2b69b6988e0a00a01f46a61a4 \
|
||||
--hash=sha256:3b4995dc971c9fb83c25aa44cf45f02ba86f71ee600d81091c2f0cbae116b06c \
|
||||
--hash=sha256:3ce58ba46e1bc2aac4f7d9290223cead56743fa6ab94a5d53292ffaac6a91614 \
|
||||
--hash=sha256:3ee190460e2fbe447175cda91b88b84ae8322a104fc27766ad09428754a618ed \
|
||||
--hash=sha256:4108d4c09fbbf2789d0c926eb4152ae1760d5a2d97612b92d508d96c861e4d31 \
|
||||
--hash=sha256:420d0e909050490d04359e7fdb5ed7e667ca5c3c402b809ae2563d7e66a92229 \
|
||||
--hash=sha256:47fb8a66058b80e509c47118ef8a75d14c455e81ac369050f20ba0d23e77fee0 \
|
||||
--hash=sha256:4c3341037c136030cb46e4b1e17b7418ea4cbd9dd207e4a6f3b2b24e0d4ac731 \
|
||||
--hash=sha256:4d7e3d356b8cd4ea5aff04f129d5f66ebdc7b6f8eae802b93739ed520c47c79b \
|
||||
--hash=sha256:4d8ae8659ab18c65ced284993c2265910f6c9e650189d4e3f68445ef82a810e4 \
|
||||
--hash=sha256:4e817a8920bfbcff8940ecfd60f23d01836408242b30f1a708d93198393a80b4 \
|
||||
--hash=sha256:50bfb6925eff619c9c023b967d5b77a54e04256c4281b0e21336a130cd7fc263 \
|
||||
--hash=sha256:556e106ee01aa13484ce9b0239bca667be5004efb0aabbed28d353df86445595 \
|
||||
--hash=sha256:582f5fcd2afa31622f317f80426a027f30dc792e9c80ffee87b993200ea115f1 \
|
||||
--hash=sha256:5be7bf2fb40769e05739dd0046e7b26f9d4670badc7b032d6ce4db64dddc0678 \
|
||||
--hash=sha256:60ee7e19e95104d4c03871d7d7dfb3d22ef8a9b9c6778c94e1c8fcc8365afd48 \
|
||||
--hash=sha256:61aa400dce22cb001a98014f647dc21cda08f7915ceb95df0c9eaf84b4b6af76 \
|
||||
--hash=sha256:68f68d13f2e1cb95163fa3b4db4bf9a159a418f5f6e7242564fc75fcae667fd0 \
|
||||
--hash=sha256:7d1f30a86d2757199cb2d56e48cce14deddf1f9c95f1ef1b64ee91ea43fe2e18 \
|
||||
--hash=sha256:7d731d4b107030987fd61a7f8ab512b25b53cef8f233a97379ede116f30eb67d \
|
||||
--hash=sha256:803812e111e75d1aa73690d2facc295eaefd4439be1023fefc4995eaea2af90d \
|
||||
--hash=sha256:80a8d7bfdf38f87ca30a5391c0c9ce4ed2926918e017c29ddf643d0ed2778ea1 \
|
||||
--hash=sha256:8293f3dea7fc929ef7240796ba231413afa7b68ce38fd21da2995549f5961981 \
|
||||
--hash=sha256:8456928655f856c6e1533ff59d5be76578a7157224dbd9ce6872f25055ab9ab7 \
|
||||
--hash=sha256:890bcb4abd5a2d3f852196437129eb3667d62630333aacc13dfd470fad3aaa82 \
|
||||
--hash=sha256:94a76daa32eb78d61339aff7952ea819b1734b46f73646a07decb40e5b3448e2 \
|
||||
--hash=sha256:9f16fbdf4da055efb21c22d81b89f155f02ba420558db21288b3d0035bafd5f4 \
|
||||
--hash=sha256:a3d1fae9863299076f05cb8a778c467578262fae09f9dc0ee9b12eb4268ce663 \
|
||||
--hash=sha256:a3d507bb6a513ca96ba84443226af944b0f7f47dcc9a399d110cd6146481d24c \
|
||||
--hash=sha256:abace499247268e3757271b2f1e244b36b06f8515cf27c4d49468fc9eb16e93d \
|
||||
--hash=sha256:ba2a27ff02f48193fc4daeadf8ad2590516fa3d0adeeb34336b96f7fa64c1e3a \
|
||||
--hash=sha256:bc84e875994c3b445871ea7181d424588171efec3e185dced958dad9e001950a \
|
||||
--hash=sha256:bfd56bb4b37ed4f330b82402f6f435845a5f5648edf1ad497da51a8452d5d62d \
|
||||
--hash=sha256:c18ff11e86df2e28854939acde2d003f7984f721eba450b56a200ad90eeb0e6b \
|
||||
--hash=sha256:c3bcce8521d785d510b2aad26ae2c966092b7daa8f45dd8f44734a104dc0bc1a \
|
||||
--hash=sha256:c4143987a42a2397f2fc3b4d7e3a7d313fbe684f67ff443999e803dd75a76826 \
|
||||
--hash=sha256:c69fd885df7d089548a42d5ec05be26050ebcd2283d89b3d30676eb32ff87dee \
|
||||
--hash=sha256:ced80795227d70549a411a4ab66e8ce307899fad2220ce5ab2f296e687eacde9 \
|
||||
--hash=sha256:d66e421495fdb797610a08f43b05269e0a5ea7f5e652a89bfd5a7d3c1dee3648 \
|
||||
--hash=sha256:d861ee9e76ace6cf36a6a89b959ec08e7bc2493ee39d07ffe5acb23ef46d27da \
|
||||
--hash=sha256:e9251e3be159d1020c4030bd2e5f84d6a43fe54b6c19c12f51cde9542a2817b2 \
|
||||
--hash=sha256:f145bba11b878005c496e93e257c1e88f154d278d2638e6450d17e0f31e558d2 \
|
||||
--hash=sha256:fe346b143ff9685e40192a4960938545c699054ba11d4f9029f94751e3f71d87
|
||||
cryptography==48.0.0 \
|
||||
--hash=sha256:0890f502ddf7d9c6426129c3f49f5c0a39278ed7cd6322c8755ffca6ee675a13 \
|
||||
--hash=sha256:0c558d2cdffd8f4bbb30fc7134c74d2ca9a476f830bb053074498fbc86f41ed6 \
|
||||
--hash=sha256:16cd65b9330583e4619939b3a3843eec1e6e789744bb01e7c7e2e62e33c239c8 \
|
||||
--hash=sha256:18349bbc56f4743c8b12dc32e2bccb2cf83ee8b69a3bba74ef8ae857e26b3d25 \
|
||||
--hash=sha256:1e2d54c8be6152856a36f0882ab231e70f8ec7f14e93cf87db8a2ed056bf160c \
|
||||
--hash=sha256:22a5cb272895dce158b2cacdfdc3debd299019659f42947dbdac6f32d68fe832 \
|
||||
--hash=sha256:27241b1dc9962e056062a8eef1991d02c3a24569c95975bd2322a8a52c6e5e12 \
|
||||
--hash=sha256:2b4d59804e8408e2fea7d1fbaf218e5ec984325221db76e6a241a9abd6cdd95c \
|
||||
--hash=sha256:2eb992bbd4661238c5a397594c83f5b4dc2bc5b848c365c8f991b6780efcc5c7 \
|
||||
--hash=sha256:369a6348999f94bbd53435c894377b20ab95f25a9065c283570e70150d8abc3c \
|
||||
--hash=sha256:3cb07a3ed6431663cd321ea8a000a1314c74211f823e4177fefa2255e057d1ec \
|
||||
--hash=sha256:40ba1f85eaa6959837b1d51c9767e230e14612eea4ef110ee8854ada22da1bf5 \
|
||||
--hash=sha256:4defde8685ae324a9eb9d818717e93b4638ef67070ac9bc15b8ca85f63048355 \
|
||||
--hash=sha256:55b7718303bf06a5753dcdccf2f3945cf18ad7bffde41b61226e4db31ab89a9c \
|
||||
--hash=sha256:561215ea3879cb1cbbf272867e2efda62476f240fb58c64de6b393ae19246741 \
|
||||
--hash=sha256:58d00498e8933e4a194f3076aee1b4a97dfec1a6da444535755822fe5d8b0b86 \
|
||||
--hash=sha256:59baa2cb386c4f0b9905bd6eb4c2a79a69a128408fd31d32ca4d7102d4156321 \
|
||||
--hash=sha256:5a5ed8fde7a1d09376ca0b40e68cd59c69fe23b1f9768bd5824f54681626032a \
|
||||
--hash=sha256:5b012212e08b8dd5edc78ef54da83dd9892fd9105323b3993eff6bea65dc21d7 \
|
||||
--hash=sha256:5c3932f4436d1cccb036cb0eaef46e6e2db91035166f1ad6505c3c9d5a635920 \
|
||||
--hash=sha256:614d0949f4790582d2cc25553abd09dd723025f0c0e7c67376a1d77196743d6e \
|
||||
--hash=sha256:76341972e1eff8b4bea859f09c0d3e64b96ce931b084f9b9b7db8ef364c30eff \
|
||||
--hash=sha256:77a2ccbbe917f6710e05ba9adaa25fb5075620bf3ea6fb751997875aff4ae4bd \
|
||||
--hash=sha256:7995ef305d7165c3f11ae07f2517e5a4f1d5c18da1376a0a9ed496336b69e5f3 \
|
||||
--hash=sha256:7ce4bfae76319a532a2dc68f82cc32f5676ee792a983187dac07183690e5c66f \
|
||||
--hash=sha256:7e8eac43dfca5c4cccc6dad9a80504436fca53bb9bc3100a2386d730fbe6b602 \
|
||||
--hash=sha256:84cf79f0dc8b36ac5da873481716e87aef31fcfa0444f9e1d8b4b2cece142855 \
|
||||
--hash=sha256:8c7378637d7d88016fa6791c159f698b3d3eed28ebf844ac36b9dc04a14dae18 \
|
||||
--hash=sha256:8cd666227ef7af430aa5914a9910e0ddd703e75f039cef0825cd0da71b6b711a \
|
||||
--hash=sha256:906cbf0670286c6e0044156bc7d4af9cbb0ef6db9f73e52c3ec56ba6bdde5336 \
|
||||
--hash=sha256:9071196d81abc88b3516ac8cdfad32e2b66dd4a5393a8e68a961e9161ddc6239 \
|
||||
--hash=sha256:9249e3cd978541d665967ac2cb2787fd6a62bddf1e75b3e347a594d7dacf4f74 \
|
||||
--hash=sha256:984a20b0f62a26f48a3396c72e4bc34c66e356d356bf370053066b3b6d54634a \
|
||||
--hash=sha256:9be5aafa5736574f8f15f262adc81b2a9869e2cfe9014d52a44633905b40d52c \
|
||||
--hash=sha256:9c459db21422be75e2809370b829a87eb37f74cd785fc4aa9ea1e5f43b47cda4 \
|
||||
--hash=sha256:9ccdac7d40688ecb5a3b4a604b8a88c8002e3442d6c60aead1db2a89a041560c \
|
||||
--hash=sha256:a0e692c683f4df67815a2d258b324e66f4738bd7a96a218c826dce4f4bd05d8f \
|
||||
--hash=sha256:a5da777e32ffed6f85a7b2b3f7c5cbc88c146bfcd0a1d7baf5fcc6c52ee35dd4 \
|
||||
--hash=sha256:a64697c641c7b1b2178e573cbc31c7c6684cd56883a478d75143dbb7118036db \
|
||||
--hash=sha256:ad64688338ed4bc1a6618076ba75fd7194a5f1797ac60b47afe926285adb3166 \
|
||||
--hash=sha256:bd72e68b06bb1e96913f97dd4901119bc17f39d4586a5adf2d3e47bc2b9d58b5 \
|
||||
--hash=sha256:c17dfe85494deaeddc5ce251aebd1d60bbe6afc8b62071bb0b469431a000124f \
|
||||
--hash=sha256:c18684a7f0cc9a3cb60328f496b8e3372def7c5d2df39ac267878b05565aaaae \
|
||||
--hash=sha256:cc90c0b39b2e3c65ef52c804b72e3c58f8a04ab2a1871272798e5f9572c17d20 \
|
||||
--hash=sha256:db63bf618e5dea46c07de12e900fe1cdd2541e6dc9dbae772a70b7d4d4765f6a \
|
||||
--hash=sha256:ea8990436d914540a40ab24b6a77c0969695ed52f4a4874c5137ccf7045a7057 \
|
||||
--hash=sha256:ecde28a596bead48b0cfd2a1b4416c3d43074c2d785e3a398d7ec1fc4d0f7fbb \
|
||||
--hash=sha256:f5333311663ea94f75dd408665686aaf426563556bb5283554a3539177e03b8c \
|
||||
--hash=sha256:fdfef35d751d510fcef5252703621574364fec16418c4a1e5e1055248401054b
|
||||
# via
|
||||
# -r requirements_formatting.txt.in
|
||||
# pyjwt
|
||||
@@ -281,9 +281,9 @@ graylint==1.1.1 \
|
||||
--hash=sha256:0fd8e02972ca03d0ef2bf0adea76b5343efcd492d7afb5f658f3e3a724f55a36 \
|
||||
--hash=sha256:b7e0eab6c159684dbf5ef84e942c3340f6a6549b02a3d11b1a1763cc4f8f0593
|
||||
# via darker
|
||||
idna==3.10 \
|
||||
--hash=sha256:12f65c9b470abda6dc35cf8e63cc574b1c52b11df2c86030af0ac09b01b13ea9 \
|
||||
--hash=sha256:946d195a0d259cbba61165e88e65941f16e9b36ea6ddb97f00452bae8b1287d3
|
||||
idna==3.16 \
|
||||
--hash=sha256:cc246e3a3f89580c3a951b5ad298ca4638078b2cdd4f115654332b5c26daded5 \
|
||||
--hash=sha256:d7a6da03db833450fca25d2358ac9ff06cd624577a4aea3a596d5c0f77b8e03d
|
||||
# via
|
||||
# -r requirements_formatting.txt.in
|
||||
# requests
|
||||
@@ -390,9 +390,9 @@ pytokens==0.4.1 \
|
||||
--hash=sha256:ee44d0f85b803321710f9239f335aafe16553b39106384cef8e6de40cb4ef2f6 \
|
||||
--hash=sha256:f66a6bbe741bd431f6d741e617e0f39ec7257ca1f89089593479347cc4d13324
|
||||
# via black
|
||||
requests==2.32.4 \
|
||||
--hash=sha256:27babd3cda2a6d50b30443204ee89830707d396671944c998b5975b031ac2b2c \
|
||||
--hash=sha256:27d0316682c8a29834d3264820024b62a36942083d52caf2f14c0591336d3422
|
||||
requests==2.34.2 \
|
||||
--hash=sha256:2a0d60c172f83ac6ab31e4554906c0f3b3588d37b5cb939b1c061f4907e278e0 \
|
||||
--hash=sha256:f288924cae4e29463698d6d60bc6a4da69c89185ad1e0bcc4104f584e960b9ed
|
||||
# via
|
||||
# -r requirements_formatting.txt.in
|
||||
# pygithub
|
||||
@@ -406,9 +406,9 @@ typing-extensions==4.14.1 \
|
||||
--hash=sha256:38b39f4aeeab64884ce9f74c94263ef78f3c22467c8724005483154c26648d36 \
|
||||
--hash=sha256:d1e1e3b58374dc93031d6eda2420a48ea44a36c2b4766a4fdeb3710755731d76
|
||||
# via pygithub
|
||||
urllib3==2.6.3 \
|
||||
--hash=sha256:1b62b6884944a57dbe321509ab94fd4d3b307075e0c2eae991ac71ee15ad38ed \
|
||||
--hash=sha256:bf272323e553dfb2e87d9bfd225ca7b0f467b919d7bbd355436d3fd37cb0acd4
|
||||
urllib3==2.7.0 \
|
||||
--hash=sha256:231e0ec3b63ceb14667c67be60f2f2c40a518cb38b03af60abc813da26505f4c \
|
||||
--hash=sha256:9fb4c81ebbb1ce9531cce37674bbc6f1360472bc18ca9a553ede278ef7276897
|
||||
# via
|
||||
# -r requirements_formatting.txt.in
|
||||
# pygithub
|
||||
|
||||
@@ -1,10 +1,10 @@
|
||||
black>=26.3.1
|
||||
darker==2.1.1
|
||||
PyGithub==2.6.1
|
||||
cryptography>=46.0.5
|
||||
urllib3>=2.6.3
|
||||
requests>=2.32.4
|
||||
idna>=3.7
|
||||
cryptography>=46.0.7
|
||||
urllib3>=2.7.0
|
||||
requests>=2.33.0
|
||||
idna>=3.15
|
||||
certifi>=2024.7.4
|
||||
PyNaCl>=1.6.2
|
||||
PyJWT>=2.12.1
|
||||
Vendored
+1
-1
Submodule External/rpmalloc updated: 1f6fb494f2...1d85c246cd.
@@ -23,6 +23,13 @@
|
||||
"Enable the code caching subsystem"
|
||||
]
|
||||
},
|
||||
"EnableLazyCodeCachingWIP": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Enable lazy loading of chunks in code caches"
|
||||
]
|
||||
},
|
||||
"EnableCodeCacheValidation": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
|
||||
@@ -53,6 +53,6 @@ FEXCore::CPUID::FunctionResults FEXCore::Context::ContextImpl::RunCPUIDFunctionN
|
||||
}
|
||||
|
||||
bool FEXCore::Context::ContextImpl::IsAddressInCodeBuffer(FEXCore::Core::InternalThreadState* Thread, uintptr_t Address) const {
|
||||
return Thread->CPUBackend->IsAddressInCodeBuffer(Address);
|
||||
return Thread->CPUBackend->IsAddressInCodeBuffer(Address) || CodeCache.IsAddressInMappedCodeBuffer(Address);
|
||||
}
|
||||
} // namespace FEXCore::Context
|
||||
@@ -64,6 +64,8 @@ struct CustomIRResult {
|
||||
using BlockDelinkerFunc = void (*)(FEXCore::Context::ExitFunctionLinkData* Record);
|
||||
constexpr uint32_t TSC_SCALE_MAXIMUM = 1'000'000'000; ///< 1Ghz
|
||||
|
||||
constexpr static bool BLOCK_DEBUGGING = false;
|
||||
|
||||
class CodeCache : public AbstractCodeCache {
|
||||
public:
|
||||
CodeCache(ContextImpl&);
|
||||
@@ -76,11 +78,17 @@ public:
|
||||
bool IsGeneratingCache = false;
|
||||
|
||||
FEX_CONFIG_OPT(EnableCodeCaching, ENABLECODECACHINGWIP);
|
||||
FEX_CONFIG_OPT(EnableLazyCodeCaching, ENABLELAZYCODECACHINGWIP);
|
||||
FEX_CONFIG_OPT(EnableCodeCacheValidation, ENABLECODECACHEVALIDATION);
|
||||
|
||||
uint64_t ComputeCodeMapId(std::string_view Filename, int FD) override;
|
||||
bool SaveData(Core::InternalThreadState&, int TargetFD, const ExecutableFileSectionInfo&, uint64_t SerializedBaseAddress) override;
|
||||
bool LoadData(Core::InternalThreadState*, std::byte* MappedCacheFile, const ExecutableFileSectionInfo&) override;
|
||||
|
||||
fextl::unique_ptr<MappedCodeCacheFile> LoadCache(std::span<std::byte> CacheFile, const ExecutableFileInfo&, uint64_t FileStartVA) override;
|
||||
|
||||
bool EnableLoadedSection(Core::InternalThreadState*, MappedCodeCacheFile&, const ExecutableFileSectionInfo&) override;
|
||||
|
||||
void FinalizeCodePages(MappedCodeCacheFile&, std::span<std::byte> CodeRange) override;
|
||||
|
||||
/**
|
||||
* Performs expensive extra validation on the loaded code cache data.
|
||||
@@ -112,12 +120,14 @@ public:
|
||||
* Note that FEX relocations are unrelated to ELF/PE relocations.
|
||||
*
|
||||
* @param GuestDelta Guest address offset to apply to RIP-relative data
|
||||
* @param RelocationOffset Offset to subtract from relocation target offsets
|
||||
* @param ForStorage True for serializing data (producing deterministic output); false for de-serializing it (resolving dynamic symbols)
|
||||
*
|
||||
* @return Returns true on success
|
||||
*/
|
||||
[[nodiscard]]
|
||||
bool ApplyCodeRelocations(uint64_t GuestDelta, std::span<std::byte> Code, std::span<const CPU::Relocation> Relocations, bool ForStorage);
|
||||
bool ApplyCodeRelocations(uint64_t GuestDelta, std::span<std::byte> Code, std::span<const CPU::Relocation> Relocations,
|
||||
uint32_t RelocationOffset, bool ForStorage);
|
||||
};
|
||||
|
||||
class ContextImpl final : public FEXCore::Context::Context, public CPU::CodeBufferManager {
|
||||
@@ -233,6 +243,83 @@ public:
|
||||
|
||||
void MarkMonoBackpatcherBlock(uint64_t BlockEntry) override;
|
||||
|
||||
// Manual debugging tooling which is useful for developers.
|
||||
struct TrackingEmpty {
|
||||
// RIP stepping handling
|
||||
virtual void AddSingleStepTarget(uint64_t GuestRIP) {}
|
||||
virtual void AllTargetSingleStep() {}
|
||||
virtual void RemoveSingleStepTarget(uint64_t GuestRIP) {}
|
||||
virtual bool IsSingleStepTarget(uint64_t GuestRIP) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Watchpoints
|
||||
virtual void AddWriteWatchPoint(uint64_t Ptr) {}
|
||||
virtual void AddReadWatchPoint(uint64_t Ptr) {}
|
||||
virtual bool ContainsWriteWatchPoint(uint64_t Ptr, size_t Size) {
|
||||
return false;
|
||||
}
|
||||
virtual bool ContainsReadWatchPoint(uint64_t Ptr, size_t Size) {
|
||||
return false;
|
||||
}
|
||||
};
|
||||
|
||||
struct TrackingPossible final : public TrackingEmpty {
|
||||
void AddSingleStepTarget(uint64_t GuestRIP) override {
|
||||
SingleStepTargets.emplace(GuestRIP);
|
||||
}
|
||||
|
||||
void RemoveSingleStepTarget(uint64_t GuestRIP) override {
|
||||
SingleStepTargets.erase(GuestRIP);
|
||||
}
|
||||
|
||||
void AllTargetSingleStep() override {
|
||||
SingleStepEverything = true;
|
||||
}
|
||||
|
||||
bool IsSingleStepTarget(uint64_t GuestRIP) override {
|
||||
return SingleStepEverything || SingleStepTargets.contains(GuestRIP);
|
||||
}
|
||||
|
||||
void AddWriteWatchPoint(uint64_t Ptr) override {
|
||||
WatchWriteTargets.emplace(Ptr);
|
||||
}
|
||||
|
||||
void AddReadWatchPoint(uint64_t Ptr) override {
|
||||
WatchReadTargets.emplace(Ptr);
|
||||
}
|
||||
|
||||
bool ContainsWriteWatchPoint(uint64_t Ptr, size_t Size) override {
|
||||
return ContainsRange(WatchWriteTargets, Ptr, Size);
|
||||
}
|
||||
|
||||
bool ContainsReadWatchPoint(uint64_t Ptr, size_t Size) override {
|
||||
return ContainsRange(WatchReadTargets, Ptr, Size);
|
||||
}
|
||||
|
||||
private:
|
||||
bool SingleStepEverything {};
|
||||
fextl::set<uint64_t> SingleStepTargets {};
|
||||
fextl::set<uint64_t> WatchWriteTargets {};
|
||||
fextl::set<uint64_t> WatchReadTargets {};
|
||||
|
||||
static bool ContainsRange(const fextl::set<uint64_t>& Set, uint64_t Ptr, size_t Size) {
|
||||
for (auto it = Set.lower_bound(Ptr); it != Set.end(); --it) {
|
||||
auto Watch = *it;
|
||||
if (Watch < Ptr) {
|
||||
break;
|
||||
}
|
||||
if (Watch >= Ptr && Watch < (Ptr + Size)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
};
|
||||
using TrackingStructure = std::conditional<BLOCK_DEBUGGING, TrackingPossible, TrackingEmpty>::type;
|
||||
|
||||
TrackingStructure BlockDebuggerTracker {};
|
||||
public:
|
||||
struct {
|
||||
uint64_t VirtualMemSize {1ULL << 36};
|
||||
|
||||
@@ -586,8 +586,8 @@ void Arm64Emitter::PushCalleeSavedRegisters() {
|
||||
{ARMEmitter::XReg::x29, ARMEmitter::XReg::x30},
|
||||
}};
|
||||
|
||||
for (auto& RegPair : CalleeSaved) {
|
||||
stp<ARMEmitter::IndexType::PRE>(RegPair.first, RegPair.second, ARMEmitter::Reg::rsp, -16);
|
||||
for (const auto& [rt, rt2] : CalleeSaved) {
|
||||
stp<ARMEmitter::IndexType::PRE>(rt, rt2, ARMEmitter::Reg::rsp, -16);
|
||||
}
|
||||
|
||||
// Additionally we need to store the lower 64bits of v8-v15
|
||||
@@ -604,9 +604,8 @@ void Arm64Emitter::PushCalleeSavedRegisters() {
|
||||
// We just saved x19 so it is safe
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r19, ARMEmitter::Reg::rsp, 0);
|
||||
|
||||
for (auto& RegQuad : FPRs) {
|
||||
st4(ARMEmitter::SubRegSize::i64Bit, std::get<0>(RegQuad), std::get<1>(RegQuad), std::get<2>(RegQuad), std::get<3>(RegQuad), 0,
|
||||
ARMEmitter::Reg::r19, 32);
|
||||
for (const auto& [rt, rt2, rt3, rt4] : FPRs) {
|
||||
st4(ARMEmitter::SubRegSize::i64Bit, rt, rt2, rt3, rt4, 0, ARMEmitter::Reg::r19, 32);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -616,9 +615,8 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
{ARMEmitter::DReg::d12, ARMEmitter::DReg::d13, ARMEmitter::DReg::d14, ARMEmitter::DReg::d15},
|
||||
}};
|
||||
|
||||
for (auto& RegQuad : FPRs) {
|
||||
ld4(ARMEmitter::SubRegSize::i64Bit, std::get<0>(RegQuad), std::get<1>(RegQuad), std::get<2>(RegQuad), std::get<3>(RegQuad), 0,
|
||||
ARMEmitter::Reg::rsp, 32);
|
||||
for (const auto& [rt, rt2, rt3, rt4] : FPRs) {
|
||||
ld4(ARMEmitter::SubRegSize::i64Bit, rt, rt2, rt3, rt4, 0, ARMEmitter::Reg::rsp, 32);
|
||||
}
|
||||
|
||||
constexpr static std::array<std::pair<ARMEmitter::XRegister, ARMEmitter::XRegister>, 6> CalleeSaved = {{
|
||||
@@ -630,8 +628,8 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
{ARMEmitter::XReg::x19, ARMEmitter::XReg::x20},
|
||||
}};
|
||||
|
||||
for (auto& RegPair : CalleeSaved) {
|
||||
ldp<ARMEmitter::IndexType::POST>(RegPair.first, RegPair.second, ARMEmitter::Reg::rsp, 16);
|
||||
for (const auto& [rt, rt2] : CalleeSaved) {
|
||||
ldp<ARMEmitter::IndexType::POST>(rt, rt2, ARMEmitter::Reg::rsp, 16);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -661,7 +659,7 @@ void Arm64Emitter::FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Regi
|
||||
}
|
||||
#endif
|
||||
|
||||
if (SetPredRegs && (EmitterCTX->HostFeatures.SupportsSVE256 || EmitterCTX->HostFeatures.SupportsSVE128)) {
|
||||
if (SetPredRegs && EmitterCTX->HostFeatures.SupportsSVE()) {
|
||||
// Set up predicate registers.
|
||||
// We don't bother spilling these in SpillStaticRegs,
|
||||
// since all that matters is we restore them on a fill.
|
||||
|
||||
@@ -43,6 +43,8 @@ namespace CPU {
|
||||
{0x0706'0504'FFFF'FFFFULL, 0x0F0E'0D0C'0B0A'0908ULL}, // NAMED_VECTOR_BLENDPS_1110B
|
||||
{0x8040'2010'0804'0201ULL, 0x8040'2010'0804'0201ULL}, // NAMED_VECTOR_MOVMASKB
|
||||
{0x8040'2010'0804'0201ULL, 0x8040'2010'0804'0201ULL}, // NAMED_VECTOR_MOVMASKB_UPPER
|
||||
{0x0706'0504'0302'0100ULL, 0x1716'1514'1312'1110ULL}, // NAMED_VECTOR_256_MID_ELEMENT_SWAP
|
||||
{0x0F0E'0D0C'0B0A'0908ULL, 0x1F1E'1D1C'1B1A'1918ULL}, // NAMED_VECTOR_256_MID_ELEMENT_SWAP_UPPER
|
||||
{0x8000'0000'0000'0000ULL, 0x0000'0000'0000'3FFFULL}, // NAMED_VECTOR_X87_ONE
|
||||
{0xD49A'784B'CD1B'8AFEULL, 0x0000'0000'0000'4000ULL}, // NAMED_VECTOR_X87_LOG2_10
|
||||
{0xB8AA'3B29'5C17'F0BCULL, 0x0000'0000'0000'3FFFULL}, // NAMED_VECTOR_X87_LOG2_E
|
||||
|
||||
@@ -1,4 +1,9 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "FEXCore/Utils/LogManager.h"
|
||||
#include "FEXCore/Utils/MathUtils.h"
|
||||
#include "FEXCore/Utils/TypeDefines.h"
|
||||
#include "FEXCore/fextl/memory.h"
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/Utils/SpinWaitLock.h>
|
||||
|
||||
#include <Interface/Context/Context.h>
|
||||
@@ -16,10 +21,14 @@
|
||||
|
||||
#include <FEXHeaderUtils/Filesystem.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <git_version.h>
|
||||
|
||||
#include <span>
|
||||
#include <xxhash.h>
|
||||
|
||||
#include <FEXCore/Utils/AllocatorHooks.h>
|
||||
|
||||
#include <fstream>
|
||||
|
||||
namespace FEXCore {
|
||||
@@ -32,6 +41,38 @@ ExecutableFileInfo::ExecutableFileInfo(fextl::unique_ptr<HLE::SourcecodeMap> Map
|
||||
#endif
|
||||
ExecutableFileInfo::~ExecutableFileInfo() = default;
|
||||
|
||||
MappedCodeCacheFile::~MappedCodeCacheFile() {
|
||||
if (CacheManager) {
|
||||
CacheManager->UnregisterMappedCodeBuffer(*this);
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
if (!CodeBuffer.empty()) {
|
||||
FEXCore::Allocator::munmap(CodeBuffer.data(), CodeBuffer.size_bytes());
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void AbstractCodeCache::RegisterMappedCodeBuffer(MappedCodeCacheFile& Code) {
|
||||
MappedCodeBuffers.push_back(Code.CodeBuffer);
|
||||
// Unregister on destruction of Code
|
||||
Code.CacheManager = this;
|
||||
}
|
||||
|
||||
void AbstractCodeCache::UnregisterMappedCodeBuffer(MappedCodeCacheFile& Code) {
|
||||
std::erase_if(MappedCodeBuffers, [&](const auto& Elem) { return Elem.data() == Code.CodeBuffer.data(); });
|
||||
}
|
||||
|
||||
bool AbstractCodeCache::IsAddressInMappedCodeBuffer(uintptr_t Address) const {
|
||||
for (const auto& Range : MappedCodeBuffers) {
|
||||
auto Start = reinterpret_cast<uintptr_t>(Range.data());
|
||||
if (Address >= Start && Address < Start + Range.size_bytes()) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
fextl::string CodeMap::GetBaseFilename(const ExecutableFileInfo& MainExecutable, bool AddNombSuffix) {
|
||||
auto FileId = MainExecutable.FileId;
|
||||
|
||||
@@ -233,7 +274,10 @@ uint64_t CodeCache::ComputeCodeMapId(std::string_view Filename, int FD) {
|
||||
|
||||
struct CodeCacheHeader {
|
||||
std::array<char, 4> Magic = ExpectedMagic;
|
||||
uint32_t FormatVersion = 1;
|
||||
// Version history:
|
||||
// 1: Initial version
|
||||
// 2: Padding code buffer data to enable direct mapping
|
||||
uint32_t FormatVersion = 2;
|
||||
uint8_t FEXVersion[20] = {};
|
||||
uint32_t NumBlocks;
|
||||
uint32_t NumCodePages;
|
||||
@@ -260,7 +304,7 @@ bool CodeCache::SaveData(Core::InternalThreadState& Thread, int fd, const Execut
|
||||
std::ranges::copy(GIT_HASH, header.FEXVersion);
|
||||
header.NumBlocks = LookupCache.BlockList.size();
|
||||
header.NumCodePages = LookupCache.CodePages.size();
|
||||
header.CodeBufferSize = CTX.LatestOffset;
|
||||
header.CodeBufferSize = FEXCore::AlignUp(CTX.LatestOffset, Utils::FEX_PAGE_SIZE);
|
||||
header.NumRelocations = Relocations.size();
|
||||
header.SerializedBaseAddress = SerializedBaseAddress;
|
||||
::write(fd, &header, sizeof(header));
|
||||
@@ -308,11 +352,17 @@ bool CodeCache::SaveData(Core::InternalThreadState& Thread, int fd, const Execut
|
||||
|
||||
// Dump the host code (relocated for position-independent serialization)
|
||||
std::span CodeBufferData(reinterpret_cast<std::byte*>(CodeBuffer->Ptr), reinterpret_cast<std::byte*>(CodeBuffer->Ptr) + CTX.LatestOffset);
|
||||
if (!ApplyCodeRelocations(SerializedBaseAddress, CodeBufferData, Relocations, true)) {
|
||||
if (!ApplyCodeRelocations(SerializedBaseAddress, CodeBufferData, Relocations, 0, true)) {
|
||||
LOGMAN_THROW_A_FMT(false, "Failed to apply code relocations");
|
||||
return false;
|
||||
}
|
||||
::write(fd, CodeBufferData.data(), CodeBufferData.size());
|
||||
// Pad to next page in file for mmap
|
||||
{
|
||||
auto PaddedSize = AlignUp(lseek(fd, 0, SEEK_CUR), Utils::FEX_PAGE_SIZE);
|
||||
::ftruncate(fd, PaddedSize);
|
||||
lseek(fd, PaddedSize, SEEK_SET);
|
||||
}
|
||||
|
||||
// Dump code pages
|
||||
static_assert(OrderedContainer<decltype(LookupCache.CodePages)>, "Non-deterministic data source");
|
||||
@@ -330,167 +380,6 @@ bool CodeCache::SaveData(Core::InternalThreadState& Thread, int fd, const Execut
|
||||
return true;
|
||||
}
|
||||
|
||||
bool CodeCache::LoadData(Core::InternalThreadState* Thread, std::byte* MappedCacheFile, const ExecutableFileSectionInfo& BinarySection) {
|
||||
if (!EnableCodeCaching) {
|
||||
return true;
|
||||
}
|
||||
|
||||
namespace ranges = std::ranges;
|
||||
|
||||
// Read file header
|
||||
CodeCacheHeader header {};
|
||||
::memcpy(&header, MappedCacheFile, sizeof(header));
|
||||
MappedCacheFile += sizeof(header);
|
||||
|
||||
LogMan::Msg::IFmt("Cache load: {:5} blocks; base={:#14x}; off={:#9x}-{:#09x}; {:016x} {}", header.NumBlocks, BinarySection.FileStartVA,
|
||||
BinarySection.BeginVA - BinarySection.FileStartVA, BinarySection.EndVA - BinarySection.FileStartVA,
|
||||
BinarySection.FileInfo.FileId, BinarySection.FileInfo.Filename);
|
||||
|
||||
if (!ranges::equal(header.Magic, header.ExpectedMagic)) {
|
||||
LogMan::Msg::EFmt("Invalid cache file header");
|
||||
return false;
|
||||
}
|
||||
|
||||
if (!ranges::equal(header.FEXVersion, GIT_HASH)) {
|
||||
LogMan::Msg::IFmt("Cache generated from old FEX version {:02x}, current is {:02x}; skipping", fmt::join(header.FEXVersion, ""),
|
||||
fmt::join(GIT_HASH, ""));
|
||||
return false;
|
||||
}
|
||||
|
||||
if (header.NumBlocks == 0) {
|
||||
// Valid caches are never empty
|
||||
LogMan::Msg::IFmt("Code cache empty, aborting");
|
||||
return false;
|
||||
}
|
||||
|
||||
// Read guest<->host block mappings
|
||||
using BlockListEntry = decltype(GuestToHostMap::BlockList)::value_type;
|
||||
fextl::vector<BlockListEntry> BlockList(header.NumBlocks);
|
||||
{
|
||||
for (auto& BlockPtr : BlockList) {
|
||||
::memcpy(&BlockPtr.first, MappedCacheFile, sizeof(BlockPtr.first));
|
||||
MappedCacheFile += sizeof(BlockPtr.first);
|
||||
::memcpy(&BlockPtr.second.HostCode, MappedCacheFile, sizeof(BlockPtr.second.HostCode));
|
||||
MappedCacheFile += sizeof(BlockPtr.second.HostCode);
|
||||
uint64_t NumGuestPages;
|
||||
::memcpy(&NumGuestPages, MappedCacheFile, sizeof(NumGuestPages));
|
||||
MappedCacheFile += sizeof(NumGuestPages);
|
||||
|
||||
BlockPtr.second.CodePages.resize(NumGuestPages);
|
||||
::memcpy(BlockPtr.second.CodePages.data(), MappedCacheFile, std::span {BlockPtr.second.CodePages}.size_bytes());
|
||||
MappedCacheFile += std::span {BlockPtr.second.CodePages}.size_bytes();
|
||||
}
|
||||
|
||||
// Constrain BlockList to the given ExecutableFileSectionInfo
|
||||
LOGMAN_THROW_A_FMT(ranges::is_sorted(BlockList, [](auto& a, auto& b) { return a.first < b.first; }), "Expected sorted block list");
|
||||
auto begin = ranges::lower_bound(BlockList, BinarySection.BeginVA - BinarySection.FileStartVA, std::less {}, &BlockListEntry::first);
|
||||
auto end =
|
||||
ranges::upper_bound(begin, BlockList.end(), BinarySection.EndVA - BinarySection.FileStartVA - 1, std::less {}, &BlockListEntry::first);
|
||||
if (begin == end) {
|
||||
// Not an error since there is just no data to load
|
||||
LogMan::Msg::IFmt("No blocks cached in this range, aborting");
|
||||
return true;
|
||||
}
|
||||
BlockList.erase(end, BlockList.end());
|
||||
BlockList.erase(BlockList.begin(), begin);
|
||||
}
|
||||
|
||||
// Read relocations
|
||||
fextl::vector<FEXCore::CPU::Relocation> Relocations(header.NumRelocations, FEXCore::CPU::Relocation::Default());
|
||||
::memcpy(Relocations.data(), MappedCacheFile, Relocations.size() * sizeof(Relocations[0]));
|
||||
MappedCacheFile += Relocations.size() * sizeof(Relocations[0]);
|
||||
|
||||
// Pad to next page in file, which contains CodeBuffer data
|
||||
MappedCacheFile = reinterpret_cast<std::byte*>(AlignUp(reinterpret_cast<uintptr_t>(MappedCacheFile), Utils::FEX_PAGE_SIZE));
|
||||
|
||||
// Prepare CodeBuffer: Page aligned and big enough to hold all cached data
|
||||
auto Lock = std::unique_lock {CTX.CodeBufferWriteMutex};
|
||||
if (Thread) {
|
||||
if (auto Prev = Thread->CPUBackend->CheckCodeBufferUpdate()) {
|
||||
Allocator::VirtualDontNeed(Thread->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
|
||||
auto lk = Thread->LookupCache->AcquireWriteLock();
|
||||
Thread->LookupCache->ChangeGuestToHostMapping(*Prev, *CTX.GetLatest()->LookupCache, lk);
|
||||
}
|
||||
}
|
||||
|
||||
auto CodeBuffer = CTX.GetLatest();
|
||||
LOGMAN_THROW_A_FMT(reinterpret_cast<uintptr_t>(CodeBuffer->Ptr) % 0x1000 == 0, "Expected CodeBuffer base to be page-aligned");
|
||||
const auto Delta = AlignUp(CTX.LatestOffset, 0x1000) - CTX.LatestOffset;
|
||||
CTX.LatestOffset += Delta;
|
||||
|
||||
while (CTX.LatestOffset + header.CodeBufferSize > CodeBuffer->UsableSize()) {
|
||||
if (Thread) {
|
||||
CTX.ClearCodeCache(Thread);
|
||||
CodeBuffer = CTX.GetLatest();
|
||||
LogMan::Msg::IFmt("Increased code buffer size to {} MiB for cache load", CodeBuffer->AllocatedSize / 1024 / 1024);
|
||||
} else {
|
||||
ERROR_AND_DIE_FMT("Cannot extend codebuffer without thread!");
|
||||
}
|
||||
}
|
||||
|
||||
// Read CodeBuffer data from file. Make sure the destination is page-aligned.
|
||||
// TODO: Only load the data needed for the selected section
|
||||
auto CodeBufferRange =
|
||||
std::as_writable_bytes(std::span {CodeBuffer->Ptr, CodeBuffer->UsableSize()}).subspan(CTX.LatestOffset, header.CodeBufferSize);
|
||||
::memcpy(CodeBufferRange.data(), MappedCacheFile, header.CodeBufferSize);
|
||||
MappedCacheFile += header.CodeBufferSize;
|
||||
CTX.LatestOffset += header.CodeBufferSize;
|
||||
|
||||
// Apply FEX relocations
|
||||
auto Ret = ApplyCodeRelocations(BinarySection.FileStartVA, CodeBufferRange, Relocations, false);
|
||||
LOGMAN_THROW_A_FMT(Ret == true, "Failed to apply code cache relocations");
|
||||
|
||||
{
|
||||
auto& LookupCache = *CodeBuffer->LookupCache;
|
||||
auto WriteLock = LookupCache.AcquireWriteLock();
|
||||
|
||||
// Register blocks to LookupCache
|
||||
for (auto& [Guest, Host] : BlockList) {
|
||||
for (auto& CodePage : Host.CodePages) {
|
||||
CodePage += BinarySection.FileStartVA;
|
||||
}
|
||||
auto HostCode = reinterpret_cast<void*>(Host.HostCode + reinterpret_cast<uintptr_t>(CodeBufferRange.data()));
|
||||
LookupCache.AddBlockMapping(Guest + BinarySection.FileStartVA, std::move(Host.CodePages), HostCode, WriteLock);
|
||||
}
|
||||
|
||||
// Register loaded code ranges
|
||||
fextl::vector<uint64_t> Entrypoints;
|
||||
for (uint32_t i = 0; i < header.NumCodePages; ++i) {
|
||||
uint64_t CodePage;
|
||||
memcpy(&CodePage, MappedCacheFile, sizeof(CodePage));
|
||||
CodePage += BinarySection.FileStartVA;
|
||||
MappedCacheFile += sizeof(CodePage);
|
||||
|
||||
uint64_t NumEntrypoints;
|
||||
memcpy(&NumEntrypoints, MappedCacheFile, sizeof(NumEntrypoints));
|
||||
MappedCacheFile += sizeof(NumEntrypoints);
|
||||
|
||||
Entrypoints.resize(NumEntrypoints);
|
||||
memcpy(Entrypoints.data(), MappedCacheFile, NumEntrypoints * sizeof(Entrypoints[0]));
|
||||
MappedCacheFile += NumEntrypoints * sizeof(Entrypoints[0]);
|
||||
for (auto& Entrypoint : Entrypoints) {
|
||||
Entrypoint += BinarySection.FileStartVA;
|
||||
}
|
||||
|
||||
if (LookupCache.AddBlockExecutableRange(Entrypoints, CodePage, FEXCore::Utils::FEX_PAGE_SIZE, WriteLock)) {
|
||||
CTX.SyscallHandler->MarkGuestExecutableRange(Thread, CodePage, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (EnableCodeCacheValidation) {
|
||||
fextl::set<uint64_t> GuestBlocks, HostBlocks;
|
||||
for (auto& [Guest, Host] : BlockList) {
|
||||
GuestBlocks.insert(Guest + BinarySection.FileStartVA);
|
||||
HostBlocks.insert(Host.HostCode);
|
||||
}
|
||||
|
||||
Validate(BinarySection, std::move(GuestBlocks), HostBlocks, CodeBufferRange);
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
void CodeCache::Validate(const ExecutableFileSectionInfo& Section, fextl::set<uint64_t> GuestBlocks, const fextl::set<uint64_t>& HostBlocks,
|
||||
std::span<std::byte> CachedCode) {
|
||||
LOGMAN_THROW_A_FMT(!HostBlocks.empty(), "Tried to validate without any host blocks");
|
||||
@@ -545,7 +434,7 @@ void CodeCache::Validate(const ExecutableFileSectionInfo& Section, fextl::set<ui
|
||||
NewRelocations.erase(std::remove_if(NewRelocations.begin(), NewRelocations.end(), [](const CPU::Relocation& Reloc) {
|
||||
return Reloc.Header.Type != CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL && Reloc.Header.Type != CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE;
|
||||
}));
|
||||
(void)ApplyCodeRelocations(Section.FileStartVA, CodeBufferRangeRef, NewRelocations, false);
|
||||
(void)ApplyCodeRelocations(Section.FileStartVA, CodeBufferRangeRef, NewRelocations, 0, false);
|
||||
|
||||
if (ValidationCTX->LatestOffset <= CodeBufferRangeRef.size()) {
|
||||
// Reference compilation produced fewer bytes than our cache, so validation is going to fail.
|
||||
@@ -603,15 +492,17 @@ void CodeCache::Validate(const ExecutableFileSectionInfo& Section, fextl::set<ui
|
||||
ValidationThread->LookupCache->ClearCache(ValidationThread->LookupCache->AcquireWriteLock());
|
||||
ValidationCTX->LatestOffset = 0;
|
||||
|
||||
LogMan::Msg::IFmt("\tSuccessfully validated cache");
|
||||
LogMan::Msg::IFmt(" successfully validated cache");
|
||||
}
|
||||
|
||||
bool CodeCache::ApplyCodeRelocations(uint64_t GuestEntry, std::span<std::byte> Code,
|
||||
std::span<const FEXCore::CPU::Relocation> EntryRelocations, bool ForStorage) {
|
||||
std::span<const FEXCore::CPU::Relocation> EntryRelocations, uint32_t RelocationOffset, bool ForStorage) {
|
||||
CPU::Arm64Emitter Emitter(&CTX, Code.data(), Code.size_bytes());
|
||||
for (size_t j = 0; j < EntryRelocations.size(); ++j) {
|
||||
const FEXCore::CPU::Relocation& Reloc = EntryRelocations[j];
|
||||
Emitter.SetCursorOffset(Reloc.Header.Offset);
|
||||
LOGMAN_THROW_A_FMT(Reloc.Header.Offset >= RelocationOffset, "Invalid relocation offset");
|
||||
LOGMAN_THROW_A_FMT(Reloc.Header.Offset - RelocationOffset < Code.size_bytes(), "Invalid relocation offset");
|
||||
Emitter.SetCursorOffset(Reloc.Header.Offset - RelocationOffset);
|
||||
|
||||
switch (Reloc.Header.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
|
||||
@@ -650,4 +541,343 @@ bool CodeCache::ApplyCodeRelocations(uint64_t GuestEntry, std::span<std::byte> C
|
||||
return true;
|
||||
}
|
||||
|
||||
fextl::unique_ptr<MappedCodeCacheFile>
|
||||
CodeCache::LoadCache(std::span<std::byte> CacheFile, const ExecutableFileInfo& FileInfo, uint64_t FileStartVA) {
|
||||
if (!EnableCodeCaching) {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
FEXCORE_PROFILE_SCOPED("LoadCache");
|
||||
|
||||
// Read file header
|
||||
CodeCacheHeader header {};
|
||||
::memcpy(&header, CacheFile.data(), sizeof(header));
|
||||
|
||||
if (!std::ranges::equal(header.Magic, header.ExpectedMagic)) {
|
||||
LogMan::Msg::EFmt("Invalid cache file header");
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
if (!std::ranges::equal(header.FEXVersion, GIT_HASH)) {
|
||||
LogMan::Msg::IFmt("Cache generated from old FEX version {:02x}, current is {:02x}; skipping", fmt::join(header.FEXVersion, ""),
|
||||
fmt::join(GIT_HASH, ""));
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
if (header.NumBlocks == 0) {
|
||||
// Valid caches are never empty
|
||||
LogMan::Msg::IFmt("Code cache empty, aborting");
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
// Skip over BlockEntry data since it won't be used until EnableLoadedSection
|
||||
// TODO: Store direct offset to relocations in the header
|
||||
auto* BlockListStart = CacheFile.data() + sizeof(header);
|
||||
auto* Cursor = BlockListStart;
|
||||
for (uint32_t i = 0; i < header.NumBlocks; ++i) {
|
||||
Cursor += sizeof(uint64_t); // guest address
|
||||
Cursor += sizeof(uint64_t); // host code address
|
||||
uint64_t NumGuestCodePages;
|
||||
::memcpy(&NumGuestCodePages, Cursor, sizeof(NumGuestCodePages));
|
||||
Cursor += sizeof(NumGuestCodePages);
|
||||
Cursor += NumGuestCodePages * sizeof(uint64_t);
|
||||
}
|
||||
|
||||
auto Relocations = std::span {reinterpret_cast<const FEXCore::CPU::Relocation*>(Cursor), header.NumRelocations};
|
||||
Cursor += Relocations.size_bytes();
|
||||
|
||||
// Pad to next page to get the code buffer data
|
||||
Cursor = reinterpret_cast<std::byte*>(AlignUp(reinterpret_cast<uintptr_t>(Cursor), Utils::FEX_PAGE_SIZE));
|
||||
auto CodeDataInFile = std::span {Cursor, header.CodeBufferSize};
|
||||
|
||||
#ifndef _WIN32
|
||||
// Allocate target memory for post-relocation code. This is PROT_NONE until
|
||||
// the first execution, so that contents can be lazily populated in a
|
||||
// frontend-provided segfault handler.
|
||||
void* CodeBufferAllocation = Allocator::mmap(nullptr, header.CodeBufferSize, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
if (CodeBufferAllocation == MAP_FAILED) {
|
||||
LogMan::Msg::EFmt("Failed to reserve target memory for code cache");
|
||||
return nullptr;
|
||||
}
|
||||
auto CodeBuffer = std::span {static_cast<std::byte*>(CodeBufferAllocation), header.CodeBufferSize};
|
||||
#else
|
||||
// TODO: Implement lazy mapping on Windows
|
||||
auto CodeBuffer = CodeDataInFile;
|
||||
#endif
|
||||
|
||||
// Group relocations by page
|
||||
size_t NumPages = header.CodeBufferSize / Utils::FEX_PAGE_SIZE;
|
||||
fextl::vector<MappedCodeCacheFile::PageRelocationRange> PageRelocationRanges(NumPages, {0, 0});
|
||||
auto RelocBaseOffset = std::as_bytes(Relocations).data() - CacheFile.data();
|
||||
auto RelocIt = Relocations.begin();
|
||||
for (size_t Page = 0; Page < NumPages; ++Page) {
|
||||
auto EndRelocIt = std::upper_bound(RelocIt, Relocations.end(), Page,
|
||||
[](auto& Page, auto& Reloc) { return Page < Reloc.Header.Offset / Utils::FEX_PAGE_SIZE; });
|
||||
PageRelocationRanges.at(Page) = {static_cast<uint32_t>(RelocBaseOffset + (RelocIt - Relocations.begin()) * sizeof(CPU::Relocation)),
|
||||
static_cast<uint32_t>(EndRelocIt - RelocIt)};
|
||||
RelocIt = EndRelocIt;
|
||||
}
|
||||
|
||||
auto Storage = FEXCore::Allocator::aligned_alloc(alignof(MappedCodeCacheFile), sizeof(MappedCodeCacheFile));
|
||||
return fextl::unique_ptr<MappedCodeCacheFile>(
|
||||
new (Storage) MappedCodeCacheFile {this, CacheFile, CodeDataInFile, CodeBuffer, BlockListStart, header.NumBlocks, header.NumCodePages,
|
||||
std::move(PageRelocationRanges), fextl::vector<bool>(NumPages), FileStartVA});
|
||||
}
|
||||
|
||||
bool CodeCache::EnableLoadedSection(Core::InternalThreadState* Thread, MappedCodeCacheFile& Code, const ExecutableFileSectionInfo& BinarySection) {
|
||||
if (!EnableCodeCaching) {
|
||||
return true;
|
||||
}
|
||||
|
||||
namespace ranges = std::ranges;
|
||||
|
||||
FEXCORE_PROFILE_SCOPED("EnableLoadedSection");
|
||||
|
||||
// Read block list from cache file
|
||||
// TODO: Store section-ized BlockLists in cache file
|
||||
using BlockListEntry = decltype(GuestToHostMap::BlockList)::value_type;
|
||||
fextl::vector<BlockListEntry> BlockList(Code.NumBlocks);
|
||||
{
|
||||
auto* Cursor = Code.BlockListInFile;
|
||||
for (auto& BlockPtr : BlockList) {
|
||||
::memcpy(&BlockPtr.first, Cursor, sizeof(BlockPtr.first));
|
||||
Cursor += sizeof(BlockPtr.first);
|
||||
::memcpy(&BlockPtr.second.HostCode, Cursor, sizeof(BlockPtr.second.HostCode));
|
||||
Cursor += sizeof(BlockPtr.second.HostCode);
|
||||
uint64_t NumGuestPages;
|
||||
::memcpy(&NumGuestPages, Cursor, sizeof(NumGuestPages));
|
||||
Cursor += sizeof(NumGuestPages);
|
||||
|
||||
BlockPtr.second.CodePages.resize(NumGuestPages);
|
||||
::memcpy(BlockPtr.second.CodePages.data(), Cursor, std::span {BlockPtr.second.CodePages}.size_bytes());
|
||||
Cursor += std::span {BlockPtr.second.CodePages}.size_bytes();
|
||||
}
|
||||
|
||||
// Constrain BlockList to the given ExecutableFileSectionInfo
|
||||
LOGMAN_THROW_A_FMT(ranges::is_sorted(BlockList, [](auto& a, auto& b) { return a.first < b.first; }), "Expected sorted block list");
|
||||
auto begin = ranges::lower_bound(BlockList, BinarySection.BeginVA - BinarySection.FileStartVA, std::less {}, &BlockListEntry::first);
|
||||
auto end =
|
||||
ranges::upper_bound(begin, BlockList.end(), BinarySection.EndVA - BinarySection.FileStartVA - 1, std::less {}, &BlockListEntry::first);
|
||||
if (begin == end) {
|
||||
LogMan::Msg::IFmt("No blocks cached in this range, aborting");
|
||||
return true;
|
||||
}
|
||||
BlockList.erase(end, BlockList.end());
|
||||
BlockList.erase(BlockList.begin(), begin);
|
||||
}
|
||||
|
||||
LogMan::Msg::IFmt("Cache load: {:5} blocks; base={:#14x}; off={:#9x}-{:#09x}; {:016x} {}", BlockList.size(), BinarySection.FileStartVA,
|
||||
BinarySection.BeginVA - BinarySection.FileStartVA, BinarySection.EndVA - BinarySection.FileStartVA,
|
||||
BinarySection.FileInfo.FileId, BinarySection.FileInfo.Filename);
|
||||
|
||||
if (EnableLazyCodeCaching) {
|
||||
LogMan::Msg::IFmt(" lazy mapping: base={:#14x} -> host={}; cache_source={}", BinarySection.FileStartVA,
|
||||
fmt::ptr(Code.CodeBuffer.data()), fmt::ptr(Code.MappedFile.data()));
|
||||
}
|
||||
// Register blocks to LookupCache.
|
||||
// The host addresses will point into the protected code buffer, so that FEX
|
||||
// can lazily apply relocations on first execution of each page.
|
||||
auto CodeBuffer = CTX.GetLatest();
|
||||
{
|
||||
FEXCORE_PROFILE_SCOPED("Decode");
|
||||
auto& LookupCache = *CodeBuffer->LookupCache;
|
||||
auto WriteLock = LookupCache.AcquireWriteLock();
|
||||
|
||||
for (auto& [Guest, Block] : BlockList) {
|
||||
for (auto& CodePage : Block.CodePages) {
|
||||
CodePage += BinarySection.FileStartVA;
|
||||
}
|
||||
LOGMAN_THROW_A_FMT(Block.HostCode < Code.CodeBuffer.size_bytes(), "Host offset {:#x} out of range ({:#x})", Block.HostCode,
|
||||
Code.CodeBuffer.size_bytes());
|
||||
auto HostCode = &Code.CodeBuffer[Block.HostCode];
|
||||
LookupCache.AddBlockMapping(Guest + BinarySection.FileStartVA, std::move(Block.CodePages), HostCode, WriteLock);
|
||||
}
|
||||
|
||||
// Guest code pages
|
||||
auto* Cursor = Code.CodeBufferInFile.data() + Code.CodeBufferInFile.size_bytes();
|
||||
fextl::vector<uint64_t> Entrypoints;
|
||||
for (uint32_t i = 0; i < Code.NumCodePages; ++i) {
|
||||
uint64_t CodePage;
|
||||
memcpy(&CodePage, Cursor, sizeof(CodePage));
|
||||
CodePage += BinarySection.FileStartVA;
|
||||
Cursor += sizeof(CodePage);
|
||||
|
||||
uint64_t NumEntrypoints;
|
||||
memcpy(&NumEntrypoints, Cursor, sizeof(NumEntrypoints));
|
||||
Cursor += sizeof(NumEntrypoints);
|
||||
|
||||
Entrypoints.resize(NumEntrypoints);
|
||||
memcpy(Entrypoints.data(), Cursor, std::span {Entrypoints}.size_bytes());
|
||||
Cursor += std::span {Entrypoints}.size_bytes();
|
||||
for (auto& Entrypoint : Entrypoints) {
|
||||
Entrypoint += BinarySection.FileStartVA;
|
||||
}
|
||||
|
||||
if (LookupCache.AddBlockExecutableRange(Entrypoints, CodePage, FEXCore::Utils::FEX_PAGE_SIZE, WriteLock)) {
|
||||
CTX.SyscallHandler->MarkGuestExecutableRange(Thread, CodePage, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
if (!EnableLazyCodeCaching || EnableCodeCacheValidation) {
|
||||
#else
|
||||
// TODO: Implement lazy mapping on Windows
|
||||
if (true) {
|
||||
#endif
|
||||
auto Range = SelectCodeRangeToFinalize(Code, 0, Code.CodeBuffer.size_bytes() / Utils::FEX_PAGE_SIZE);
|
||||
FinalizeCodePages(Code, Range);
|
||||
}
|
||||
|
||||
if (EnableCodeCacheValidation) {
|
||||
fextl::set<uint64_t> GuestBlocks, HostBlocks;
|
||||
for (auto& [Guest, Host] : BlockList) {
|
||||
GuestBlocks.insert(Guest + BinarySection.FileStartVA);
|
||||
HostBlocks.insert(Host.HostCode);
|
||||
}
|
||||
|
||||
Validate(BinarySection, std::move(GuestBlocks), HostBlocks, Code.CodeBuffer);
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
} // namespace FEXCore::Context
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
static std::span<CPU::Relocation> SpanPageRelocations(const MappedCodeCacheFile& Code, size_t PageIndex) {
|
||||
auto [Offset, Count] = Code.PageRelocationRanges.at(PageIndex);
|
||||
return std::span {reinterpret_cast<FEXCore::CPU::Relocation*>(Code.MappedFile.data() + Offset), Count};
|
||||
}
|
||||
|
||||
std::span<std::byte> AbstractCodeCache::SelectCodeRangeToFinalize(MappedCodeCacheFile& Code, size_t StartPage, size_t EndPage) {
|
||||
// First, check if we were racing another thread in loading this range
|
||||
if (std::find(Code.LoadedPages.begin() + StartPage, Code.LoadedPages.begin() + EndPage, false) == Code.LoadedPages.begin() + EndPage) {
|
||||
return {};
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(StartPage < EndPage, "Invalid page range [{}, {})", StartPage, EndPage);
|
||||
LOGMAN_THROW_A_FMT(EndPage <= Code.NumPages(), "End page {} out of range ({})", EndPage, Code.NumPages());
|
||||
|
||||
// Include any pages that have relocations or block link records crossing
|
||||
// into the current page range. This ensures we don't attempt to finalize
|
||||
// any page twice, partially apply FEX relocations, or trigger page loads
|
||||
// during block linking.
|
||||
while (EndPage < Code.NumPages()) {
|
||||
auto PageRelocs = SpanPageRelocations(Code, EndPage - 1);
|
||||
if (!PageRelocs.empty()) {
|
||||
auto It = std::prev(PageRelocs.end());
|
||||
size_t RelocEnd = It->Header.Offset + 16 /* Upper bound for relocation size */;
|
||||
if (RelocEnd > EndPage * Utils::FEX_PAGE_SIZE) {
|
||||
++EndPage;
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
// Check for trailing block link
|
||||
{
|
||||
auto PageRelocs = SpanPageRelocations(Code, EndPage);
|
||||
if (!PageRelocs.empty() && PageRelocs.begin()->Header.Offset < EndPage * Utils::FEX_PAGE_SIZE + 0x18) {
|
||||
++EndPage;
|
||||
continue;
|
||||
}
|
||||
}
|
||||
break;
|
||||
};
|
||||
while (StartPage != 0) {
|
||||
auto PageRelocs = SpanPageRelocations(Code, StartPage - 1);
|
||||
if (!PageRelocs.empty()) {
|
||||
auto It = std::prev(PageRelocs.end());
|
||||
size_t RelocEnd = It->Header.Offset + 16 /* Upper bound for relocation size */;
|
||||
if (RelocEnd > StartPage * Utils::FEX_PAGE_SIZE) {
|
||||
--StartPage;
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
// Check for trailing block link
|
||||
{
|
||||
auto PageRelocs = SpanPageRelocations(Code, StartPage);
|
||||
if (!PageRelocs.empty() && PageRelocs.begin()->Header.Offset < StartPage * Utils::FEX_PAGE_SIZE + 0x18) {
|
||||
--StartPage;
|
||||
continue;
|
||||
}
|
||||
}
|
||||
break;
|
||||
};
|
||||
|
||||
return Code.CodeBuffer.subspan(StartPage * Utils::FEX_PAGE_SIZE, (EndPage - StartPage) * Utils::FEX_PAGE_SIZE);
|
||||
}
|
||||
} // namespace FEXCore
|
||||
|
||||
namespace FEXCore::Context {
|
||||
|
||||
void CodeCache::FinalizeCodePages(MappedCodeCacheFile& Code, std::span<std::byte> CodeRange) {
|
||||
const size_t StartOffset = CodeRange.data() - Code.CodeBuffer.data();
|
||||
const auto StartPage = StartOffset / Utils::FEX_PAGE_SIZE;
|
||||
const auto EndPage = StartPage + CodeRange.size_bytes() / Utils::FEX_PAGE_SIZE;
|
||||
const size_t Size = CodeRange.size_bytes();
|
||||
|
||||
// None of the selected pages should be loaded at all; otherwise, SelectCodeRangeToFinalize returned inconsistent ranges
|
||||
LOGMAN_THROW_A_FMT(std::find(Code.LoadedPages.begin() + StartPage, Code.LoadedPages.begin() + EndPage, true) == Code.LoadedPages.begin() + EndPage,
|
||||
"Inconsistent page load state");
|
||||
|
||||
FEXCORE_PROFILE_SCOPED("FinalizeCodePages");
|
||||
|
||||
#ifndef _WIN32
|
||||
// Atomicity is critical when making the finalized code data visible.
|
||||
// We ensure this by remapping a temporary buffer onto the PROT_NONE
|
||||
// placeholder page in CodeBuffer. Some constraints to keep in mind are:
|
||||
// 1. Pages can't be write-only (readability is implicitly added), so
|
||||
// we can't change CodeBuffer from PROT_NONE to PROT_WRITE even for just
|
||||
// a short duration
|
||||
// 2. Naive mremap from CodeBufferInFile to CodeBuffer would leave a gap in
|
||||
// the former, which would make cleanup overly complicated
|
||||
//
|
||||
// Due to (1), we can't apply relocations in place (CodeBufferInFile); at
|
||||
// least a secondary buffer is needed for execution (CodeBuffer).
|
||||
// Due to (2), a third buffer is temporarily allocated here and freed on
|
||||
// completion. The final code data is computed here and then the memory
|
||||
// is remapped onto CodeBuffer.
|
||||
auto* Staging = reinterpret_cast<std::byte*>(Allocator::VirtualAlloc(nullptr, Size, true));
|
||||
if (!Staging) {
|
||||
ERROR_AND_DIE_FMT("Failed to allocate {} bytes of staging memory for code-cache finalization", Size);
|
||||
}
|
||||
|
||||
// Copy code from the cache file to the staging buffer
|
||||
memcpy(Staging, Code.CodeBufferInFile.data() + StartOffset, Size);
|
||||
|
||||
// Apply relocations
|
||||
auto StagingSpan = std::span {Staging, Size};
|
||||
for (size_t i = StartPage; i < EndPage; ++i) {
|
||||
auto PageRelocations = SpanPageRelocations(Code, i);
|
||||
(void)ApplyCodeRelocations(Code.GuestBase, StagingSpan, PageRelocations, static_cast<uint32_t>(StartOffset), false);
|
||||
Code.LoadedPages[i] = true;
|
||||
}
|
||||
|
||||
// Atomically make the finalized code data visible by remapping the staging
|
||||
// buffer onto the requested CodeBuffer window. MREMAP_DONTUNMAP is used to
|
||||
// leave the old VA range reserved so that we can cleanly deallocate it
|
||||
// through Allocator.
|
||||
void* RemapResult = ::mremap(Staging, Size, Size, MREMAP_FIXED | MREMAP_MAYMOVE | MREMAP_DONTUNMAP, CodeRange.data());
|
||||
if (RemapResult == MAP_FAILED) {
|
||||
ERROR_AND_DIE_FMT("{}: mremap failed: {}", __FUNCTION__, errno);
|
||||
}
|
||||
Allocator::VirtualFree(Staging, Size);
|
||||
|
||||
// Release resident file pages that will no longer be needed. The VA range is left allocated to allow cleanup with a single VirtualFree.
|
||||
Allocator::VirtualDontNeed(Code.CodeBufferInFile.data() + StartOffset, Size);
|
||||
#else
|
||||
// TODO: Implement lazy mapping on Windows
|
||||
for (size_t i = StartPage; i < EndPage; ++i) {
|
||||
auto PageRelocations = SpanPageRelocations(Code, i);
|
||||
(void)ApplyCodeRelocations(Code.GuestBase, Code.CodeBuffer, PageRelocations, 0, false);
|
||||
Code.LoadedPages[i] = true;
|
||||
}
|
||||
#endif
|
||||
|
||||
ARMEmitter::Emitter::ClearICache(CodeRange.data(), Size);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::Context
|
||||
@@ -358,6 +358,16 @@ bool ContextImpl::InitCore() {
|
||||
Config.NeedsPendingInterruptFaultCheck = true;
|
||||
}
|
||||
|
||||
if constexpr (BLOCK_DEBUGGING) {
|
||||
// If the developer wants to do any single-stepping points or watch points.
|
||||
// Add them here.
|
||||
//
|
||||
// eg:
|
||||
// BlockDebuggerTracker.AllTargetSingleStep();
|
||||
// BlockDebuggerTracker.AddSingleStepTarget(0x14000'0000ULL);
|
||||
// BlockDebuggerTracker.AddWriteWatchPoint(0x420BA5ED);
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -376,7 +386,7 @@ void ContextImpl::ExecuteThread(FEXCore::Core::InternalThreadState* Thread) {
|
||||
}
|
||||
|
||||
void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread) {
|
||||
Thread->OpDispatcher = fextl::make_unique<FEXCore::IR::OpDispatchBuilder>(this);
|
||||
Thread->OpDispatcher = fextl::make_unique<FEXCore::IR::OpDispatchBuilder>(this, Thread);
|
||||
Thread->OpDispatcher->SetMultiblock(Config.Multiblock);
|
||||
Thread->LookupCache = fextl::make_unique<FEXCore::LookupCache>(this);
|
||||
Thread->FrontendDecoder = fextl::make_unique<FEXCore::Frontend::Decoder>(Thread);
|
||||
@@ -456,7 +466,6 @@ void ContextImpl::UnlockAfterFork(FEXCore::Core::InternalThreadState* LiveThread
|
||||
if (Config.StrictInProcessSplitLocks) {
|
||||
FEXCore::Utils::SpinWaitLock::unlock(&StrictSplitLockMutex);
|
||||
}
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -641,6 +650,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
Thread->OpDispatcher->SetTrueJumpTarget(InvalidateCodeCond, CodeWasChangedBlock);
|
||||
|
||||
Thread->OpDispatcher->SetCurrentCodeBlock(CodeWasChangedBlock);
|
||||
Thread->OpDispatcher->StartNewBlock();
|
||||
Thread->OpDispatcher->_ThreadRemoveCodeEntry();
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_InlineEntrypointOffset(GPRSize, InstAddress - GuestRIP));
|
||||
|
||||
@@ -648,6 +658,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
Thread->OpDispatcher->SetFalseJumpTarget(InvalidateCodeCond, NextOpBlock);
|
||||
Thread->OpDispatcher->SetCurrentCodeBlock(NextOpBlock);
|
||||
Thread->OpDispatcher->StartNewBlock();
|
||||
}
|
||||
|
||||
if (TableInfo && TableInfo->OpcodeDispatcher.OpDispatch) {
|
||||
@@ -695,6 +706,8 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
if (Block.BlockStatus == Frontend::Decoder::DecodedBlockStatus::INVALID_INST ||
|
||||
Block.BlockStatus == Frontend::Decoder::DecodedBlockStatus::BAD_RELOCATION) {
|
||||
Thread->OpDispatcher->InvalidOp(DecodedInfo);
|
||||
} else if (Block.BlockStatus == Frontend::Decoder::DecodedBlockStatus::UNIMPLEMENTED_INST) {
|
||||
Thread->OpDispatcher->UnimplementedOp(DecodedInfo);
|
||||
} else {
|
||||
Thread->OpDispatcher->NoExecOp(DecodedInfo);
|
||||
}
|
||||
@@ -818,6 +831,17 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
}
|
||||
|
||||
uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP, uint64_t MaxInst) {
|
||||
if constexpr (BLOCK_DEBUGGING) {
|
||||
// Block debugging logic is hand-written and needs to be handled with care.
|
||||
// Force MaxInst to only be one in this case.
|
||||
MaxInst = 1;
|
||||
|
||||
// If the entrypoint is part of the single step targets then single step it.
|
||||
if (BlockDebuggerTracker.IsSingleStepTarget(GuestRIP)) {
|
||||
return CompileSingleStep(Frame, GuestRIP);
|
||||
}
|
||||
}
|
||||
|
||||
auto Thread = Frame->Thread;
|
||||
FEXCORE_PROFILE_SCOPED("CompileBlock");
|
||||
FEXCORE_PROFILE_ACCUMULATION(Thread, AccumulatedJITTime);
|
||||
|
||||
@@ -153,9 +153,8 @@ void Dispatcher::EmitDispatcher() {
|
||||
ldr(TMP1, ARMEmitter::XReg::x18, TEB_PEB_OFFSET);
|
||||
ldr(TMP1, TMP1, PEB_EC_CODE_BITMAP_OFFSET);
|
||||
|
||||
lsr(ARMEmitter::Size::i64Bit, TMP2, RipReg, 15);
|
||||
and_(ARMEmitter::Size::i64Bit, TMP2, TMP2, 0x1fffffffffff8);
|
||||
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
lsr(ARMEmitter::Size::i64Bit, TMP2, RipReg, 18);
|
||||
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 3);
|
||||
lsr(ARMEmitter::Size::i64Bit, TMP2, RipReg, 12);
|
||||
lsrv(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP2);
|
||||
(void)tbz(TMP1, 0, &l_NotECCode);
|
||||
@@ -552,7 +551,12 @@ void Dispatcher::EmitDispatcher() {
|
||||
EmitF64F2XM1();
|
||||
EmitF64Scale();
|
||||
EmitF64Atan();
|
||||
EmitF64FYL2X();
|
||||
F64Log2Constants Log2C;
|
||||
EmitF64FYL2X(Log2C);
|
||||
EmitF64FYL2XP1(Log2C);
|
||||
EmitF64Log2Constants(Log2C);
|
||||
EmitF64FPREM();
|
||||
EmitF64FPREM1();
|
||||
|
||||
// Interpreter fallbacks
|
||||
{
|
||||
@@ -1605,18 +1609,13 @@ void Dispatcher::EmitF64Atan() {
|
||||
|
||||
// JIT-inlined double-precision y * log2(x) for the F64 reduced precision x87 path.
|
||||
// Input: VTMP1 = x, VTMP2 = y. Output: VTMP1 = y * log2(x).
|
||||
// Algorithm: atanh-based log via s = f/(2+f) with 9-term polynomial, scaled by 1/ln(2),
|
||||
// then multiplied by y.
|
||||
void Dispatcher::EmitF64FYL2X() {
|
||||
// Constants in `C` are shared with EmitF64FYL2XP1 and emitted by EmitF64Log2Constants.
|
||||
void Dispatcher::EmitF64FYL2X(F64Log2Constants& C) {
|
||||
F64FYL2XHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
constexpr auto Accum = ARMEmitter::VReg::v2;
|
||||
|
||||
ARMEmitter::ForwardLabel Fallback;
|
||||
ARMEmitter::ForwardLabel NoNorm;
|
||||
ARMEmitter::ForwardLabel Sqrt2Label, Log2eLabel;
|
||||
ARMEmitter::ForwardLabel BiasLabel;
|
||||
ARMEmitter::ForwardLabel P0Label, P1Label, P2Label, P3Label, P4Label, P5Label, P6Label, P7Label, P8Label;
|
||||
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
@@ -1633,63 +1632,58 @@ void Dispatcher::EmitF64FYL2X() {
|
||||
(void)b(ARMEmitter::Condition::CC_EQ, &Fallback);
|
||||
fmov(ARMEmitter::Size::i64Bit, TMP3, VTMP2.D());
|
||||
|
||||
// Extract k and normalize mantissa m into [1.0, 2.0).
|
||||
// k = unbiased exponent; m bits = mantissa | (0x3FF << 52).
|
||||
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, 1023);
|
||||
ubfx(ARMEmitter::Size::i64Bit, TMP1, TMP1, 0, 52);
|
||||
ldr(TMP4, &BiasLabel);
|
||||
movz(ARMEmitter::Size::i64Bit, TMP4, 0x3FF0, 48);
|
||||
orr(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP4);
|
||||
|
||||
// Index = top 6 mantissa bits (bits 51..46 of m).
|
||||
ubfx(ARMEmitter::Size::i64Bit, TMP4, TMP1, 46, 6);
|
||||
|
||||
// Set m as F64 in VTMP1.
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP1);
|
||||
|
||||
// If m > sqrt(2), halve m and increment k.
|
||||
ldr(VTMP2.D(), &Sqrt2Label);
|
||||
fcmp(VTMP1.D(), VTMP2.D());
|
||||
(void)b(ARMEmitter::Condition::CC_LE, &NoNorm);
|
||||
fmov(ARMEmitter::ScalarRegSize::i64Bit, VTMP2, 0.5f);
|
||||
fmul(VTMP1.D(), VTMP1.D(), VTMP2.D());
|
||||
add(ARMEmitter::Size::i64Bit, TMP2, TMP2, 1);
|
||||
(void)Bind(&NoNorm);
|
||||
// Load (recip, logc) from LUT[index].
|
||||
(void)adr(TMP1, &C.Table);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP4, ARMEmitter::ShiftType::LSL, 4);
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(VTMP2.D(), Accum.D(), TMP1, 0);
|
||||
|
||||
// f = m - 1; s = f / (2 + f); TMP1 stashes s, VTMP1 holds s^2.
|
||||
fmov(ARMEmitter::ScalarRegSize::i64Bit, VTMP2, 1.0f);
|
||||
// Stash logc bits in TMP4 so Accum can be reused for the polynomial.
|
||||
fmov(ARMEmitter::Size::i64Bit, TMP4, Accum.D());
|
||||
|
||||
// r = recip * m - 1.
|
||||
fmul(VTMP1.D(), VTMP2.D(), VTMP1.D());
|
||||
ldr(VTMP2.D(), &C.One);
|
||||
fsub(VTMP1.D(), VTMP1.D(), VTMP2.D());
|
||||
fmov(ARMEmitter::ScalarRegSize::i64Bit, VTMP2, 2.0f);
|
||||
fadd(VTMP2.D(), VTMP1.D(), VTMP2.D());
|
||||
fdiv(Accum.D(), VTMP1.D(), VTMP2.D());
|
||||
fmul(VTMP1.D(), Accum.D(), Accum.D());
|
||||
fmov(ARMEmitter::Size::i64Bit, TMP1, Accum.D());
|
||||
|
||||
// 9-term Horner from 1/19 down to 1/3.
|
||||
ldr(Accum.D(), &P8Label);
|
||||
ldr(VTMP2.D(), &P7Label);
|
||||
// Horner: poly = a0 + r*(a1 + r*(a2 + ... + r*a7)).
|
||||
ldr(Accum.D(), &C.A7);
|
||||
ldr(VTMP2.D(), &C.A6);
|
||||
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
|
||||
ldr(VTMP2.D(), &P6Label);
|
||||
ldr(VTMP2.D(), &C.A5);
|
||||
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
|
||||
ldr(VTMP2.D(), &P5Label);
|
||||
ldr(VTMP2.D(), &C.A4);
|
||||
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
|
||||
ldr(VTMP2.D(), &P4Label);
|
||||
ldr(VTMP2.D(), &C.A3);
|
||||
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
|
||||
ldr(VTMP2.D(), &P3Label);
|
||||
ldr(VTMP2.D(), &C.A2);
|
||||
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
|
||||
ldr(VTMP2.D(), &P2Label);
|
||||
ldr(VTMP2.D(), &C.A1);
|
||||
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
|
||||
ldr(VTMP2.D(), &P1Label);
|
||||
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
|
||||
ldr(VTMP2.D(), &P0Label);
|
||||
ldr(VTMP2.D(), &C.A0);
|
||||
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
|
||||
|
||||
// ln(1+f) = 2 * s * (1 + s^2 * P(s^2)).
|
||||
fmul(Accum.D(), VTMP1.D(), Accum.D());
|
||||
fmov(ARMEmitter::ScalarRegSize::i64Bit, VTMP2, 1.0f);
|
||||
fadd(Accum.D(), Accum.D(), VTMP2.D());
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP2.D(), TMP1);
|
||||
fmul(Accum.D(), VTMP2.D(), Accum.D());
|
||||
fadd(Accum.D(), Accum.D(), Accum.D());
|
||||
// log2(1+r) = r * Accum.
|
||||
fmul(VTMP1.D(), VTMP1.D(), Accum.D());
|
||||
|
||||
// log2(x) = k + ln(1+f)/ln(2); multiply by y.
|
||||
ldr(VTMP1.D(), &Log2eLabel);
|
||||
fmul(VTMP1.D(), Accum.D(), VTMP1.D());
|
||||
// log2(x) = log2(1+r) + log2(center[i]) + k.
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP2.D(), TMP4);
|
||||
fadd(VTMP1.D(), VTMP1.D(), VTMP2.D());
|
||||
scvtf(ARMEmitter::Size::i64Bit, VTMP2.D(), TMP2);
|
||||
fadd(VTMP1.D(), VTMP1.D(), VTMP2.D());
|
||||
|
||||
// result = y * log2(x).
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP2.D(), TMP3);
|
||||
fmul(VTMP1.D(), VTMP1.D(), VTMP2.D());
|
||||
|
||||
@@ -1710,33 +1704,483 @@ void Dispatcher::EmitF64FYL2X() {
|
||||
blr(TMP1);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
ret();
|
||||
}
|
||||
|
||||
// Constant pool: bias=1.0, sqrt(2), log2(e)=1/ln(2), P8..P0 = 1/19, 1/17, 1/15, ..., 1/3.
|
||||
// JIT-inlined double-precision y * log2(1 + x) for the F64 reduced precision x87 path.
|
||||
// Input: VTMP1 = x, VTMP2 = y. Output: VTMP1 = y * log2(1 + x).
|
||||
// Computes v = 1 + x in F64 and runs the same LUT-based log2 as F64FYL2X.
|
||||
// Loses 1-2 ulps of precision near x=0 (FYL2XP1's original purpose) but
|
||||
// matches main's lowering and avoids the range-check cliff into a fallback.
|
||||
void Dispatcher::EmitF64FYL2XP1(F64Log2Constants& C) {
|
||||
F64FYL2XP1HandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
constexpr auto Accum = ARMEmitter::VReg::v2;
|
||||
|
||||
ARMEmitter::ForwardLabel Fallback;
|
||||
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
|
||||
str(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
|
||||
|
||||
// v = 1 + x in Accum so VTMP1/VTMP2 still hold the original x/y at the fallback.
|
||||
ldr(Accum.D(), &C.One);
|
||||
fadd(Accum.D(), VTMP1.D(), Accum.D());
|
||||
|
||||
// Reject v <= 0, subnormal, NaN, Inf via v's bits in TMP1.
|
||||
fmov(ARMEmitter::Size::i64Bit, TMP1, Accum.D());
|
||||
(void)tbnz(TMP1, 63, &Fallback);
|
||||
lsr(ARMEmitter::Size::i64Bit, TMP2, TMP1, 52);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP2, &Fallback);
|
||||
cmp(ARMEmitter::Size::i64Bit, TMP2, 0x7FF);
|
||||
(void)b(ARMEmitter::Condition::CC_EQ, &Fallback);
|
||||
fmov(ARMEmitter::Size::i64Bit, TMP3, VTMP2.D());
|
||||
|
||||
// k = unbiased exponent; m bits = mantissa | (0x3FF << 52).
|
||||
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, 1023);
|
||||
ubfx(ARMEmitter::Size::i64Bit, TMP1, TMP1, 0, 52);
|
||||
movz(ARMEmitter::Size::i64Bit, TMP4, 0x3FF0, 48);
|
||||
orr(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP4);
|
||||
|
||||
// Index = top 6 mantissa bits (bits 51..46 of m).
|
||||
ubfx(ARMEmitter::Size::i64Bit, TMP4, TMP1, 46, 6);
|
||||
|
||||
// Set m as F64 in VTMP1.
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP1);
|
||||
|
||||
// Load (recip, logc) from LUT[index].
|
||||
(void)adr(TMP1, &C.Table);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, TMP1, TMP4, ARMEmitter::ShiftType::LSL, 4);
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(VTMP2.D(), Accum.D(), TMP1, 0);
|
||||
|
||||
// Stash logc bits in TMP4 so Accum can be reused for the polynomial.
|
||||
fmov(ARMEmitter::Size::i64Bit, TMP4, Accum.D());
|
||||
|
||||
// r = recip * m - 1.
|
||||
fmul(VTMP1.D(), VTMP2.D(), VTMP1.D());
|
||||
ldr(VTMP2.D(), &C.One);
|
||||
fsub(VTMP1.D(), VTMP1.D(), VTMP2.D());
|
||||
|
||||
// Horner: poly = a0 + r*(a1 + r*(a2 + ... + r*a7)).
|
||||
ldr(Accum.D(), &C.A7);
|
||||
ldr(VTMP2.D(), &C.A6);
|
||||
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
|
||||
ldr(VTMP2.D(), &C.A5);
|
||||
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
|
||||
ldr(VTMP2.D(), &C.A4);
|
||||
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
|
||||
ldr(VTMP2.D(), &C.A3);
|
||||
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
|
||||
ldr(VTMP2.D(), &C.A2);
|
||||
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
|
||||
ldr(VTMP2.D(), &C.A1);
|
||||
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
|
||||
ldr(VTMP2.D(), &C.A0);
|
||||
fmadd(Accum.D(), VTMP1.D(), Accum.D(), VTMP2.D());
|
||||
|
||||
// log2(1+r) = r * Accum.
|
||||
fmul(VTMP1.D(), VTMP1.D(), Accum.D());
|
||||
|
||||
// log2(v) = log2(1+r) + log2(center[i]) + k.
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP2.D(), TMP4);
|
||||
fadd(VTMP1.D(), VTMP1.D(), VTMP2.D());
|
||||
scvtf(ARMEmitter::Size::i64Bit, VTMP2.D(), TMP2);
|
||||
fadd(VTMP1.D(), VTMP1.D(), VTMP2.D());
|
||||
|
||||
// result = y * log2(v).
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP2.D(), TMP3);
|
||||
fmul(VTMP1.D(), VTMP1.D(), VTMP2.D());
|
||||
|
||||
ldr(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, 16);
|
||||
ret();
|
||||
|
||||
// Fallback path: VTMP1/VTMP2 still hold the original x/y.
|
||||
(void)Bind(&Fallback);
|
||||
ldr(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, 16);
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[FEXCore::Core::OPINDEX_F64FYL2XP1].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[FEXCore::Core::OPINDEX_F64FYL2XP1].Func));
|
||||
blr(TMP1);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
ret();
|
||||
}
|
||||
|
||||
// Shared constants pool for the LUT-based F64 log2 path used by FYL2X and FYL2XP1.
|
||||
// Emitted once after both handlers; their forward `ldr`/`adr` references are
|
||||
// patched here. Layout: One (1.0), 8 Horner coefficients A0..A7, then a 64-entry
|
||||
// LUT of (1/center[i], log2(center[i])) pairs where center[i] = 1 + (i+0.5)/64.
|
||||
void Dispatcher::EmitF64Log2Constants(F64Log2Constants& C) {
|
||||
Align(16);
|
||||
(void)Bind(&BiasLabel);
|
||||
dc64(0x3FF0'0000'0000'0000ULL);
|
||||
(void)Bind(&Sqrt2Label);
|
||||
dc64(0x3FF6'A09E'667F'3BCDULL);
|
||||
(void)Bind(&Log2eLabel);
|
||||
dc64(0x3FF7'1547'652B'82FEULL);
|
||||
(void)Bind(&P8Label);
|
||||
dc64(0x3FAA'F286'BCA1'AF28ULL);
|
||||
(void)Bind(&P7Label);
|
||||
dc64(0x3FAE'1E1E'1E1E'1E1EULL);
|
||||
(void)Bind(&P6Label);
|
||||
dc64(0x3FB1'1111'1111'1111ULL);
|
||||
(void)Bind(&P5Label);
|
||||
dc64(0x3FB3'B13B'13B1'3B14ULL);
|
||||
(void)Bind(&P4Label);
|
||||
dc64(0x3FB7'45D1'745D'1746ULL);
|
||||
(void)Bind(&P3Label);
|
||||
dc64(0x3FBC'71C7'1C71'C71CULL);
|
||||
(void)Bind(&P2Label);
|
||||
dc64(0x3FC2'4924'9249'2492ULL);
|
||||
(void)Bind(&P1Label);
|
||||
dc64(0x3FC9'9999'9999'999AULL);
|
||||
(void)Bind(&P0Label);
|
||||
dc64(0x3FD5'5555'5555'5555ULL);
|
||||
(void)Bind(&C.One);
|
||||
dc64(0x3FF0000000000000ULL); // 1.0
|
||||
(void)Bind(&C.A0);
|
||||
dc64(0x3FF71547652B82FEULL); // log2(e) * 1/1
|
||||
(void)Bind(&C.A1);
|
||||
dc64(0xBFE71547652B82FEULL); // log2(e) * -1/2
|
||||
(void)Bind(&C.A2);
|
||||
dc64(0x3FDEC709DC3A03FDULL); // log2(e) * 1/3
|
||||
(void)Bind(&C.A3);
|
||||
dc64(0xBFD71547652B82FEULL); // log2(e) * -1/4
|
||||
(void)Bind(&C.A4);
|
||||
dc64(0x3FD2776C50EF9BFEULL); // log2(e) * 1/5
|
||||
(void)Bind(&C.A5);
|
||||
dc64(0xBFCEC709DC3A03FDULL); // log2(e) * -1/6
|
||||
(void)Bind(&C.A6);
|
||||
dc64(0x3FCA61762A7ADED9ULL); // log2(e) * 1/7
|
||||
(void)Bind(&C.A7);
|
||||
dc64(0xBFC71547652B82FEULL); // log2(e) * -1/8
|
||||
|
||||
Align(16);
|
||||
(void)Bind(&C.Table);
|
||||
dc64(0x3FEFC07F01FC07F0ULL);
|
||||
dc64(0x3F86FE50B6EF0851ULL); // i= 0
|
||||
dc64(0x3FEF44659E4A4271ULL);
|
||||
dc64(0x3FA11CD1D5133413ULL); // i= 1
|
||||
dc64(0x3FEECC07B301ECC0ULL);
|
||||
dc64(0x3FAC4DFAB90AAB5FULL); // i= 2
|
||||
dc64(0x3FEE573AC901E574ULL);
|
||||
dc64(0x3FB3AA2FDD27F1C3ULL); // i= 3
|
||||
dc64(0x3FEDE5D6E3F8868AULL);
|
||||
dc64(0x3FB918A16E46335BULL); // i= 4
|
||||
dc64(0x3FED77B654B82C34ULL);
|
||||
dc64(0x3FBE72EC117FA5B2ULL); // i= 5
|
||||
dc64(0x3FED0CB58F6EC074ULL);
|
||||
dc64(0x3FC1DCD197552B7BULL); // i= 6
|
||||
dc64(0x3FECA4B3055EE191ULL);
|
||||
dc64(0x3FC476A9F983F74DULL); // i= 7
|
||||
dc64(0x3FEC3F8F01C3F8F0ULL);
|
||||
dc64(0x3FC70742D4EF027FULL); // i= 8
|
||||
dc64(0x3FEBDD2B899406F7ULL);
|
||||
dc64(0x3FC98EDD077E70DFULL); // i= 9
|
||||
dc64(0x3FEB7D6C3DDA338BULL);
|
||||
dc64(0x3FCC0DB6CDD94DEEULL); // i=10
|
||||
dc64(0x3FEB2036406C80D9ULL);
|
||||
dc64(0x3FCE840BE74E6A4DULL); // i=11
|
||||
dc64(0x3FEAC5701AC5701BULL);
|
||||
dc64(0x3FD0790ADBB03009ULL); // i=12
|
||||
dc64(0x3FEA6D01A6D01A6DULL);
|
||||
dc64(0x3FD1AC05B291F070ULL); // i=13
|
||||
dc64(0x3FEA16D3F97A4B02ULL);
|
||||
dc64(0x3FD2DB10FC4D9AAFULL); // i=14
|
||||
dc64(0x3FE9C2D14EE4A102ULL);
|
||||
dc64(0x3FD406463B1B0449ULL); // i=15
|
||||
dc64(0x3FE970E4F80CB872ULL);
|
||||
dc64(0x3FD52DBDFC4C96B3ULL); // i=16
|
||||
dc64(0x3FE920FB49D0E229ULL);
|
||||
dc64(0x3FD6518FE4677BA7ULL); // i=17
|
||||
dc64(0x3FE8D3018D3018D3ULL);
|
||||
dc64(0x3FD771D2BA7EFB3CULL); // i=18
|
||||
dc64(0x3FE886E5F0ABB04AULL);
|
||||
dc64(0x3FD88E9C72E0B226ULL); // i=19
|
||||
dc64(0x3FE83C977AB2BEDDULL);
|
||||
dc64(0x3FD9A802391E232FULL); // i=20
|
||||
dc64(0x3FE7F405FD017F40ULL);
|
||||
dc64(0x3FDABE18797F1F49ULL); // i=21
|
||||
dc64(0x3FE7AD2208E0ECC3ULL);
|
||||
dc64(0x3FDBD0F2E9E79031ULL); // i=22
|
||||
dc64(0x3FE767DCE434A9B1ULL);
|
||||
dc64(0x3FDCE0A4923A587DULL); // i=23
|
||||
dc64(0x3FE724287F46DEBCULL);
|
||||
dc64(0x3FDDED3FD442364CULL); // i=24
|
||||
dc64(0x3FE6E1F76B4337C7ULL);
|
||||
dc64(0x3FDEF6D67328E220ULL); // i=25
|
||||
dc64(0x3FE6A13CD1537290ULL);
|
||||
dc64(0x3FDFFD799A83FF9BULL); // i=26
|
||||
dc64(0x3FE661EC6A5122F9ULL);
|
||||
dc64(0x3FE0809CF27F703DULL); // i=27
|
||||
dc64(0x3FE623FA77016240ULL);
|
||||
dc64(0x3FE10113B153C8EAULL); // i=28
|
||||
dc64(0x3FE5E75BB8D015E7ULL);
|
||||
dc64(0x3FE18028CF72976AULL); // i=29
|
||||
dc64(0x3FE5AC056B015AC0ULL);
|
||||
dc64(0x3FE1FDE3D30E8126ULL); // i=30
|
||||
dc64(0x3FE571ED3C506B3AULL);
|
||||
dc64(0x3FE27A4C0585CBF8ULL); // i=31
|
||||
dc64(0x3FE5390948F40FEBULL);
|
||||
dc64(0x3FE2F56875EB3F26ULL); // i=32
|
||||
dc64(0x3FE5015015015015ULL);
|
||||
dc64(0x3FE36F3FFB6D9162ULL); // i=33
|
||||
dc64(0x3FE4CAB88725AF6EULL);
|
||||
dc64(0x3FE3E7D9379F7016ULL); // i=34
|
||||
dc64(0x3FE49539E3B2D067ULL);
|
||||
dc64(0x3FE45F3A98A20739ULL); // i=35
|
||||
dc64(0x3FE460CBC7F5CF9AULL);
|
||||
dc64(0x3FE4D56A5B33CEC4ULL); // i=36
|
||||
dc64(0x3FE42D6625D51F87ULL);
|
||||
dc64(0x3FE54A6E8CA5438EULL); // i=37
|
||||
dc64(0x3FE3FB013FB013FBULL);
|
||||
dc64(0x3FE5BE4D0CB51435ULL); // i=38
|
||||
dc64(0x3FE3C995A47BABE7ULL);
|
||||
dc64(0x3FE6310B8F553048ULL); // i=39
|
||||
dc64(0x3FE3991C2C187F63ULL);
|
||||
dc64(0x3FE6A2AF9E5A0F0AULL); // i=40
|
||||
dc64(0x3FE3698DF3DE0748ULL);
|
||||
dc64(0x3FE7133E9B156C7CULL); // i=41
|
||||
dc64(0x3FE33AE45B57BCB2ULL);
|
||||
dc64(0x3FE782BDBFDDA657ULL); // i=42
|
||||
dc64(0x3FE30D190130D190ULL);
|
||||
dc64(0x3FE7F1322182CF16ULL); // i=43
|
||||
dc64(0x3FE2E025C04B8097ULL);
|
||||
dc64(0x3FE85EA0B0B27B26ULL); // i=44
|
||||
dc64(0x3FE2B404AD012B40ULL);
|
||||
dc64(0x3FE8CB0E3B4B3BBEULL); // i=45
|
||||
dc64(0x3FE288B01288B013ULL);
|
||||
dc64(0x3FE9367F6DA0AB2FULL); // i=46
|
||||
dc64(0x3FE25E22708092F1ULL);
|
||||
dc64(0x3FE9A0F8D3B0E050ULL); // i=47
|
||||
dc64(0x3FE23456789ABCDFULL);
|
||||
dc64(0x3FEA0A7EDA4C112DULL); // i=48
|
||||
dc64(0x3FE20B470C67C0D9ULL);
|
||||
dc64(0x3FEA7315D02F20C8ULL); // i=49
|
||||
dc64(0x3FE1E2EF3B3FB874ULL);
|
||||
dc64(0x3FEADAC1E711C833ULL); // i=50
|
||||
dc64(0x3FE1BB4A4046ED29ULL);
|
||||
dc64(0x3FEB418734A9008CULL); // i=51
|
||||
dc64(0x3FE19453808CA29CULL);
|
||||
dc64(0x3FEBA769B39E4964ULL); // i=52
|
||||
dc64(0x3FE16E0689427379ULL);
|
||||
dc64(0x3FEC0C6D447C5DD3ULL); // i=53
|
||||
dc64(0x3FE1485F0E0ACD3BULL);
|
||||
dc64(0x3FEC7095AE91E1C7ULL); // i=54
|
||||
dc64(0x3FE12358E75D3033ULL);
|
||||
dc64(0x3FECD3E6A0CA8907ULL); // i=55
|
||||
dc64(0x3FE0FEF010FEF011ULL);
|
||||
dc64(0x3FED3663B27F31D5ULL); // i=56
|
||||
dc64(0x3FE0DB20A88F4696ULL);
|
||||
dc64(0x3FED9810643D6615ULL); // i=57
|
||||
dc64(0x3FE0B7E6EC259DC8ULL);
|
||||
dc64(0x3FEDF8F02086AF2CULL); // i=58
|
||||
dc64(0x3FE0953F39010954ULL);
|
||||
dc64(0x3FEE59063C8822CEULL); // i=59
|
||||
dc64(0x3FE073260A47F7C6ULL);
|
||||
dc64(0x3FEEB855F8CA88FBULL); // i=60
|
||||
dc64(0x3FE05197F7D73404ULL);
|
||||
dc64(0x3FEF16E281DB7630ULL); // i=61
|
||||
dc64(0x3FE03091B51F5E1AULL);
|
||||
dc64(0x3FEF74AEF0EFAFAEULL); // i=62
|
||||
dc64(0x3FE0101010101010ULL);
|
||||
dc64(0x3FEFD1BE4C7F2AF9ULL); // i=63
|
||||
}
|
||||
|
||||
void Dispatcher::EmitF64FPREM() {
|
||||
// JIT-inlined double-precision FPREM (C-library style truncated remainder).
|
||||
// Input: VTMP1 = dividend (src1), VTMP2 = divisor (src2). Output: VTMP1.
|
||||
F64FPREMHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
constexpr auto Scratch = ARMEmitter::VReg::v2;
|
||||
ARMEmitter::ForwardLabel ReturnX;
|
||||
ARMEmitter::ForwardLabel MaybeExact;
|
||||
ARMEmitter::ForwardLabel Fallback;
|
||||
ARMEmitter::ForwardLabel NonZeroResult;
|
||||
|
||||
// Save q2.
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
// save nzcv
|
||||
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
|
||||
str(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
|
||||
|
||||
fmov(ARMEmitter::Size::i64Bit, TMP3, VTMP1.D());
|
||||
fmov(ARMEmitter::Size::i64Bit, TMP4, VTMP2.D());
|
||||
|
||||
fdiv(Scratch.D(), VTMP1.D(), VTMP2.D());
|
||||
frintz(VTMP1.D(), Scratch.D());
|
||||
|
||||
fmov(ARMEmitter::Size::i64Bit, TMP1, VTMP1.D());
|
||||
lsl(ARMEmitter::Size::i64Bit, TMP2, TMP1, 1);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP2, &ReturnX);
|
||||
|
||||
fmov(ARMEmitter::Size::i64Bit, TMP2, Scratch.D());
|
||||
ubfx(ARMEmitter::Size::i64Bit, TMP2, TMP2, 52, 11);
|
||||
cmp(ARMEmitter::Size::i64Bit, TMP2, 0x7FF);
|
||||
(void)b(ARMEmitter::Condition::CC_EQ, &Fallback);
|
||||
cmp(ARMEmitter::Size::i64Bit, TMP2, 1076);
|
||||
(void)b(ARMEmitter::Condition::CC_HS, &Fallback);
|
||||
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP1);
|
||||
fsub(VTMP2.D(), Scratch.D(), VTMP1.D());
|
||||
fabs(VTMP2.D(), VTMP2.D());
|
||||
|
||||
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, 53);
|
||||
lsl(ARMEmitter::Size::i64Bit, TMP2, TMP2, 52);
|
||||
fmov(ARMEmitter::Size::i64Bit, Scratch.D(), TMP2);
|
||||
fcmp(VTMP2.D(), Scratch.D());
|
||||
// err <= 0.5 ULP could mean fdiv rounded across an integer boundary; defer to MaybeExact
|
||||
// which distinguishes the (safe) exact-quotient case from the (unsafe) rounded case.
|
||||
(void)b(ARMEmitter::Condition::CC_LS, &MaybeExact);
|
||||
|
||||
fmov(ARMEmitter::ScalarRegSize::i64Bit, Scratch, 1.0f);
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP2);
|
||||
fsub(Scratch.D(), Scratch.D(), VTMP1.D());
|
||||
fcmp(VTMP2.D(), Scratch.D());
|
||||
(void)b(ARMEmitter::Condition::CC_HS, &Fallback);
|
||||
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP1);
|
||||
fmov(ARMEmitter::Size::i64Bit, Scratch.D(), TMP4);
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP2.D(), TMP3);
|
||||
fmsub(VTMP2.D(), VTMP1.D(), Scratch.D(), VTMP2.D());
|
||||
|
||||
fmov(ARMEmitter::Size::i64Bit, TMP2, VTMP2.D());
|
||||
lsl(ARMEmitter::Size::i64Bit, TMP2, TMP2, 1);
|
||||
(void)cbnz(ARMEmitter::Size::i64Bit, TMP2, &NonZeroResult);
|
||||
lsr(ARMEmitter::Size::i64Bit, TMP2, TMP3, 63);
|
||||
lsl(ARMEmitter::Size::i64Bit, TMP2, TMP2, 63);
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP2.D(), TMP2);
|
||||
(void)Bind(&NonZeroResult);
|
||||
|
||||
fmov(VTMP1.D(), VTMP2.D());
|
||||
|
||||
// restore nzcv
|
||||
ldr(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, 16);
|
||||
ret();
|
||||
|
||||
(void)Bind(&ReturnX);
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP3);
|
||||
|
||||
// restore nzcv
|
||||
ldr(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, 16);
|
||||
ret();
|
||||
|
||||
// err <= 0.5 ULP. Two cases:
|
||||
// err == 0: x/y rounded to an exact integer. Either truly exact (q is right) or fdiv
|
||||
// rounded a near-integer onto an integer (q is off by one). Verify with
|
||||
// fmsub(q, y, x): if exactly zero, x == q*y exactly => result is sign(x)*0.
|
||||
// err > 0: q_fp is genuinely between integers but within rounding slack; fall back.
|
||||
(void)Bind(&MaybeExact);
|
||||
fmov(ARMEmitter::Size::i64Bit, TMP2, VTMP2.D());
|
||||
(void)cbnz(ARMEmitter::Size::i64Bit, TMP2, &Fallback);
|
||||
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP1);
|
||||
fmov(ARMEmitter::Size::i64Bit, Scratch.D(), TMP4);
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP2.D(), TMP3);
|
||||
fmsub(VTMP2.D(), VTMP1.D(), Scratch.D(), VTMP2.D());
|
||||
|
||||
fmov(ARMEmitter::Size::i64Bit, TMP2, VTMP2.D());
|
||||
lsl(ARMEmitter::Size::i64Bit, TMP2, TMP2, 1);
|
||||
(void)cbnz(ARMEmitter::Size::i64Bit, TMP2, &Fallback);
|
||||
|
||||
lsr(ARMEmitter::Size::i64Bit, TMP2, TMP3, 63);
|
||||
lsl(ARMEmitter::Size::i64Bit, TMP2, TMP2, 63);
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP2);
|
||||
|
||||
// restore nzcv
|
||||
ldr(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, 16);
|
||||
ret();
|
||||
|
||||
(void)Bind(&Fallback);
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP3);
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP2.D(), TMP4);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, 16);
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[FEXCore::Core::OPINDEX_F64FPREM].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[FEXCore::Core::OPINDEX_F64FPREM].Func));
|
||||
blr(TMP1);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
ret();
|
||||
}
|
||||
|
||||
void Dispatcher::EmitF64FPREM1() {
|
||||
// JIT-inlined double-precision FPREM1 (IEEE round-to-nearest remainder).
|
||||
// Input: VTMP1 = dividend (src1), VTMP2 = divisor (src2). Output: VTMP1.
|
||||
F64FPREM1HandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
constexpr auto Scratch = ARMEmitter::VReg::v2;
|
||||
ARMEmitter::ForwardLabel ReturnX;
|
||||
ARMEmitter::ForwardLabel Fallback;
|
||||
ARMEmitter::ForwardLabel NonZeroResult;
|
||||
|
||||
// Save q2.
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, -16);
|
||||
|
||||
// save nzcv
|
||||
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
|
||||
str(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
|
||||
|
||||
fmov(ARMEmitter::Size::i64Bit, TMP3, VTMP1.D());
|
||||
fmov(ARMEmitter::Size::i64Bit, TMP4, VTMP2.D());
|
||||
|
||||
fdiv(Scratch.D(), VTMP1.D(), VTMP2.D());
|
||||
frintn(VTMP1.D(), Scratch.D());
|
||||
|
||||
fmov(ARMEmitter::Size::i64Bit, TMP1, VTMP1.D());
|
||||
lsl(ARMEmitter::Size::i64Bit, TMP2, TMP1, 1);
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP2, &ReturnX);
|
||||
|
||||
fmov(ARMEmitter::Size::i64Bit, TMP2, Scratch.D());
|
||||
ubfx(ARMEmitter::Size::i64Bit, TMP2, TMP2, 52, 11);
|
||||
cmp(ARMEmitter::Size::i64Bit, TMP2, 0x7FF);
|
||||
(void)b(ARMEmitter::Condition::CC_EQ, &Fallback);
|
||||
cmp(ARMEmitter::Size::i64Bit, TMP2, 1076);
|
||||
(void)b(ARMEmitter::Condition::CC_HS, &Fallback);
|
||||
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP1);
|
||||
fsub(VTMP2.D(), Scratch.D(), VTMP1.D());
|
||||
fabs(VTMP2.D(), VTMP2.D());
|
||||
|
||||
sub(ARMEmitter::Size::i64Bit, TMP2, TMP2, 53);
|
||||
lsl(ARMEmitter::Size::i64Bit, TMP2, TMP2, 52);
|
||||
fmov(ARMEmitter::ScalarRegSize::i64Bit, Scratch, 0.5f);
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP2);
|
||||
fsub(Scratch.D(), Scratch.D(), VTMP1.D());
|
||||
fcmp(VTMP2.D(), Scratch.D());
|
||||
(void)b(ARMEmitter::Condition::CC_HS, &Fallback);
|
||||
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP1);
|
||||
fmov(ARMEmitter::Size::i64Bit, Scratch.D(), TMP4);
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP2.D(), TMP3);
|
||||
fmsub(VTMP2.D(), VTMP1.D(), Scratch.D(), VTMP2.D());
|
||||
|
||||
fmov(ARMEmitter::Size::i64Bit, TMP2, VTMP2.D());
|
||||
lsl(ARMEmitter::Size::i64Bit, TMP2, TMP2, 1);
|
||||
(void)cbnz(ARMEmitter::Size::i64Bit, TMP2, &NonZeroResult);
|
||||
lsr(ARMEmitter::Size::i64Bit, TMP2, TMP3, 63);
|
||||
lsl(ARMEmitter::Size::i64Bit, TMP2, TMP2, 63);
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP2.D(), TMP2);
|
||||
(void)Bind(&NonZeroResult);
|
||||
|
||||
fmov(VTMP1.D(), VTMP2.D());
|
||||
|
||||
// restore nzcv
|
||||
ldr(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, 16);
|
||||
ret();
|
||||
|
||||
(void)Bind(&ReturnX);
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP3);
|
||||
|
||||
// restore nzcv
|
||||
ldr(TMP1.W(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.flags[24]));
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, 16);
|
||||
ret();
|
||||
|
||||
(void)Bind(&Fallback);
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP1.D(), TMP3);
|
||||
fmov(ARMEmitter::Size::i64Bit, VTMP2.D(), TMP4);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::QReg::q2, ARMEmitter::Reg::rsp, 16);
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[FEXCore::Core::OPINDEX_F64FPREM1].ABIHandler));
|
||||
ldr(TMP4, STATE_PTR(CpuStateFrame, Pointers.FallbackHandlerPointers[FEXCore::Core::OPINDEX_F64FPREM1].Func));
|
||||
blr(TMP1);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
ret();
|
||||
}
|
||||
|
||||
uint64_t Dispatcher::GenerateABICall(FallbackABI ABI) {
|
||||
@@ -2190,6 +2634,9 @@ void Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState* Thread)
|
||||
Ptrs.F64ScaleHandler = F64ScaleHandlerAddress;
|
||||
Ptrs.F64AtanHandler = F64AtanHandlerAddress;
|
||||
Ptrs.F64FYL2XHandler = F64FYL2XHandlerAddress;
|
||||
Ptrs.F64FYL2XP1Handler = F64FYL2XP1HandlerAddress;
|
||||
Ptrs.F64FPREMHandler = F64FPREMHandlerAddress;
|
||||
Ptrs.F64FPREM1Handler = F64FPREM1HandlerAddress;
|
||||
|
||||
// Fill in the fallback handlers
|
||||
InterpreterOps::FillFallbackIndexPointers(Ptrs.FallbackHandlerPointers, &ABIPointers[0]);
|
||||
|
||||
@@ -107,6 +107,9 @@ private:
|
||||
uint64_t F64ScaleHandlerAddress {};
|
||||
uint64_t F64AtanHandlerAddress {};
|
||||
uint64_t F64FYL2XHandlerAddress {};
|
||||
uint64_t F64FYL2XP1HandlerAddress {};
|
||||
uint64_t F64FPREMHandlerAddress {};
|
||||
uint64_t F64FPREM1HandlerAddress {};
|
||||
|
||||
void EmitDispatcher();
|
||||
uint64_t GenerateABICall(FallbackABI ABI);
|
||||
@@ -118,13 +121,25 @@ private:
|
||||
void EmitF32ToExtF80();
|
||||
void EmitF64ToExtF80();
|
||||
|
||||
// Shared label set for the LUT-based F64 log2 path used by both FYL2X and
|
||||
// FYL2XP1. The pool is emitted once via EmitF64Log2Constants.
|
||||
struct F64Log2Constants {
|
||||
ARMEmitter::ForwardLabel One;
|
||||
ARMEmitter::ForwardLabel A0, A1, A2, A3, A4, A5, A6, A7;
|
||||
ARMEmitter::ForwardLabel Table;
|
||||
};
|
||||
|
||||
void EmitF64Sin();
|
||||
void EmitF64Cos();
|
||||
void EmitF64Tan();
|
||||
void EmitF64F2XM1();
|
||||
void EmitF64Scale();
|
||||
void EmitF64Atan();
|
||||
void EmitF64FYL2X();
|
||||
void EmitF64FYL2X(F64Log2Constants& C);
|
||||
void EmitF64FYL2XP1(F64Log2Constants& C);
|
||||
void EmitF64Log2Constants(F64Log2Constants& C);
|
||||
void EmitF64FPREM();
|
||||
void EmitF64FPREM1();
|
||||
|
||||
FEX_CONFIG_OPT(DisableL2Cache, DISABLEL2CACHE);
|
||||
};
|
||||
|
||||
@@ -124,9 +124,9 @@ uint8_t Decoder::ReadByte() {
|
||||
}
|
||||
|
||||
std::optional<uint8_t> Decoder::PeekByte(uint8_t Offset) {
|
||||
uint64_t ByteAddress = reinterpret_cast<uint64_t>(InstStream + InstructionSize + Offset);
|
||||
uint64_t ByteAddress = reinterpret_cast<uint64_t>(InstStream.InstStream + InstructionSize + Offset);
|
||||
if (CheckRangeExecutable(ByteAddress, 1)) {
|
||||
return InstStream[InstructionSize + Offset];
|
||||
return InstStream.AdjustedInstStream[InstructionSize + Offset];
|
||||
} else {
|
||||
return std::nullopt;
|
||||
}
|
||||
@@ -136,9 +136,9 @@ std::pair<uint64_t, bool> Decoder::ReadData(uint8_t Size) {
|
||||
LOGMAN_THROW_A_FMT(Size != 0 && Size <= sizeof(uint64_t), "Unknown data size to read");
|
||||
|
||||
uint64_t Res = 0;
|
||||
uint64_t Address = reinterpret_cast<uint64_t>(InstStream + InstructionSize);
|
||||
uint64_t Address = reinterpret_cast<uint64_t>(InstStream.InstStream + InstructionSize);
|
||||
if (CheckRangeExecutable(Address, Size)) {
|
||||
std::memcpy(&Res, &InstStream[InstructionSize], Size);
|
||||
std::memcpy(&Res, &InstStream.AdjustedInstStream[InstructionSize], Size);
|
||||
} else {
|
||||
HitNonExecutableRange = true;
|
||||
// See PeekByte, this specific case may cause some executable memory to read as 0 but it doesn't matter as the entire instruction will be rolled back anyway.
|
||||
@@ -342,7 +342,7 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand* Operand, X86Tables::ModR
|
||||
}
|
||||
}
|
||||
|
||||
bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op, DecodedHeader Options) {
|
||||
Decoder::DecodedBlockStatus Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op, DecodedHeader Options) {
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_ARCH_DISPATCHER) [[unlikely]] {
|
||||
// Dispatcher Op.
|
||||
// TODO: Move this in to `NormalOpHeader`, Dispatch tables have a bug currently where some subtables don't inherit flags correctly.
|
||||
@@ -354,11 +354,16 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
DecodeInst->TableInfo = Info;
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
|
||||
return false;
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_INVALID) {
|
||||
return false;
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
}
|
||||
|
||||
if (!(Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SUPPORTS_LOCK) && (DecodeInst->Flags & DecodeFlags::FLAG_LOCK)) {
|
||||
// Instruction has lock prefix but doesn't support lock.
|
||||
return DecodedBlockStatus::UNIMPLEMENTED_INST;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(!(Info->Type >= FEXCore::X86Tables::TYPE_GROUP_1 && Info->Type <= FEXCore::X86Tables::TYPE_GROUP_P), "Group Ops "
|
||||
@@ -390,15 +395,15 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
const bool Has16BitAddressing = !BlockInfo.Is64BitMode && DecodeInst->Flags & DecodeFlags::FLAG_ADDRESS_SIZE;
|
||||
|
||||
if (Options.w && (Info->Flags & InstFlags::FLAGS_REX_W_0)) {
|
||||
return false;
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
} else if (!Options.w && (Info->Flags & InstFlags::FLAGS_REX_W_1)) {
|
||||
return false;
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
}
|
||||
|
||||
if (Options.L && (Info->Flags & InstFlags::FLAGS_VEX_L_0)) {
|
||||
return false;
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
} else if (!Options.L && (Info->Flags & InstFlags::FLAGS_VEX_L_1)) {
|
||||
return false;
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
}
|
||||
|
||||
const bool UseVEXL = Options.L && !(Info->Flags & InstFlags::FLAGS_VEX_L_IGNORE);
|
||||
@@ -507,7 +512,7 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, Op & 0b111, Is8BitDest, HasREX, false, false);
|
||||
|
||||
if (CurrentDest->Data.GPR.GPR == FEXCore::X86State::REG_INVALID) {
|
||||
return false;
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -576,7 +581,7 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
|
||||
const auto VEXOperand = Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_VEX_SRC_MASK;
|
||||
if (VEXOperand == FEXCore::X86Tables::InstFlags::FLAGS_VEX_NO_OPERAND && Options.vvvv) {
|
||||
return false;
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
}
|
||||
|
||||
if (VEXOperand == FEXCore::X86Tables::InstFlags::FLAGS_VEX_1ST_SRC) {
|
||||
@@ -594,11 +599,11 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
if (Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_MODRM) {
|
||||
if (Info->Flags & FEXCore::X86Tables::InstFlags::FLAGS_SF_MOD_DST) {
|
||||
if (!ModRMOperand(DecodeInst->Src[CurrentSrc], DecodeInst->Dest, HasXMMSrc, HasXMMDst, HasMMSrc, HasMMDst, Is8BitSrc, Is8BitDest)) {
|
||||
return false;
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
}
|
||||
} else {
|
||||
if (!ModRMOperand(DecodeInst->Dest, DecodeInst->Src[CurrentSrc], HasXMMDst, HasXMMSrc, HasMMDst, HasMMSrc, Is8BitDest, Is8BitSrc)) {
|
||||
return false;
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
}
|
||||
}
|
||||
++CurrentSrc;
|
||||
@@ -660,22 +665,27 @@ bool Decoder::NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op,
|
||||
Bytes = 0;
|
||||
}
|
||||
|
||||
if ((DecodeInst->Flags & DecodeFlags::FLAG_LOCK) && DecodeInst->Dest.IsGPR()) {
|
||||
// Instruction has lock prefix, but the destination isn't memory, this is invalid.
|
||||
return DecodedBlockStatus::UNIMPLEMENTED_INST;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Bytes == 0, "Inst at 0x{:x}: 0x{:04x} '{}' Had an instruction of size {} with {} remaining", DecodeInst->PC,
|
||||
DecodeInst->OP, DecodeInst->TableInfo->Name ?: "UND", InstructionSize, Bytes);
|
||||
DecodeInst->InstSize = InstructionSize;
|
||||
return true;
|
||||
return DecodedBlockStatus::SUCCESS;
|
||||
}
|
||||
|
||||
bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op) {
|
||||
Decoder::DecodedBlockStatus Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op) {
|
||||
DecodeInst->OPRaw = DecodeInst->OP = Op;
|
||||
DecodeInst->TableInfo = Info;
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_UNKNOWN) {
|
||||
return false;
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
}
|
||||
|
||||
if (Info->Type == FEXCore::X86Tables::TYPE_INVALID) {
|
||||
return false;
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_A_FMT(Info->Type != FEXCore::X86Tables::TYPE_REX_PREFIX, "REX PREFIX should have been decoded before this!");
|
||||
@@ -732,7 +742,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
};
|
||||
uint8_t Field = RegToField[ModRM.reg];
|
||||
if (Field == 255) {
|
||||
return false;
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
}
|
||||
|
||||
LocalOp = (Field << 3) | ModRM.rm;
|
||||
@@ -751,7 +761,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
} else if (Info->Type == FEXCore::X86Tables::TYPE_VEX_TABLE_PREFIX) {
|
||||
if (!VEXTable) {
|
||||
// AVX not enabled.
|
||||
return false;
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
}
|
||||
|
||||
uint16_t map_select = 1;
|
||||
@@ -761,7 +771,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
|
||||
if ((Byte1 & 0b10000000) == 0) {
|
||||
if (!BlockInfo.Is64BitMode) {
|
||||
return false;
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
}
|
||||
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_R;
|
||||
@@ -772,7 +782,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
const uint8_t vvvv = ((Byte1 & 0b01111000) >> 3);
|
||||
if (!BlockInfo.Is64BitMode && vvvv <= 0b0111) {
|
||||
// Invalid on 32-bit, can't use the high registers.
|
||||
return false;
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
}
|
||||
options.vvvv = 15 - vvvv;
|
||||
options.L = (Byte1 & 0b100) != 0;
|
||||
@@ -783,14 +793,14 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
const uint8_t vvvv = ((Byte2 & 0b01111000) >> 3);
|
||||
if (!BlockInfo.Is64BitMode && vvvv <= 0b0111) {
|
||||
// Invalid on 32-bit, can't use the high registers.
|
||||
return false;
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
}
|
||||
options.vvvv = 15 - vvvv;
|
||||
options.w = (Byte2 & 0b10000000) != 0;
|
||||
options.L = (Byte2 & 0b100) != 0;
|
||||
if ((Byte1 & 0b01000000) == 0) {
|
||||
if (!BlockInfo.Is64BitMode) {
|
||||
return false;
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
}
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_REX_XGPR_X;
|
||||
}
|
||||
@@ -801,7 +811,7 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
DecodeInst->Flags |= DecodeFlags::FLAG_OPTION_AVX_W;
|
||||
}
|
||||
if (!(map_select >= 1 && map_select <= 3)) {
|
||||
return false;
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -831,14 +841,14 @@ bool Decoder::NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16
|
||||
} else if (Info->Type == FEXCore::X86Tables::TYPE_GROUP_EVEX) {
|
||||
FEXCORE_TELEMETRY_SET(TYPE_USES_EVEX_OPS, 1);
|
||||
// EVEX unsupported
|
||||
return false;
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
}
|
||||
|
||||
LOGMAN_MSG_A_FMT("Invalid instruction decoding type");
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
bool Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
Decoder::DecodedBlockStatus Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
InstructionSize = 0;
|
||||
LastEscapePrefix = 0;
|
||||
Instruction.fill(0);
|
||||
@@ -849,7 +859,7 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
|
||||
for (;;) {
|
||||
if (InstructionSize >= MAX_INST_SIZE) {
|
||||
return false;
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
}
|
||||
uint8_t Op = ReadByte();
|
||||
switch (Op) {
|
||||
@@ -1035,10 +1045,10 @@ bool Decoder::DecodeInstructionImpl(uint64_t PC) {
|
||||
}
|
||||
|
||||
if (DecodeInst->Dest.IsGPR()) {
|
||||
return false;
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
}
|
||||
|
||||
return true;
|
||||
return DecodedBlockStatus::SUCCESS;
|
||||
}
|
||||
|
||||
void Decoder::DecodeREXIfValid(int8_t ExpectedOffset) {
|
||||
@@ -1076,16 +1086,16 @@ Decoder::DecodedBlockStatus Decoder::DecodeInstruction(uint64_t PC) {
|
||||
// Will be set if DecodeInstructionImpl tries to read non-executable memory
|
||||
HitNonExecutableRange = false;
|
||||
HitBadRelocation = false;
|
||||
bool ErrorDuringDecoding = !DecodeInstructionImpl(PC);
|
||||
auto ErrorDuringDecoding = DecodeInstructionImpl(PC);
|
||||
|
||||
if (ErrorDuringDecoding || HitNonExecutableRange || HitBadRelocation) [[unlikely]] {
|
||||
if (ErrorDuringDecoding != DecodedBlockStatus::SUCCESS || HitNonExecutableRange || HitBadRelocation) [[unlikely]] {
|
||||
// Put an invalid instruction in the stream so the core can raise SIGILL if hit
|
||||
// Error while decoding instruction. We don't know the table or instruction size
|
||||
DecodeInst->TableInfo = nullptr;
|
||||
auto Result = ErrorDuringDecoding ? DecodedBlockStatus::INVALID_INST :
|
||||
DecodeInst->InstSize ? DecodedBlockStatus::PARTIAL_DECODE_INST :
|
||||
HitNonExecutableRange ? DecodedBlockStatus::NOEXEC_INST :
|
||||
DecodedBlockStatus::BAD_RELOCATION;
|
||||
auto Result = ErrorDuringDecoding != DecodedBlockStatus::SUCCESS ? ErrorDuringDecoding :
|
||||
DecodeInst->InstSize ? DecodedBlockStatus::PARTIAL_DECODE_INST :
|
||||
HitNonExecutableRange ? DecodedBlockStatus::NOEXEC_INST :
|
||||
DecodedBlockStatus::BAD_RELOCATION;
|
||||
DecodeInst->InstSize = 0;
|
||||
return Result;
|
||||
} else if (!DecodeInst->TableInfo || (DecodeInst->TableInfo->Type == TYPE_INST && !DecodeInst->TableInfo->OpcodeDispatcher.OpDispatch)) {
|
||||
@@ -1331,7 +1341,7 @@ void Decoder::AddBranchTarget(uint64_t Target) {
|
||||
}
|
||||
}
|
||||
|
||||
const uint8_t* Decoder::AdjustAddrForSpecialRegion(const uint8_t* _InstStream, uint64_t EntryPoint, uint64_t RIP) {
|
||||
const Decoder::DecodeStream Decoder::AdjustAddrForSpecialRegion(const uint8_t* _InstStream, uint64_t EntryPoint, uint64_t RIP) {
|
||||
constexpr uint64_t VSyscall_Base = 0xFFFF'FFFF'FF60'0000ULL;
|
||||
constexpr uint64_t VSyscall_End = VSyscall_Base + 0x1000;
|
||||
|
||||
@@ -1342,10 +1352,16 @@ const uint8_t* Decoder::AdjustAddrForSpecialRegion(const uint8_t* _InstStream, u
|
||||
// Offset 0x400: vtime
|
||||
// Offset 0x800: vgetcpu
|
||||
uint64_t Offset = RIP - VSyscall_Base;
|
||||
return VSyscallData + Offset;
|
||||
return DecodeStream {
|
||||
.InstStream = _InstStream - EntryPoint + RIP,
|
||||
.AdjustedInstStream = VSyscallData + Offset,
|
||||
};
|
||||
}
|
||||
|
||||
return _InstStream - EntryPoint + RIP;
|
||||
return DecodeStream {
|
||||
.InstStream = _InstStream - EntryPoint + RIP,
|
||||
.AdjustedInstStream = _InstStream - EntryPoint + RIP,
|
||||
};
|
||||
}
|
||||
|
||||
bool Decoder::CheckIfCacheable(FEXCore::Core::InternalThreadState& Thread, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst) {
|
||||
@@ -1373,7 +1389,6 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
|
||||
|
||||
EntryPoint = PC;
|
||||
BlockInfo.EntryPoints = {PC};
|
||||
InstStream = _InstStream;
|
||||
|
||||
uint64_t TotalInstructions {};
|
||||
|
||||
@@ -1507,10 +1522,11 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
|
||||
EraseBlock = true;
|
||||
} else {
|
||||
LogMan::Msg::EFmt("{} instruction in entry block: {:X}",
|
||||
BlockIt->BlockStatus == DecodedBlockStatus::INVALID_INST ? "Invalid" :
|
||||
BlockIt->BlockStatus == DecodedBlockStatus::NOEXEC_INST ? "NoExec" :
|
||||
BlockIt->BlockStatus == DecodedBlockStatus::BAD_RELOCATION ? "BadRelocation" :
|
||||
"PartialDecode",
|
||||
BlockIt->BlockStatus == DecodedBlockStatus::INVALID_INST ? "Invalid" :
|
||||
BlockIt->BlockStatus == DecodedBlockStatus::NOEXEC_INST ? "NoExec" :
|
||||
BlockIt->BlockStatus == DecodedBlockStatus::BAD_RELOCATION ? "BadRelocation" :
|
||||
BlockIt->BlockStatus == DecodedBlockStatus::UNIMPLEMENTED_INST ? "Unimplemented" :
|
||||
"PartialDecode",
|
||||
OpAddress);
|
||||
}
|
||||
break;
|
||||
|
||||
@@ -32,6 +32,7 @@ public:
|
||||
NOEXEC_INST,
|
||||
PARTIAL_DECODE_INST,
|
||||
BAD_RELOCATION,
|
||||
UNIMPLEMENTED_INST,
|
||||
};
|
||||
|
||||
// New Frontend decoding
|
||||
@@ -55,6 +56,7 @@ public:
|
||||
|
||||
Decoder(FEXCore::Core::InternalThreadState* Thread);
|
||||
bool CheckIfCacheable(FEXCore::Core::InternalThreadState&, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst);
|
||||
|
||||
void DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thread, const uint8_t* InstStream, uint64_t PC, uint64_t MaxInst);
|
||||
|
||||
const DecodedBlockInformation* GetDecodedBlockInfo() const {
|
||||
@@ -91,7 +93,7 @@ private:
|
||||
|
||||
FEX_CONFIG_OPT(EnableCodeCacheValidation, ENABLECODECACHEVALIDATION);
|
||||
|
||||
bool DecodeInstructionImpl(uint64_t PC);
|
||||
DecodedBlockStatus DecodeInstructionImpl(uint64_t PC);
|
||||
DecodedBlockStatus DecodeInstruction(uint64_t PC);
|
||||
|
||||
void BranchTargetInMultiblockRange();
|
||||
@@ -110,8 +112,8 @@ private:
|
||||
InstructionSize += Size;
|
||||
}
|
||||
|
||||
bool NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op, DecodedHeader Options = {});
|
||||
bool NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op);
|
||||
DecodedBlockStatus NormalOp(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op, DecodedHeader Options = {});
|
||||
DecodedBlockStatus NormalOpHeader(const FEXCore::X86Tables::X86InstInfo* Info, uint16_t Op);
|
||||
|
||||
void DecodeREXIfValid(int8_t ExpectedOffset = -1);
|
||||
|
||||
@@ -126,7 +128,27 @@ private:
|
||||
bool HitNonExecutableRange {};
|
||||
bool HitBadRelocation {};
|
||||
|
||||
const uint8_t* InstStream {};
|
||||
struct DecodeStream {
|
||||
// Original instruction stream RIP location.
|
||||
const uint8_t* InstStream;
|
||||
|
||||
// Adjusted location for FEX actually decodes from.
|
||||
const uint8_t* AdjustedInstStream;
|
||||
|
||||
DecodeStream& operator-=(size_t offset) noexcept {
|
||||
InstStream -= offset;
|
||||
AdjustedInstStream -= offset;
|
||||
return *this;
|
||||
}
|
||||
|
||||
DecodeStream& operator+=(size_t offset) noexcept {
|
||||
InstStream += offset;
|
||||
AdjustedInstStream += offset;
|
||||
return *this;
|
||||
}
|
||||
};
|
||||
|
||||
DecodeStream InstStream;
|
||||
IR::OpSize GetGPROpSize() const {
|
||||
return BlockInfo.Is64BitMode ? IR::OpSize::i64Bit : IR::OpSize::i32Bit;
|
||||
}
|
||||
@@ -168,6 +190,6 @@ private:
|
||||
const std::array<X86Tables::X86InstInfo, X86Tables::MAX_VEX_TABLE_SIZE>* VEXTable {};
|
||||
const std::array<X86Tables::X86InstInfo, X86Tables::MAX_VEX_GROUP_TABLE_SIZE>* VEXTableGroup {};
|
||||
|
||||
const uint8_t* AdjustAddrForSpecialRegion(const uint8_t* _InstStream, uint64_t EntryPoint, uint64_t RIP);
|
||||
const DecodeStream AdjustAddrForSpecialRegion(const uint8_t* _InstStream, uint64_t EntryPoint, uint64_t RIP);
|
||||
};
|
||||
} // namespace FEXCore::Frontend
|
||||
@@ -302,6 +302,16 @@ struct OpHandlers<IR::OP_F80FYL2X> {
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FYL2XP1> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
ScopedSoftFloatState State {FCW, Frame, true};
|
||||
const X80SoftFloat One {&State.State, 1.0};
|
||||
return X80SoftFloat::FYL2X(&State.State, X80SoftFloat::FADD(&State.State, Src1, One), Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ATAN> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle(uint16_t FCW, VectorRegType Src1, VectorRegType Src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
@@ -417,6 +427,14 @@ struct OpHandlers<IR::OP_F64FYL2X> {
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FYL2XP1> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
|
||||
return src2 * log2(1.0 + src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SCALE> {
|
||||
FEXCORE_PRESERVE_ALL_ATTR static double handle(double src1, double src2, FEXCore::Core::CpuStateFrame* Frame) {
|
||||
|
||||
@@ -72,6 +72,8 @@ void InterpreterOps::FillFallbackIndexPointers(Core::FallbackABIInfo* Info, uint
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80DIV>::handle)};
|
||||
Info[Core::OPINDEX_F80FYL2X] = {ABIHandlers[FABI_F80_I16_F80_F80_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80FYL2X>::handle)};
|
||||
Info[Core::OPINDEX_F80FYL2XP1] = {ABIHandlers[FABI_F80_I16_F80_F80_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80FYL2XP1>::handle)};
|
||||
Info[Core::OPINDEX_F80ATAN] = {ABIHandlers[FABI_F80_I16_F80_F80_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F80ATAN>::handle)};
|
||||
Info[Core::OPINDEX_F80FPREM1] = {ABIHandlers[FABI_F80_I16_F80_F80_PTR],
|
||||
@@ -97,6 +99,8 @@ void InterpreterOps::FillFallbackIndexPointers(Core::FallbackABIInfo* Info, uint
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64FPREM1>::handle)};
|
||||
Info[Core::OPINDEX_F64FYL2X] = {ABIHandlers[FABI_F64_F64_F64_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64FYL2X>::handle)};
|
||||
Info[Core::OPINDEX_F64FYL2XP1] = {ABIHandlers[FABI_F64_F64_F64_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64FYL2XP1>::handle)};
|
||||
Info[Core::OPINDEX_F64SCALE] = {ABIHandlers[FABI_F64_F64_F64_PTR],
|
||||
reinterpret_cast<uint64_t>(&FEXCore::CPU::OpHandlers<IR::OP_F64SCALE>::handle)};
|
||||
|
||||
@@ -254,6 +258,7 @@ bool InterpreterOps::GetFallbackHandler(const IR::IROp_Header* IROp, FallbackInf
|
||||
COMMON_BINARY_X87_OP(MUL)
|
||||
COMMON_BINARY_X87_OP(DIV)
|
||||
COMMON_BINARY_X87_OP(FYL2X)
|
||||
COMMON_BINARY_X87_OP(FYL2XP1)
|
||||
COMMON_BINARY_X87_OP(ATAN)
|
||||
COMMON_BINARY_X87_OP(FPREM1)
|
||||
COMMON_BINARY_X87_OP(FPREM)
|
||||
@@ -268,6 +273,7 @@ bool InterpreterOps::GetFallbackHandler(const IR::IROp_Header* IROp, FallbackInf
|
||||
|
||||
// Double Precision Binary
|
||||
COMMON_BINARY_F64_OP(FYL2X)
|
||||
COMMON_BINARY_F64_OP(FYL2XP1)
|
||||
COMMON_BINARY_F64_OP(ATAN)
|
||||
COMMON_BINARY_F64_OP(FPREM1)
|
||||
COMMON_BINARY_F64_OP(FPREM)
|
||||
|
||||
@@ -523,7 +523,7 @@ DEF_OP(AndWithFlags) {
|
||||
}
|
||||
|
||||
DEF_OP(AndShift) {
|
||||
auto Op = IROp->C<IR::IROp_XorShift>();
|
||||
auto Op = IROp->C<IR::IROp_AndShift>();
|
||||
|
||||
and_(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1), GetReg(Op->Src2), ConvertIRShiftType(Op->Shift), Op->ShiftAmount);
|
||||
}
|
||||
@@ -721,7 +721,7 @@ DEF_OP(Extr) {
|
||||
}
|
||||
|
||||
DEF_OP(PDep) {
|
||||
auto Op = IROp->C<IR::IROp_PExt>();
|
||||
auto Op = IROp->C<IR::IROp_PDep>();
|
||||
const auto EmitSize = ConvertSize48(IROp);
|
||||
|
||||
const auto Dest = GetReg(Node);
|
||||
|
||||
@@ -55,6 +55,29 @@ DEF_OP(ExitFunction) {
|
||||
|
||||
uint64_t NewRIP;
|
||||
|
||||
if constexpr (Context::BLOCK_DEBUGGING) {
|
||||
// Skip block linking when BLOCK_DEBUGGING as it adds overhead and is unncessary.
|
||||
// This is a debug only feature and doesn't need caching help.
|
||||
bool IsInlineRIP = IsInlineConstant(Op->NewRIP, &NewRIP) || IsInlineEntrypointOffset(Op->NewRIP, &NewRIP);
|
||||
ARMEmitter::ForwardLabel l_ExitLink;
|
||||
if (IsInlineRIP) {
|
||||
ldr(TMP1, &l_ExitLink);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
|
||||
} else {
|
||||
auto RipReg = GetReg(Op->NewRIP);
|
||||
str(RipReg.X(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip));
|
||||
}
|
||||
ldr(TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.DispatcherLoopTop));
|
||||
br(TMP2);
|
||||
|
||||
if (IsInlineRIP) {
|
||||
BindOrRestart(&l_ExitLink);
|
||||
dc64(NewRIP);
|
||||
}
|
||||
|
||||
return;
|
||||
}
|
||||
|
||||
if (IsInlineConstant(Op->NewRIP, &NewRIP) || IsInlineEntrypointOffset(Op->NewRIP, &NewRIP)) {
|
||||
#ifdef ARCHITECTURE_arm64ec
|
||||
if (NewRIP < EC_CODE_BITMAP_MAX_ADDRESS && RtlIsEcCode(NewRIP)) {
|
||||
|
||||
@@ -776,8 +776,15 @@ void Arm64JITCore::EmitSuspendInterruptCheck() {
|
||||
if (CTX->Config.NeedsPendingInterruptFaultCheck) {
|
||||
// Trigger a fault if there are any pending interrupts
|
||||
// Used only for suspend on WIN32 at the moment
|
||||
strb(ARMEmitter::XReg::zr, STATE,
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState));
|
||||
constexpr size_t InterruptPageOffset =
|
||||
offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState);
|
||||
if constexpr (InterruptPageOffset <= 32760) {
|
||||
str(ARMEmitter::XReg::zr, STATE, InterruptPageOffset);
|
||||
} else {
|
||||
// Need to use vector 128-bit store for this range.
|
||||
// Doesn't matter which register we use to store.
|
||||
str(ARMEmitter::QReg::q0, STATE, InterruptPageOffset);
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef ARCHITECTURE_arm64ec
|
||||
|
||||
@@ -568,7 +568,7 @@ DEF_OP(LoadDF) {
|
||||
|
||||
DEF_OP(ContextClear) {
|
||||
auto Op = IROp->C<IR::IROp_ContextClear>();
|
||||
if (CTX->HostFeatures.SupportsCLZERO) {
|
||||
if (CTX->HostFeatures.PreferZVAForVZero) {
|
||||
// We can use CLZero directly when hardware supports it.
|
||||
// Provides a fairly generous speed-up on Ampere1A hardware.
|
||||
// TODO: When FEAT_MOPS hardware ships, test memset using MOPS.
|
||||
@@ -2400,11 +2400,12 @@ DEF_OP(CacheLineClear) {
|
||||
|
||||
// Clear dcache only
|
||||
// icache doesn't matter here since the guest application shouldn't be calling clflush on JIT code.
|
||||
// check host cacheline size again x86_64 size to ensure at least 64 bytes are cleaned
|
||||
if (CTX->HostFeatures.DCacheLineSize >= 64U) {
|
||||
dc(ARMEmitter::DataCacheOperation::CIVAC, MemReg);
|
||||
} else {
|
||||
auto CurrentWorkingReg = MemReg.X();
|
||||
for (size_t i = 0; i < std::max(1U, CTX->HostFeatures.DCacheLineSize / 64U); ++i) {
|
||||
for (size_t i = 0; i < std::max(1U, 64U / CTX->HostFeatures.DCacheLineSize); ++i) {
|
||||
dc(ARMEmitter::DataCacheOperation::CIVAC, TMP1);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, CurrentWorkingReg, CTX->HostFeatures.DCacheLineSize);
|
||||
CurrentWorkingReg = TMP1;
|
||||
@@ -2428,11 +2429,12 @@ DEF_OP(CacheLineClean) {
|
||||
auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
// Clean dcache only
|
||||
// check host cacheline size again x86_64 size to ensure at least 64 bytes are cleaned
|
||||
if (CTX->HostFeatures.DCacheLineSize >= 64U) {
|
||||
dc(ARMEmitter::DataCacheOperation::CVAC, MemReg);
|
||||
} else {
|
||||
auto CurrentWorkingReg = MemReg.X();
|
||||
for (size_t i = 0; i < std::max(1U, CTX->HostFeatures.DCacheLineSize / 64U); ++i) {
|
||||
for (size_t i = 0; i < std::max(1U, 64U / CTX->HostFeatures.DCacheLineSize); ++i) {
|
||||
dc(ARMEmitter::DataCacheOperation::CVAC, TMP1);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, CurrentWorkingReg, CTX->HostFeatures.DCacheLineSize);
|
||||
CurrentWorkingReg = TMP1;
|
||||
|
||||
@@ -73,8 +73,8 @@ DEF_OP(Break) {
|
||||
uint64_t Constant {};
|
||||
memcpy(&Constant, &State, sizeof(State));
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, Constant);
|
||||
str(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Constant);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
|
||||
|
||||
switch (Op->Reason.Signal) {
|
||||
case Core::FAULT_SIGILL:
|
||||
|
||||
@@ -1036,24 +1036,28 @@ DEF_OP(VMov) {
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Source = GetVReg(Op->Source);
|
||||
const auto Sub64BitHandler = [&](ARMEmitter::SubRegSize InsertSize) {
|
||||
if (Dst != Source) {
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), 0);
|
||||
ins(InsertSize, Dst, 0, Source, 0);
|
||||
} else {
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), 0);
|
||||
ins(InsertSize, VTMP1, 0, Source, 0);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
};
|
||||
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: {
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), 0);
|
||||
ins(ARMEmitter::SubRegSize::i8Bit, VTMP1, 0, Source, 0);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
Sub64BitHandler(ARMEmitter::SubRegSize::i8Bit);
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i16Bit: {
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), 0);
|
||||
ins(ARMEmitter::SubRegSize::i16Bit, VTMP1, 0, Source, 0);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
Sub64BitHandler(ARMEmitter::SubRegSize::i16Bit);
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i32Bit: {
|
||||
movi(ARMEmitter::SubRegSize::i64Bit, VTMP1.Q(), 0);
|
||||
ins(ARMEmitter::SubRegSize::i32Bit, VTMP1, 0, Source, 0);
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
Sub64BitHandler(ARMEmitter::SubRegSize::i32Bit);
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
@@ -1095,16 +1099,21 @@ DEF_OP(VAddP) {
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Pred = PRED_TMP_32B.Merging();
|
||||
|
||||
// SVE ADDP is a destructive operation, so we need a temporary
|
||||
movprfx(VTMP1.Z(), VectorLower.Z());
|
||||
// SVE ADDP is a destructive operation, so we need a temporary if
|
||||
// the destination and the lower vector don't alias.
|
||||
auto LHS = Dst;
|
||||
if (Dst != VectorLower) {
|
||||
movprfx(VTMP1.Z(), VectorLower.Z());
|
||||
LHS = VTMP1;
|
||||
}
|
||||
|
||||
// Unlike Adv. SIMD's version of ADDP, which acts like it concats the
|
||||
// upper vector onto the end of the lower vector and then performs
|
||||
// pairwise addition, the SVE version actually interleaves the
|
||||
// results of the pairwise addition (gross!), so we need to undo that.
|
||||
addp(SubRegSize, VTMP1.Z(), Pred, VTMP1.Z(), VectorUpper.Z());
|
||||
uzp1(SubRegSize, Dst.Z(), VTMP1.Z(), VTMP1.Z());
|
||||
uzp2(SubRegSize, VTMP2.Z(), VTMP1.Z(), VTMP1.Z());
|
||||
addp(SubRegSize, LHS.Z(), Pred, LHS.Z(), VectorUpper.Z());
|
||||
uzp1(SubRegSize, Dst.Z(), LHS.Z(), LHS.Z());
|
||||
uzp2(SubRegSize, VTMP2.Z(), LHS.Z(), LHS.Z());
|
||||
|
||||
// Merge upper half with lower half.
|
||||
splice<ARMEmitter::OpType::Destructive>(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), PRED_TMP_16B, Dst.Z(), VTMP2.Z());
|
||||
@@ -1298,16 +1307,21 @@ DEF_OP(VFAddP) {
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
const auto Pred = PRED_TMP_32B.Merging();
|
||||
|
||||
// SVE FADDP is a destructive operation, so we need a temporary
|
||||
movprfx(VTMP1.Z(), VectorLower.Z());
|
||||
// SVE FADDP is a destructive operation, so we need a temporary if
|
||||
// the destination and the lower vector don't alias.
|
||||
auto LHS = Dst;
|
||||
if (Dst != VectorLower) {
|
||||
movprfx(VTMP1.Z(), VectorLower.Z());
|
||||
LHS = VTMP1;
|
||||
}
|
||||
|
||||
// Unlike Adv. SIMD's version of FADDP, which acts like it concats the
|
||||
// upper vector onto the end of the lower vector and then performs
|
||||
// pairwise addition, the SVE version actually interleaves the
|
||||
// results of the pairwise addition (gross!), so we need to undo that.
|
||||
faddp(SubRegSize, VTMP1.Z(), Pred, VTMP1.Z(), VectorUpper.Z());
|
||||
uzp1(SubRegSize, Dst.Z(), VTMP1.Z(), VTMP1.Z());
|
||||
uzp2(SubRegSize, VTMP2.Z(), VTMP1.Z(), VTMP1.Z());
|
||||
faddp(SubRegSize, LHS.Z(), Pred, LHS.Z(), VectorUpper.Z());
|
||||
uzp1(SubRegSize, Dst.Z(), LHS.Z(), LHS.Z());
|
||||
uzp2(SubRegSize, VTMP2.Z(), LHS.Z(), LHS.Z());
|
||||
|
||||
// Merge upper half with lower half.
|
||||
splice<ARMEmitter::OpType::Destructive>(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), PRED_TMP_16B, Dst.Z(), VTMP2.Z());
|
||||
@@ -1526,9 +1540,14 @@ DEF_OP(VFRecp) {
|
||||
return;
|
||||
}
|
||||
|
||||
fmov(SubRegSize.Vector, VTMP1.Z(), 1.0);
|
||||
fdiv(SubRegSize.Vector, VTMP1.Z(), Pred, VTMP1.Z(), Vector.Z());
|
||||
mov(Dst.Z(), VTMP1.Z());
|
||||
if (Dst != Vector) {
|
||||
fmov(SubRegSize.Vector, Dst.Z(), 1.0);
|
||||
fdiv(SubRegSize.Vector, Dst.Z(), Pred, Dst.Z(), Vector.Z());
|
||||
} else {
|
||||
fmov(SubRegSize.Vector, VTMP1.Z(), 1.0);
|
||||
fdiv(SubRegSize.Vector, VTMP1.Z(), Pred, VTMP1.Z(), Vector.Z());
|
||||
mov(Dst.Z(), VTMP1.Z());
|
||||
}
|
||||
} else {
|
||||
if (IsScalar) {
|
||||
if (ElementSize == IR::OpSize::i32Bit && HostSupportsRPRES) {
|
||||
@@ -1780,10 +1799,14 @@ DEF_OP(VUMin) {
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
cmhi(SubRegSize, VTMP1.Q(), Vector2.Q(), Vector1.Q());
|
||||
mov(VTMP2.Q(), Vector1.Q());
|
||||
bif(VTMP2.Q(), Vector2.Q(), VTMP1.Q());
|
||||
mov(Dst.Q(), VTMP2.Q());
|
||||
if (Dst != Vector1 && Dst != Vector2) {
|
||||
cmhi(SubRegSize, Dst.Q(), Vector1.Q(), Vector2.Q());
|
||||
bsl(Dst.Q(), Vector2.Q(), Vector1.Q());
|
||||
} else {
|
||||
cmhi(SubRegSize, VTMP1.Q(), Vector1.Q(), Vector2.Q());
|
||||
bsl(VTMP1.Q(), Vector2.Q(), Vector1.Q());
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: break;
|
||||
@@ -1829,10 +1852,14 @@ DEF_OP(VSMin) {
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
cmgt(SubRegSize, VTMP1.Q(), Vector1.Q(), Vector2.Q());
|
||||
mov(VTMP2.Q(), Vector1.Q());
|
||||
bif(VTMP2.Q(), Vector2.Q(), VTMP1.Q());
|
||||
mov(Dst.Q(), VTMP2.Q());
|
||||
if (Dst != Vector1 && Dst != Vector2) {
|
||||
cmgt(SubRegSize, Dst.Q(), Vector1.Q(), Vector2.Q());
|
||||
bsl(Dst.Q(), Vector2.Q(), Vector1.Q());
|
||||
} else {
|
||||
cmgt(SubRegSize, VTMP1.Q(), Vector1.Q(), Vector2.Q());
|
||||
bsl(VTMP1.Q(), Vector2.Q(), Vector1.Q());
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: break;
|
||||
@@ -1878,10 +1905,14 @@ DEF_OP(VUMax) {
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
cmhi(SubRegSize, VTMP1.Q(), Vector2.Q(), Vector1.Q());
|
||||
mov(VTMP2.Q(), Vector1.Q());
|
||||
bif(VTMP2.Q(), Vector2.Q(), VTMP1.Q());
|
||||
mov(Dst.Q(), VTMP2.Q());
|
||||
if (Dst != Vector1 && Dst != Vector2) {
|
||||
cmhi(SubRegSize, Dst.Q(), Vector1.Q(), Vector2.Q());
|
||||
bsl(Dst.Q(), Vector1.Q(), Vector2.Q());
|
||||
} else {
|
||||
cmhi(SubRegSize, VTMP1.Q(), Vector1.Q(), Vector2.Q());
|
||||
bsl(VTMP1.Q(), Vector1.Q(), Vector2.Q());
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: break;
|
||||
@@ -1927,10 +1958,14 @@ DEF_OP(VSMax) {
|
||||
break;
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
cmgt(SubRegSize, VTMP1.Q(), Vector2.Q(), Vector1.Q());
|
||||
mov(VTMP2.Q(), Vector1.Q());
|
||||
bif(VTMP2.Q(), Vector2.Q(), VTMP1.Q());
|
||||
mov(Dst.Q(), VTMP2.Q());
|
||||
if (Dst != Vector1 && Dst != Vector2) {
|
||||
cmgt(SubRegSize, Dst.Q(), Vector1.Q(), Vector2.Q());
|
||||
bsl(Dst.Q(), Vector1.Q(), Vector2.Q());
|
||||
} else {
|
||||
cmgt(SubRegSize, VTMP1.Q(), Vector1.Q(), Vector2.Q());
|
||||
bsl(VTMP1.Q(), Vector1.Q(), Vector2.Q());
|
||||
mov(Dst.Q(), VTMP1.Q());
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: break;
|
||||
@@ -2980,9 +3015,14 @@ DEF_OP(VInsElement) {
|
||||
auto Reg = GetVReg(Op->DestVector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// Broadcast our source value across a temporary,
|
||||
// then combine with the destination.
|
||||
dup(SubRegSize, VTMP2.Z(), SrcVector.Z(), SrcIdx);
|
||||
// Broadcast our source value across a temporary, then combine
|
||||
// with the destination.
|
||||
//
|
||||
// We don't need to perform the dup if we're just merging a 128-bit vector into
|
||||
// into an equivalent position since we have a predicate set up already.
|
||||
if (!(ElementSize == IR::OpSize::i128Bit && SrcIdx == DestIdx)) {
|
||||
dup(SubRegSize, VTMP2.Z(), SrcVector.Z(), SrcIdx);
|
||||
}
|
||||
|
||||
// We don't need to move the data unnecessarily if
|
||||
// DestVector just so happens to also be the IR op
|
||||
@@ -2995,10 +3035,12 @@ DEF_OP(VInsElement) {
|
||||
|
||||
if (ElementSize == IR::OpSize::i128Bit) {
|
||||
if (DestIdx == 0) {
|
||||
mov(ARMEmitter::SubRegSize::i8Bit, Dst.Z(), PRED_TMP_16B.Merging(), VTMP2.Z());
|
||||
const auto Source = SrcIdx == 0 ? SrcVector : VTMP2;
|
||||
mov(ARMEmitter::SubRegSize::i8Bit, Dst.Z(), PRED_TMP_16B.Merging(), Source.Z());
|
||||
} else {
|
||||
const auto Source = SrcIdx == 1 ? SrcVector : VTMP2;
|
||||
not_(Predicate, PRED_TMP_32B.Zeroing(), PRED_TMP_16B);
|
||||
mov(ARMEmitter::SubRegSize::i8Bit, Dst.Z(), Predicate.Merging(), VTMP2.Z());
|
||||
mov(ARMEmitter::SubRegSize::i8Bit, Dst.Z(), Predicate.Merging(), Source.Z());
|
||||
}
|
||||
} else {
|
||||
const auto UpperBound = 16 >> FEXCore::ilog2(IR::OpSizeToSize(ElementSize));
|
||||
@@ -4435,7 +4477,7 @@ DEF_OP(VFNMLA) {
|
||||
// - SVE - FMLS
|
||||
// - ASIMD - FMLS
|
||||
// - Scalar - FMSUB
|
||||
const auto Op = IROp->C<IR::IROp_VFMLA>();
|
||||
const auto Op = IROp->C<IR::IROp_VFNMLA>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
@@ -4503,7 +4545,7 @@ DEF_OP(VFNMLS) {
|
||||
// - ASIMD - FMLS (With Negated addend)
|
||||
// - Scalar - FNMADD
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VFMLS>();
|
||||
const auto Op = IROp->C<IR::IROp_VFNMLS>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
@@ -4612,6 +4654,36 @@ DEF_OP(VFCopySign) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(F64FPREM) {
|
||||
const auto Op = IROp->C<IR::IROp_F64FPREM>();
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
|
||||
fmov(VTMP1.D(), Src1.D());
|
||||
fmov(VTMP2.D(), Src2.D());
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.F64FPREMHandler));
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(TMP1);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
fmov(Dst.D(), VTMP1.D());
|
||||
}
|
||||
|
||||
DEF_OP(F64FPREM1) {
|
||||
const auto Op = IROp->C<IR::IROp_F64FPREM1>();
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
|
||||
fmov(VTMP1.D(), Src1.D());
|
||||
fmov(VTMP2.D(), Src2.D());
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.F64FPREM1Handler));
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(TMP1);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
fmov(Dst.D(), VTMP1.D());
|
||||
}
|
||||
|
||||
DEF_OP(F64SIN) {
|
||||
const auto Op = IROp->C<IR::IROp_F64SIN>();
|
||||
const auto Src = GetVReg(Op->Src);
|
||||
@@ -4683,6 +4755,22 @@ DEF_OP(F64FYL2X) {
|
||||
fmov(Dst.D(), VTMP1.D());
|
||||
}
|
||||
|
||||
// Src=x(ST0), Src2=y(ST1). Marshal into VTMP1/VTMP2 and dispatch the shared handler.
|
||||
DEF_OP(F64FYL2XP1) {
|
||||
const auto Op = IROp->C<IR::IROp_F64FYL2XP1>();
|
||||
const auto Src = GetVReg(Op->Src);
|
||||
const auto Src2 = GetVReg(Op->Src2);
|
||||
const auto Dst = GetVReg(Node);
|
||||
|
||||
fmov(VTMP1.D(), Src.D());
|
||||
fmov(VTMP2.D(), Src2.D());
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.F64FYL2XP1Handler));
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(TMP1);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
fmov(Dst.D(), VTMP1.D());
|
||||
}
|
||||
|
||||
DEF_OP(F64SCALE) {
|
||||
const auto Op = IROp->C<IR::IROp_F64SCALE>();
|
||||
const auto Src1 = GetVReg(Op->Src1);
|
||||
|
||||
@@ -87,8 +87,11 @@ void LookupCache::ClearL2Cache(const FEXCore::LookupCacheBaseLockToken& lk) {
|
||||
}
|
||||
|
||||
void LookupCache::ClearThreadLocalCaches(const LookupCacheWriteLockToken&) {
|
||||
// TODO: Preserve code cache entries?
|
||||
// Clear L1 and L2 by clearing the full cache.
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), TotalCacheSize, false);
|
||||
|
||||
// TODO: Rename this member to avoid confusion with code caching
|
||||
CachedCodePages.clear();
|
||||
}
|
||||
|
||||
|
||||
@@ -502,7 +502,7 @@ void OpDispatchBuilder::LEAVEOp(OpcodeArgs) {
|
||||
auto NewGPR = Pop(OperandSize, SP);
|
||||
|
||||
// Store the new stack pointer
|
||||
StoreGPRRegister(X86State::REG_RSP, SP, OperandSize);
|
||||
StoreGPRRegister(X86State::REG_RSP, SP, GPRSize);
|
||||
|
||||
// Store what we loaded to RBP
|
||||
StoreGPRRegister(X86State::REG_RBP, NewGPR, OperandSize);
|
||||
@@ -2427,9 +2427,10 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
auto BitSelect = (Size == (LshrSize * 8)) ? Src : Src.And(Mask);
|
||||
auto LshrOpSize = IR::SizeToOpSize(LshrSize);
|
||||
|
||||
// OF/SF/AF/PF undefined. ZF must be preserved. We choose to preserve OF/SF
|
||||
// too since we just use an rmif to insert into CF directly. We could
|
||||
// optimize perhaps.
|
||||
// AMD: OF/SF/ZF/AF/PF undefined.
|
||||
// Intel: OF/SF/AF/PF undefined. ZF must be preserved.
|
||||
// We choose to preserve ZF/OF/SF since we just use an rmif
|
||||
// to insert into CF directly. We could optimize perhaps.
|
||||
//
|
||||
// Set CF before the action to save a move, except for complements where we
|
||||
// can reuse the invert.
|
||||
@@ -2547,7 +2548,10 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) {
|
||||
Value = _Lshr(std::max(OpSize::i32Bit, GetOpSize(Value)), Value, BitSelect.Ref());
|
||||
}
|
||||
|
||||
// OF/SF/ZF/AF/PF undefined.
|
||||
// AMD: OF/SF/ZF/AF/PF undefined.
|
||||
// Intel: OF/SF/AF/PF undefined. ZF must be preserved.
|
||||
// We choose to preserve ZF/OF/SF since we just use an rmif
|
||||
// to insert into CF directly. We could optimize perhaps.
|
||||
SetCFDirect(Value, 0, true);
|
||||
}
|
||||
}
|
||||
@@ -3449,30 +3453,38 @@ void OpDispatchBuilder::CMPSOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::LODSOp(OpcodeArgs) {
|
||||
if (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_ADDRESS_SIZE) {
|
||||
LogMan::Msg::EFmt("LODSOp: Can't handle address size override (OP: 0x{:04X}, Flags: 0x{:08X})", Op->OP, Op->Flags);
|
||||
if (!Is64BitMode && (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_ADDRESS_SIZE)) {
|
||||
LogMan::Msg::EFmt("LODSOp: Address size override (0x67) not supported in 32-bit mode (OP: 0x{:04X}).", Op->OP);
|
||||
DecodeFailure = true;
|
||||
return;
|
||||
}
|
||||
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
OpSize AddrSize = GetStringOpSize(Op);
|
||||
|
||||
const bool Repeat = (Op->Flags & (FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX | FEXCore::X86Tables::DecodeFlags::FLAG_REPNE_PREFIX)) != 0;
|
||||
|
||||
if (!Repeat) {
|
||||
Ref Dest_RSI = MakeSegmentAddress(X86State::REG_RSI, Op->Flags, X86Tables::DecodeFlags::FLAG_DS_PREFIX);
|
||||
Ref Src_RSI = LoadGPRRegister(X86State::REG_RSI, AddrSize);
|
||||
Ref Dest_RSI = AppendSegmentOffset(Src_RSI, 0, X86Tables::DecodeFlags::FLAG_DS_PREFIX, true);
|
||||
|
||||
auto Src = _LoadMemGPRAutoTSO(Size, Dest_RSI, Size);
|
||||
|
||||
StoreResultGPR(Op, Src);
|
||||
|
||||
// Offset the pointer
|
||||
Ref TailDest_RSI = LoadGPRRegister(X86State::REG_RSI);
|
||||
StoreGPRRegister(X86State::REG_RSI, OffsetByDir(TailDest_RSI, IR::OpSizeToSize(Size)));
|
||||
Ref TailDest_RSI = OffsetByDir(Src_RSI, IR::OpSizeToSize(Size));
|
||||
if (Is64BitMode && AddrSize == OpSize::i32Bit) {
|
||||
TailDest_RSI = _Bfe(OpSize::i64Bit, 32, 0, TailDest_RSI);
|
||||
StoreGPRRegister(X86State::REG_RSI, TailDest_RSI);
|
||||
} else {
|
||||
StoreGPRRegister(X86State::REG_RSI, TailDest_RSI, AddrSize);
|
||||
}
|
||||
} else {
|
||||
// Calculate flags early. because end of block
|
||||
CalculateDeferredFlags();
|
||||
|
||||
ForeachDirection([this, Op, Size](int32_t PtrDir) {
|
||||
ForeachDirection([this, Op, Size, AddrSize](int32_t PtrDir) {
|
||||
// XXX: Theoretically LODS could be optimized to
|
||||
// RSI += {-}(RCX * Size)
|
||||
// RAX = [RSI - Size]
|
||||
@@ -3500,7 +3512,8 @@ void OpDispatchBuilder::LODSOp(OpcodeArgs) {
|
||||
|
||||
// Working loop
|
||||
{
|
||||
Ref Dest_RSI = MakeSegmentAddress(X86State::REG_RSI, Op->Flags, X86Tables::DecodeFlags::FLAG_DS_PREFIX);
|
||||
Ref Src_RSI = LoadGPRRegister(X86State::REG_RSI, AddrSize);
|
||||
Ref Dest_RSI = AppendSegmentOffset(Src_RSI, 0, X86Tables::DecodeFlags::FLAG_DS_PREFIX, true);
|
||||
|
||||
auto Src = _LoadMemGPRAutoTSO(Size, Dest_RSI, Size);
|
||||
|
||||
@@ -3516,8 +3529,13 @@ void OpDispatchBuilder::LODSOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RCX, TailCounter);
|
||||
|
||||
// Offset the pointer
|
||||
TailDest_RSI = Add(OpSize::i64Bit, TailDest_RSI, PtrDir * static_cast<int32_t>(IR::OpSizeToSize(Size)));
|
||||
StoreGPRRegister(X86State::REG_RSI, TailDest_RSI);
|
||||
TailDest_RSI = Add(AddrSize, TailDest_RSI, PtrDir * static_cast<int32_t>(IR::OpSizeToSize(Size)));
|
||||
if (Is64BitMode && AddrSize == OpSize::i32Bit) {
|
||||
TailDest_RSI = _Bfe(OpSize::i64Bit, 32, 0, TailDest_RSI);
|
||||
StoreGPRRegister(X86State::REG_RSI, TailDest_RSI);
|
||||
} else {
|
||||
StoreGPRRegister(X86State::REG_RSI, TailDest_RSI, AddrSize);
|
||||
}
|
||||
|
||||
// Jump back to the start, we have more work to do
|
||||
Jump(LoopStart);
|
||||
@@ -4216,6 +4234,94 @@ void OpDispatchBuilder::UpdatePrefixFromSegment(Ref Segment, uint32_t SegmentReg
|
||||
}
|
||||
}
|
||||
|
||||
uint64_t OpDispatchBuilder::CalcAddress(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, bool IsLoad) {
|
||||
if constexpr (!Context::BLOCK_DEBUGGING) {
|
||||
LOGMAN_MSG_A_FMT("Tried to calculate address without block debugging enabled!");
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
const auto GPRMask = GPRSize == OpSize::i64Bit ? ~0ULL : ~0U;
|
||||
|
||||
// This makes the assumption that InternalThreadState is synchronized at the point of call!
|
||||
uint64_t Ptr {};
|
||||
if (Operand.IsLiteral()) {
|
||||
Ptr = Operand.Literal();
|
||||
|
||||
if (Operand.Data.Literal.Size != 8 && IsLoad) {
|
||||
// zero extend
|
||||
uint64_t width = Operand.Data.Literal.Size * 8;
|
||||
Ptr &= ((1ULL << width) - 1);
|
||||
}
|
||||
} else if (Operand.IsGPR()) {
|
||||
// Not a memory source.
|
||||
return ~0ULL;
|
||||
} else if (Operand.IsGPRDirect()) {
|
||||
Ptr = Thread->CurrentFrame->State.gregs[Operand.Data.GPR.GPR] & GPRMask;
|
||||
} else if (Operand.IsGPRIndirect() || Operand.IsGPRIndirectRelocation()) {
|
||||
Ptr = Thread->CurrentFrame->State.gregs[Operand.Data.GPR.GPR] & GPRMask;
|
||||
Ptr += static_cast<int32_t>(Operand.Data.GPRIndirect.Displacement);
|
||||
} else if (Operand.IsRIPRelative() || Operand.IsRIPRelativeRelocation()) {
|
||||
// 64-bit is RIP relative, while 32-bit is absolute.
|
||||
if (Is64BitMode) {
|
||||
Ptr = Op->PC + Op->InstSize + static_cast<int32_t>(Operand.Data.RIPLiteral.Value) - Entry;
|
||||
} else {
|
||||
Ptr = Operand.Data.RIPLiteral.Value;
|
||||
}
|
||||
} else if (Operand.IsSIB() || Operand.IsSIBRelocation()) {
|
||||
const bool IsVSIB = IsLoad && ((Op->Flags & X86Tables::DecodeFlags::FLAG_VSIB_BYTE) != 0);
|
||||
if (IsVSIB) {
|
||||
// TODO: Unhandled.
|
||||
return ~0ULL;
|
||||
}
|
||||
if (Operand.Data.SIB.Base != FEXCore::X86State::REG_INVALID) {
|
||||
Ptr = Thread->CurrentFrame->State.gregs[Operand.Data.SIB.Base] & GPRMask;
|
||||
}
|
||||
|
||||
if (Operand.Data.SIB.Index != FEXCore::X86State::REG_INVALID) {
|
||||
Ptr += (Thread->CurrentFrame->State.gregs[Operand.Data.SIB.Index] * Operand.Data.SIB.Scale) & GPRMask;
|
||||
}
|
||||
|
||||
Ptr += static_cast<int32_t>(Operand.Data.SIB.Offset);
|
||||
}
|
||||
|
||||
auto AppendSegment = [&](uint64_t Ptr, uint32_t Flags, uint32_t DefaultPrefix = FEXCore::X86Tables::DecodeFlags::FLAG_NO_PREFIX,
|
||||
bool Override = false) -> uint64_t {
|
||||
uint32_t Prefix = Flags & FEXCore::X86Tables::DecodeFlags::FLAG_SEGMENTS;
|
||||
|
||||
if (Is64BitMode) {
|
||||
if (Prefix == FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX) {
|
||||
return Ptr + Thread->CurrentFrame->State.fs_cached;
|
||||
} else if (Prefix == FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX) {
|
||||
return Ptr + Thread->CurrentFrame->State.gs_cached;
|
||||
}
|
||||
// If there was any other segment in 64bit then it is ignored
|
||||
} else {
|
||||
if (Prefix == FEXCore::X86Tables::DecodeFlags::FLAG_NO_PREFIX || Override) {
|
||||
// If there was no prefix then use the default one if available
|
||||
// Or the argument only uses a specific prefix (with override set)
|
||||
Prefix = DefaultPrefix;
|
||||
}
|
||||
// With the segment register optimization we store the GDT bases directly in the segment register to remove indexed loads
|
||||
switch (Prefix) {
|
||||
[[likely]] case FEXCore::X86Tables::DecodeFlags::FLAG_NO_PREFIX:
|
||||
return Ptr;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX: return Ptr + Thread->CurrentFrame->State.es_cached;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX: return Ptr + Thread->CurrentFrame->State.cs_cached;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX: return Ptr + Thread->CurrentFrame->State.ss_cached;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX: return Ptr + Thread->CurrentFrame->State.ds_cached;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX: return Ptr + Thread->CurrentFrame->State.fs_cached;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX: return Ptr + Thread->CurrentFrame->State.gs_cached;
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
|
||||
return Ptr;
|
||||
};
|
||||
|
||||
return AppendSegment(Ptr, Op->Flags);
|
||||
};
|
||||
|
||||
AddressMode OpDispatchBuilder::DecodeAddress(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand,
|
||||
MemoryAccessType AccessType, bool IsLoad) {
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
@@ -4306,6 +4412,7 @@ Ref OpDispatchBuilder::LoadSource_WithOpSize(RegClass Class, const X86Tables::De
|
||||
auto [Align, LoadData, ForceLoad, AccessType, AllowUpperGarbage] = Options;
|
||||
AddressMode A = DecodeAddress(Op, Operand, AccessType, true /* IsLoad */);
|
||||
|
||||
Ref Result {};
|
||||
if (Operand.IsGPR()) {
|
||||
const auto gpr = Operand.Data.GPR.GPR;
|
||||
const auto highIndex = Operand.Data.GPR.HighBits ? 1 : 0;
|
||||
@@ -4335,22 +4442,35 @@ Ref OpDispatchBuilder::LoadSource_WithOpSize(RegClass Class, const X86Tables::De
|
||||
}
|
||||
}
|
||||
|
||||
if ((IsOperandMem(Operand, true) && LoadData) || ForceLoad) {
|
||||
const bool ShouldLoad = (IsOperandMem(Operand, true) && LoadData) || ForceLoad;
|
||||
if (ShouldLoad) {
|
||||
if (OpSize == OpSize::f80Bit) {
|
||||
Ref MemSrc = LoadEffectiveAddress(this, A, GetGPROpSize(), true);
|
||||
if (CTX->HostFeatures.SupportsSVE128 || CTX->HostFeatures.SupportsSVE256) {
|
||||
return _LoadMemX87SVEOptPredicate(OpSize::i128Bit, OpSize::i16Bit, MemSrc);
|
||||
if (CTX->HostFeatures.SupportsSVE()) {
|
||||
Result = _LoadMemX87SVEOptPredicate(OpSize::i128Bit, OpSize::i16Bit, MemSrc);
|
||||
} else {
|
||||
// For X87 extended doubles, Split the load.
|
||||
auto Res = _LoadMem(Class, OpSize::i64Bit, MemSrc, Align == OpSize::iInvalid ? OpSize : Align);
|
||||
return _VLoadVectorElement(OpSize::i128Bit, OpSize::i16Bit, Res, 4, Add(OpSize::i64Bit, MemSrc, 8));
|
||||
Result = _VLoadVectorElement(OpSize::i128Bit, OpSize::i16Bit, Res, 4, Add(OpSize::i64Bit, MemSrc, 8));
|
||||
}
|
||||
} else {
|
||||
Result = _LoadMemAutoTSO(Class, OpSize, A, Align == OpSize::iInvalid ? OpSize : Align);
|
||||
}
|
||||
} else {
|
||||
Result = LoadEffectiveAddress(this, A, GetGPROpSize(), false, AllowUpperGarbage);
|
||||
}
|
||||
|
||||
if constexpr (Context::BLOCK_DEBUGGING) {
|
||||
if (ShouldLoad && CTX->BlockDebuggerTracker.IsSingleStepTarget(Entry)) {
|
||||
uint64_t Ptr = CalcAddress(Op, Operand, true);
|
||||
if (CTX->BlockDebuggerTracker.ContainsReadWatchPoint(Ptr, OpSizeToSize(OpSize))) {
|
||||
// It's up to the developer if they want more advanced debugging logic here.
|
||||
LogMan::Msg::IFmt("Entrypoint 0x{:x} will hit read watch: [0x{:x}, 0x{:x})", Entry, Ptr, Ptr + OpSizeToSize(OpSize));
|
||||
}
|
||||
}
|
||||
|
||||
return _LoadMemAutoTSO(Class, OpSize, A, Align == OpSize::iInvalid ? OpSize : Align);
|
||||
} else {
|
||||
return LoadEffectiveAddress(this, A, GetGPROpSize(), false, AllowUpperGarbage);
|
||||
}
|
||||
|
||||
return Result;
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::LoadGPRRegister(uint32_t GPR, IR::OpSize Size, uint8_t Offset, bool AllowUpperGarbage) {
|
||||
@@ -4465,7 +4585,7 @@ void OpDispatchBuilder::StoreResult_WithOpSize(RegClass Class, FEXCore::X86Table
|
||||
|
||||
if (OpSize == OpSize::f80Bit) {
|
||||
Ref MemStoreDst = LoadEffectiveAddress(this, A, GetGPROpSize(), true);
|
||||
if (CTX->HostFeatures.SupportsSVE128 || CTX->HostFeatures.SupportsSVE256) {
|
||||
if (CTX->HostFeatures.SupportsSVE()) {
|
||||
_StoreMemX87SVEOptPredicate(OpSize::i128Bit, OpSize::i16Bit, Src, MemStoreDst);
|
||||
} else {
|
||||
// For X87 extended doubles, split before storing
|
||||
@@ -4476,6 +4596,16 @@ void OpDispatchBuilder::StoreResult_WithOpSize(RegClass Class, FEXCore::X86Table
|
||||
} else {
|
||||
_StoreMemAutoTSO(Class, OpSize, A, Src, Align == OpSize::iInvalid ? OpSize : Align);
|
||||
}
|
||||
|
||||
if constexpr (Context::BLOCK_DEBUGGING) {
|
||||
if (CTX->BlockDebuggerTracker.IsSingleStepTarget(Entry)) {
|
||||
uint64_t Ptr = CalcAddress(Op, Operand, false);
|
||||
if (CTX->BlockDebuggerTracker.ContainsWriteWatchPoint(Ptr, OpSizeToSize(OpSize))) {
|
||||
// It's up to the developer if they want more advanced debugging logic here.
|
||||
LogMan::Msg::IFmt("Entrypoint 0x{:x} will hit write watch: [0x{:x}, 0x{:x})", Entry, Ptr, Ptr + OpSizeToSize(OpSize));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::StoreResult(RegClass Class, X86Tables::DecodedOp Op, const X86Tables::DecodedOperand& Operand, Ref Src,
|
||||
@@ -4487,9 +4617,10 @@ void OpDispatchBuilder::StoreResult(RegClass Class, X86Tables::DecodedOp Op, Ref
|
||||
StoreResult(Class, Op, Op->Dest, Src, Align, AccessType);
|
||||
}
|
||||
|
||||
OpDispatchBuilder::OpDispatchBuilder(FEXCore::Context::ContextImpl* ctx)
|
||||
OpDispatchBuilder::OpDispatchBuilder(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread)
|
||||
: IREmitter {ctx->OpDispatcherAllocator, ctx->HostFeatures.SupportsTSOImm9}
|
||||
, CTX {ctx} {
|
||||
, CTX {ctx}
|
||||
, Thread {Thread} {
|
||||
if (CTX->HostFeatures.SupportsAVX && CTX->HostFeatures.SupportsSVE256) {
|
||||
SaveAVXStateFunc = &OpDispatchBuilder::SaveAVXState;
|
||||
RestoreAVXStateFunc = &OpDispatchBuilder::RestoreAVXState;
|
||||
@@ -4647,35 +4778,36 @@ void OpDispatchBuilder::INTOp(OpcodeArgs) {
|
||||
case 0xCD: { // INT imm8
|
||||
uint8_t Literal = Op->Src[0].Literal();
|
||||
|
||||
#ifndef _WIN32
|
||||
constexpr uint8_t SYSCALL_LITERAL = 0x80;
|
||||
if (Literal == SYSCALL_LITERAL) {
|
||||
if (Is64BitMode) [[unlikely]] {
|
||||
LogMan::Msg::EFmt("[Unsupported] Trying to execute 32-bit syscall from a 64-bit process.");
|
||||
UnhandledOp(Op);
|
||||
if (CTX->HostFeatures.HostType == FEXCore::HostFeatures::HostTypeEnum::Linux) {
|
||||
constexpr uint8_t SYSCALL_LITERAL = 0x80;
|
||||
if (Literal == SYSCALL_LITERAL) {
|
||||
if (Is64BitMode) [[unlikely]] {
|
||||
LogMan::Msg::EFmt("[Unsupported] Trying to execute 32-bit syscall from a 64-bit process.");
|
||||
UnhandledOp(Op);
|
||||
return;
|
||||
}
|
||||
// Syscall on linux
|
||||
SyscallOp(Op, false);
|
||||
return;
|
||||
}
|
||||
} else if (CTX->HostFeatures.HostType == FEXCore::HostFeatures::HostTypeEnum::Wow64 ||
|
||||
CTX->HostFeatures.HostType == FEXCore::HostFeatures::HostTypeEnum::Arm64ec) {
|
||||
constexpr uint8_t SYSCALL_LITERAL = 0x2E;
|
||||
if (Literal == SYSCALL_LITERAL) {
|
||||
// Can be used for both 64-bit and 32-bit syscalls on windows
|
||||
SyscallOp(Op, false);
|
||||
return;
|
||||
}
|
||||
// Syscall on linux
|
||||
SyscallOp(Op, false);
|
||||
return;
|
||||
}
|
||||
#else
|
||||
constexpr uint8_t SYSCALL_LITERAL = 0x2E;
|
||||
if (Literal == SYSCALL_LITERAL) {
|
||||
// Can be used for both 64-bit and 32-bit syscalls on windows
|
||||
SyscallOp(Op, false);
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifdef ARCHITECTURE_arm64ec
|
||||
// This is used when QueryPerformanceCounter is called on recent Windows versions, it causes CNTVCT to be written into RAX.
|
||||
constexpr uint8_t GET_CNTVCT_LITERAL = 0x81;
|
||||
if (Literal == GET_CNTVCT_LITERAL) {
|
||||
StoreGPRRegister(X86State::REG_RAX, _CycleCounter(false));
|
||||
return;
|
||||
if (CTX->HostFeatures.HostType == FEXCore::HostFeatures::HostTypeEnum::Arm64ec) {
|
||||
// This is used when QueryPerformanceCounter is called on recent Windows versions, it causes CNTVCT to be written into RAX.
|
||||
constexpr uint8_t GET_CNTVCT_LITERAL = 0x81;
|
||||
if (Literal == GET_CNTVCT_LITERAL) {
|
||||
StoreGPRRegister(X86State::REG_RAX, _CycleCounter(false));
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
Reason.ErrorRegister = Literal << 3 | (0b010);
|
||||
Reason.Signal = Core::FAULT_SIGSEGV;
|
||||
@@ -4925,6 +5057,7 @@ void OpDispatchBuilder::CRC32(OpcodeArgs) {
|
||||
return;
|
||||
}
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
const auto SrcSize = OpSizeFromSrc(Op);
|
||||
|
||||
// Destination GPR size is always 4 or 8 bytes depending on widening
|
||||
const auto DstSize = Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REX_WIDENING ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
@@ -4933,16 +5066,15 @@ void OpDispatchBuilder::CRC32(OpcodeArgs) {
|
||||
// Incoming memory is 8, 16, 32, or 64
|
||||
Ref Src {};
|
||||
if (Op->Src[0].IsGPR()) {
|
||||
Src = LoadSourceGPR_WithOpSize(Op, Op->Src[0], GPRSize, Op->Flags);
|
||||
Src = LoadSourceGPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags, {.AllowUpperGarbage = true});
|
||||
} else {
|
||||
Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.Align = OpSize::i8Bit});
|
||||
}
|
||||
auto Result = _CRC32(Dest, Src, OpSizeFromSrc(Op));
|
||||
auto Result = _CRC32(Dest, Src, SrcSize);
|
||||
StoreResultGPR_WithOpSize(Op, Op->Dest, Result, DstSize);
|
||||
}
|
||||
|
||||
template<bool Reseed>
|
||||
void OpDispatchBuilder::RDRANDOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::RDRANDOp(OpcodeArgs, bool Reseed) {
|
||||
if (!CTX->HostFeatures.SupportsRAND) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
@@ -4967,9 +5099,6 @@ void OpDispatchBuilder::RDRANDOp(OpcodeArgs) {
|
||||
}
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::RDRANDOp<true>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::RDRANDOp<false>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::BreakOp(OpcodeArgs, FEXCore::IR::BreakDefinition BreakDefinition) {
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
|
||||
|
||||
@@ -303,7 +303,7 @@ public:
|
||||
StartNewBlock();
|
||||
}
|
||||
|
||||
OpDispatchBuilder(FEXCore::Context::ContextImpl* ctx);
|
||||
OpDispatchBuilder(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread);
|
||||
|
||||
// Should only be called at the start of IR Emission.
|
||||
void ResetWorkingList();
|
||||
@@ -360,7 +360,7 @@ public:
|
||||
void MOVGPRNTOp(OpcodeArgs);
|
||||
void MOVVectorAlignedOp(OpcodeArgs);
|
||||
void MOVVectorUnalignedOp(OpcodeArgs);
|
||||
void MOVVectorNTOp(OpcodeArgs);
|
||||
void MOVVectorNTOp(OpcodeArgs, bool IsAVX);
|
||||
void ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp, unsigned SrcIdx);
|
||||
void LSLOp(OpcodeArgs);
|
||||
void INTOp(OpcodeArgs);
|
||||
@@ -470,8 +470,7 @@ public:
|
||||
void AAMOp(OpcodeArgs);
|
||||
void AADOp(OpcodeArgs);
|
||||
void XLATOp(OpcodeArgs);
|
||||
template<bool Reseed>
|
||||
void RDRANDOp(OpcodeArgs);
|
||||
void RDRANDOp(OpcodeArgs, bool Reseed);
|
||||
|
||||
enum class Segment {
|
||||
FS,
|
||||
@@ -501,8 +500,7 @@ public:
|
||||
void VectorALUROp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
|
||||
void VectorUnaryOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
|
||||
void RSqrt3DNowOp(OpcodeArgs, bool Duplicate);
|
||||
template<FEXCore::IR::IROps IROp, IR::OpSize ElementSize>
|
||||
void VectorUnaryDuplicateOp(OpcodeArgs);
|
||||
void VectorUnaryDuplicateOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
|
||||
|
||||
void MOVQOp(OpcodeArgs, VectorOpType VectorType);
|
||||
void MOVQMMXOp(OpcodeArgs);
|
||||
@@ -524,36 +522,24 @@ public:
|
||||
void PSLLDQ(OpcodeArgs);
|
||||
void PSRAIOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void MOVDDUPOp(OpcodeArgs);
|
||||
template<IR::OpSize DstElementSize>
|
||||
void CVTGPR_To_FPR(OpcodeArgs);
|
||||
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
|
||||
void CVTFPR_To_GPR(OpcodeArgs);
|
||||
template<IR::OpSize SrcElementSize, bool Widen>
|
||||
void Vector_CVT_Int_To_Float(OpcodeArgs);
|
||||
template<IR::OpSize DstElementSize, IR::OpSize SrcElementSize>
|
||||
void Scalar_CVT_Float_To_Float(OpcodeArgs);
|
||||
void CVTFPR_To_GPR(OpcodeArgs, IR::OpSize SrcElementSize, bool HostRoundingMode);
|
||||
void Vector_CVT_Int_To_Float(OpcodeArgs, IR::OpSize SrcElementSize, bool Widen, bool IsAVX);
|
||||
void Vector_CVT_Float_To_Float(OpcodeArgs, IR::OpSize DstElementSize, IR::OpSize SrcElementSize, bool IsAVX);
|
||||
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
|
||||
void Vector_CVT_Float_To_Int(OpcodeArgs);
|
||||
void Vector_CVT_Float_To_Int(OpcodeArgs, IR::OpSize SrcElementSize, bool HostRoundingMode, bool IsAVX);
|
||||
void MMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs);
|
||||
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
|
||||
void XMM_To_MMX_Vector_CVT_Float_To_Int(OpcodeArgs);
|
||||
void XMM_To_MMX_Vector_CVT_Float_To_Int(OpcodeArgs, IR::OpSize SrcElementSize, bool HostRoundingMode);
|
||||
void MASKMOVOp(OpcodeArgs);
|
||||
void MOVBetweenGPR_FPR(OpcodeArgs, VectorOpType VectorType);
|
||||
void TZCNT(OpcodeArgs);
|
||||
void LZCNT(OpcodeArgs);
|
||||
template<IR::OpSize ElementSize>
|
||||
void VFCMPOp(OpcodeArgs);
|
||||
void VFCMPOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void SHUFOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
template<IR::OpSize ElementSize>
|
||||
void PINSROp(OpcodeArgs);
|
||||
void PINSROp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void InsertPSOp(OpcodeArgs);
|
||||
void PExtrOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void PSIGN(OpcodeArgs);
|
||||
template<IR::OpSize ElementSize>
|
||||
void VPSIGN(OpcodeArgs);
|
||||
void PSIGN(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void VPSIGN(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
// BMI1 Ops
|
||||
void ANDNBMIOp(OpcodeArgs);
|
||||
@@ -576,53 +562,32 @@ public:
|
||||
// AVX Ops
|
||||
void AVXVectorXOROp(OpcodeArgs);
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void AVXVectorRound(OpcodeArgs);
|
||||
void AVXVectorRound(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
template<IR::OpSize DstElementSize, IR::OpSize SrcElementSize>
|
||||
void AVXScalar_CVT_Float_To_Float(OpcodeArgs);
|
||||
void VectorScalarInsertALUOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
|
||||
void AVXVectorScalarInsertALUOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
|
||||
|
||||
template<FEXCore::IR::IROps IROp, IR::OpSize ElementSize>
|
||||
void VectorScalarInsertALUOp(OpcodeArgs);
|
||||
template<FEXCore::IR::IROps IROp, IR::OpSize ElementSize>
|
||||
void AVXVectorScalarInsertALUOp(OpcodeArgs);
|
||||
|
||||
template<FEXCore::IR::IROps IROp, IR::OpSize ElementSize>
|
||||
void VectorScalarUnaryInsertALUOp(OpcodeArgs);
|
||||
template<FEXCore::IR::IROps IROp, IR::OpSize ElementSize>
|
||||
void AVXVectorScalarUnaryInsertALUOp(OpcodeArgs);
|
||||
void VectorScalarUnaryInsertALUOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
|
||||
void AVXVectorScalarUnaryInsertALUOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
|
||||
|
||||
void InsertMMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs);
|
||||
template<IR::OpSize DstElementSize>
|
||||
void InsertCVTGPR_To_FPR(OpcodeArgs);
|
||||
template<IR::OpSize DstElementSize>
|
||||
void AVXInsertCVTGPR_To_FPR(OpcodeArgs);
|
||||
void InsertCVTGPR_To_FPR(OpcodeArgs, IR::OpSize DstElementSize);
|
||||
void AVXInsertCVTGPR_To_FPR(OpcodeArgs, IR::OpSize DstElementSize);
|
||||
|
||||
template<IR::OpSize DstElementSize, IR::OpSize SrcElementSize>
|
||||
void InsertScalar_CVT_Float_To_Float(OpcodeArgs);
|
||||
template<IR::OpSize DstElementSize, IR::OpSize SrcElementSize>
|
||||
void AVXInsertScalar_CVT_Float_To_Float(OpcodeArgs);
|
||||
void InsertScalar_CVT_Float_To_Float(OpcodeArgs, IR::OpSize DstElementSize, IR::OpSize SrcElementSize);
|
||||
void AVXInsertScalar_CVT_Float_To_Float(OpcodeArgs, IR::OpSize DstElementSize, IR::OpSize SrcElementSize);
|
||||
|
||||
RoundMode TranslateRoundType(uint8_t Mode);
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void InsertScalarRound(OpcodeArgs);
|
||||
template<IR::OpSize ElementSize>
|
||||
void AVXInsertScalarRound(OpcodeArgs);
|
||||
void InsertScalarRound(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void AVXInsertScalarRound(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void InsertScalarFCMPOp(OpcodeArgs);
|
||||
template<IR::OpSize ElementSize>
|
||||
void AVXInsertScalarFCMPOp(OpcodeArgs);
|
||||
void InsertScalarFCMPOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void AVXInsertScalarFCMPOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
template<IR::OpSize DstElementSize>
|
||||
void AVXCVTGPR_To_FPR(OpcodeArgs);
|
||||
void AVXVFCMPOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void AVXVFCMPOp(OpcodeArgs);
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void VADDSUBPOp(OpcodeArgs);
|
||||
void VADDSUBPOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
void VAESDecOp(OpcodeArgs);
|
||||
void VAESDecLastOp(OpcodeArgs);
|
||||
@@ -631,34 +596,31 @@ public:
|
||||
|
||||
void VANDNOp(OpcodeArgs);
|
||||
|
||||
Ref VBLENDOpImpl(IR::OpSize VecSize, IR::OpSize ElementSize, Ref Src1, Ref Src2, Ref ZeroRegister, uint64_t Selector);
|
||||
Ref VBLENDOpImpl(IR::OpSize VecSize, IR::OpSize ElementSize, Ref Src1, Ref Src2, uint64_t Selector);
|
||||
void VBLENDPDOp(OpcodeArgs);
|
||||
void VPBLENDDOp(OpcodeArgs);
|
||||
void VPBLENDWOp(OpcodeArgs);
|
||||
|
||||
void VBROADCASTOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void VDPPOp(OpcodeArgs);
|
||||
void VDPPOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
void VEXTRACT128Op(OpcodeArgs);
|
||||
|
||||
template<IROps IROp, IR::OpSize ElementSize>
|
||||
void VHADDPOp(OpcodeArgs);
|
||||
void VHADDPOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize);
|
||||
void VHSUBPOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
void VINSERTOp(OpcodeArgs);
|
||||
void VINSERTPSOp(OpcodeArgs);
|
||||
|
||||
template<IR::OpSize ElementSize, bool IsStore>
|
||||
void VMASKMOVOp(OpcodeArgs);
|
||||
void VMASKMOVOp(OpcodeArgs, IR::OpSize ElementSize, bool IsStore);
|
||||
|
||||
void VMOVHPOp(OpcodeArgs);
|
||||
void VMOVLPOp(OpcodeArgs);
|
||||
|
||||
void VMOVDDUPOp(OpcodeArgs);
|
||||
void VMOVSHDUPOp(OpcodeArgs);
|
||||
void VMOVSLDUPOp(OpcodeArgs);
|
||||
void VMOVSHDUPOp(OpcodeArgs, bool IsAVX);
|
||||
void VMOVSLDUPOp(OpcodeArgs, bool IsAVX);
|
||||
|
||||
void VMOVSDOp(OpcodeArgs);
|
||||
void VMOVSSOp(OpcodeArgs);
|
||||
@@ -669,15 +631,14 @@ public:
|
||||
void VMPSADBWOp(OpcodeArgs);
|
||||
|
||||
void VPACKSSOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
void VPACKUSOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
void VPALIGNROp(OpcodeArgs);
|
||||
|
||||
void VPCMPESTRIOp(OpcodeArgs);
|
||||
void VPCMPESTRMOp(OpcodeArgs);
|
||||
void VPCMPISTRIOp(OpcodeArgs);
|
||||
void VPCMPISTRMOp(OpcodeArgs);
|
||||
void VPCMPESTRIOp(OpcodeArgs, bool IsAVX);
|
||||
void VPCMPESTRMOp(OpcodeArgs, bool IsAVX);
|
||||
void VPCMPISTRIOp(OpcodeArgs, bool IsAVX);
|
||||
void VPCMPISTRMOp(OpcodeArgs, bool IsAVX);
|
||||
|
||||
void VCVTPH2PSOp(OpcodeArgs);
|
||||
void VCVTPS2PHOp(OpcodeArgs);
|
||||
@@ -690,36 +651,28 @@ public:
|
||||
void VPERMILImmOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
Ref VPERMILRegOpImpl(OpSize DstSize, IR::OpSize ElementSize, Ref Src, Ref Indices);
|
||||
template<IR::OpSize ElementSize>
|
||||
void VPERMILRegOp(OpcodeArgs);
|
||||
void VPERMILRegOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
void VPHADDSWOp(OpcodeArgs);
|
||||
|
||||
void VPHSUBOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void VPHSUBSWOp(OpcodeArgs);
|
||||
|
||||
void VPINSRBOp(OpcodeArgs);
|
||||
void VPINSRBWOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void VPINSRDQOp(OpcodeArgs);
|
||||
void VPINSRWOp(OpcodeArgs);
|
||||
|
||||
void VPMADDUBSWOp(OpcodeArgs);
|
||||
void VPMADDWDOp(OpcodeArgs);
|
||||
|
||||
template<bool IsStore>
|
||||
void VPMASKMOVOp(OpcodeArgs);
|
||||
void VPMASKMOVOp(OpcodeArgs, bool IsStore);
|
||||
|
||||
void VPMULHRSWOp(OpcodeArgs);
|
||||
|
||||
template<bool Signed>
|
||||
void VPMULHWOp(OpcodeArgs);
|
||||
|
||||
template<IR::OpSize ElementSize, bool Signed>
|
||||
void VPMULLOp(OpcodeArgs);
|
||||
void VPMULHWOp(OpcodeArgs, bool Signed);
|
||||
void VPMULLOp(OpcodeArgs, IR::OpSize ElementSize, bool Signed);
|
||||
|
||||
void VPSADBWOp(OpcodeArgs);
|
||||
|
||||
void VPSHUFBOp(OpcodeArgs);
|
||||
|
||||
void VPSHUFWOp(OpcodeArgs, IR::OpSize ElementSize, bool Low);
|
||||
|
||||
void VPSLLOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
@@ -728,7 +681,6 @@ public:
|
||||
void VPSLLVOp(OpcodeArgs);
|
||||
|
||||
void VPSRAOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
void VPSRAIOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
void VPSRAVDOp(OpcodeArgs);
|
||||
@@ -736,17 +688,14 @@ public:
|
||||
|
||||
void VPSRLDOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void VPSRLDQOp(OpcodeArgs);
|
||||
void VPSRLIOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
void VPUNPCKHOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
void VPUNPCKLOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
void VPSRLIOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
void VSHUFOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void VTESTPOp(OpcodeArgs);
|
||||
void VTESTPOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
void VZEROOp(OpcodeArgs);
|
||||
|
||||
@@ -830,32 +779,24 @@ public:
|
||||
void XSaveOp(OpcodeArgs);
|
||||
|
||||
void PAlignrOp(OpcodeArgs);
|
||||
template<IR::OpSize ElementSize>
|
||||
void UCOMISxOp(OpcodeArgs);
|
||||
void UCOMISxOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void LDMXCSR(OpcodeArgs);
|
||||
void STMXCSR(OpcodeArgs);
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void PACKUSOp(OpcodeArgs);
|
||||
void PACKUSOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void PACKSSOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void PACKSSOp(OpcodeArgs);
|
||||
void PMULLOp(OpcodeArgs, IR::OpSize ElementSize, bool Signed);
|
||||
|
||||
template<IR::OpSize ElementSize, bool Signed>
|
||||
void PMULLOp(OpcodeArgs);
|
||||
void MOVQ2DQ(OpcodeArgs, bool ToXMM);
|
||||
|
||||
template<bool ToXMM>
|
||||
void MOVQ2DQ(OpcodeArgs);
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void ADDSUBPOp(OpcodeArgs);
|
||||
void ADDSUBPOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
void PFNACCOp(OpcodeArgs);
|
||||
void PFPNACCOp(OpcodeArgs);
|
||||
void PSWAPDOp(OpcodeArgs);
|
||||
|
||||
template<uint8_t CompType>
|
||||
void VPFCMPOp(OpcodeArgs);
|
||||
void VPFCMPOp(OpcodeArgs, uint8_t CompType);
|
||||
void PI2FWOp(OpcodeArgs);
|
||||
void PF2IWOp(OpcodeArgs);
|
||||
|
||||
@@ -864,16 +805,12 @@ public:
|
||||
void PMADDWD(OpcodeArgs);
|
||||
void PMADDUBSW(OpcodeArgs);
|
||||
|
||||
template<bool Signed>
|
||||
void PMULHW(OpcodeArgs);
|
||||
|
||||
void PMULHW(OpcodeArgs, bool Signed);
|
||||
void PMULHRSW(OpcodeArgs);
|
||||
|
||||
void MOVBEOp(OpcodeArgs);
|
||||
template<IR::OpSize ElementSize>
|
||||
void HSUBP(OpcodeArgs);
|
||||
template<IR::OpSize ElementSize>
|
||||
void PHSUB(OpcodeArgs);
|
||||
void HSUBP(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void PHSUB(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
void PHADDS(OpcodeArgs);
|
||||
void PHSUBS(OpcodeArgs);
|
||||
@@ -920,25 +857,24 @@ public:
|
||||
};
|
||||
|
||||
RefVSIB LoadVSIB(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags);
|
||||
template<OpSize AddrElementSize>
|
||||
void VPGATHER(OpcodeArgs);
|
||||
void VPGATHER(OpcodeArgs, OpSize AddrElementSize);
|
||||
|
||||
template<IR::OpSize ElementSize, IR::OpSize DstElementSize, bool Signed>
|
||||
void ExtendVectorElements(OpcodeArgs);
|
||||
template<IR::OpSize ElementSize>
|
||||
void VectorRound(OpcodeArgs);
|
||||
void AVXExtendVectorElements(OpcodeArgs, IR::OpSize ElementSize, IR::OpSize DstElementSize, bool Signed);
|
||||
void ExtendVectorElements(OpcodeArgs, IR::OpSize ElementSize, IR::OpSize DstElementSize, bool Signed);
|
||||
|
||||
Ref VectorBlend(OpSize Size, IR::OpSize ElementSize, Ref Src1, Ref Src2, uint8_t Selector);
|
||||
void VectorRound(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void VectorBlend(OpcodeArgs);
|
||||
Ref VectorBlendImpl(OpSize Size, IR::OpSize ElementSize, Ref Src1, Ref Src2, uint8_t Selector);
|
||||
void VectorBlend(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
void VectorVariableBlend(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void PTestOpImpl(OpSize Size, Ref Dest, Ref Src);
|
||||
void PTestOp(OpcodeArgs);
|
||||
|
||||
void AVXPHMINPOSUWOp(OpcodeArgs);
|
||||
void PHMINPOSUWOp(OpcodeArgs);
|
||||
template<IR::OpSize ElementSize>
|
||||
void DPPOp(OpcodeArgs);
|
||||
|
||||
void DPPOp(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
void MPSADBWOp(OpcodeArgs);
|
||||
void PCLMULQDQOp(OpcodeArgs);
|
||||
@@ -1376,6 +1312,7 @@ private:
|
||||
};
|
||||
|
||||
FEXCore::Context::ContextImpl* CTX {};
|
||||
FEXCore::Core::InternalThreadState* Thread;
|
||||
|
||||
constexpr static unsigned FullNZCVMask = (1U << FEXCore::X86State::RFLAG_CF_RAW_LOC) | (1U << FEXCore::X86State::RFLAG_ZF_RAW_LOC) |
|
||||
(1U << FEXCore::X86State::RFLAG_SF_RAW_LOC) | (1U << FEXCore::X86State::RFLAG_OF_RAW_LOC);
|
||||
@@ -1443,7 +1380,7 @@ private:
|
||||
Ref PALIGNROpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm, bool IsAVX);
|
||||
|
||||
void PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask);
|
||||
void PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask, bool IsAVX);
|
||||
|
||||
Ref PHADDSOpImpl(OpSize Size, Ref Src1, Ref Src2);
|
||||
|
||||
@@ -1481,7 +1418,7 @@ private:
|
||||
|
||||
Ref PSRLDOpImpl(OpcodeArgs, IR::OpSize ElementSize, Ref Src, Ref ShiftVec);
|
||||
|
||||
Ref SHUFOpImpl(OpcodeArgs, IR::OpSize DstSize, IR::OpSize ElementSize, Ref Src1, Ref Src2, uint8_t Shuffle);
|
||||
Ref SHUFOpImpl(IR::OpSize DstSize, IR::OpSize ElementSize, Ref Src1, Ref Src2, uint8_t Shuffle);
|
||||
|
||||
void VMASKMOVOpImpl(OpcodeArgs, IR::OpSize ElementSize, IR::OpSize DataSize, bool IsStore, const X86Tables::DecodedOperand& MaskOp,
|
||||
const X86Tables::DecodedOperand& DataOp);
|
||||
@@ -1591,6 +1528,7 @@ private:
|
||||
}
|
||||
|
||||
AddressMode DecodeAddress(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, MemoryAccessType AccessType, bool IsLoad);
|
||||
uint64_t CalcAddress(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, bool IsLoad);
|
||||
|
||||
Ref LoadSource(RegClass Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags,
|
||||
const LoadSourceOptions& Options = {});
|
||||
|
||||
@@ -630,7 +630,7 @@ void OpDispatchBuilder::AVX128_VPSIGN(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_UCOMISx(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
const auto SrcSize = Op->Src[0].IsGPR() ? GetGuestVectorLength() : ElementSize;
|
||||
const auto SrcSize = Op->Src[0].IsGPR() ? OpSize::i128Bit : ElementSize;
|
||||
|
||||
auto Src1 = AVX128_LoadSource_WithOpSize(Op, Op->Dest, Op->Flags, false);
|
||||
|
||||
@@ -1260,26 +1260,26 @@ void OpDispatchBuilder::AVX128_VAESKeyGenAssist(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VPCMPESTRI(OpcodeArgs) {
|
||||
PCMPXSTRXOpImpl(Op, true, false);
|
||||
PCMPXSTRXOpImpl(Op, true, false, true);
|
||||
|
||||
///< Does not zero anything.
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VPCMPESTRM(OpcodeArgs) {
|
||||
PCMPXSTRXOpImpl(Op, true, true);
|
||||
PCMPXSTRXOpImpl(Op, true, true, true);
|
||||
|
||||
///< Zero the upper 128-bits of hardcoded YMM0
|
||||
AVX128_StoreXMMRegister(0, LoadZeroVector(OpSize::i128Bit), true);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VPCMPISTRI(OpcodeArgs) {
|
||||
PCMPXSTRXOpImpl(Op, false, false);
|
||||
PCMPXSTRXOpImpl(Op, false, false, true);
|
||||
|
||||
///< Does not zero anything.
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VPCMPISTRM(OpcodeArgs) {
|
||||
PCMPXSTRXOpImpl(Op, false, true);
|
||||
PCMPXSTRXOpImpl(Op, false, true, true);
|
||||
|
||||
///< Zero the upper 128-bits of hardcoded YMM0
|
||||
AVX128_StoreXMMRegister(0, LoadZeroVector(OpSize::i128Bit), true);
|
||||
@@ -1399,13 +1399,13 @@ void OpDispatchBuilder::AVX128_VSHUF(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
auto Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, !Is128Bit);
|
||||
|
||||
RefPair Result {};
|
||||
Result.Low = SHUFOpImpl(Op, OpSize::i128Bit, ElementSize, Src1.Low, Src2.Low, Shuffle);
|
||||
Result.Low = SHUFOpImpl(OpSize::i128Bit, ElementSize, Src1.Low, Src2.Low, Shuffle);
|
||||
|
||||
if (Is128Bit) {
|
||||
Result.High = LoadZeroVector(OpSize::i128Bit);
|
||||
} else {
|
||||
const uint8_t ShiftAmount = ElementSize == OpSize::i32Bit ? 0 : 2;
|
||||
Result.High = SHUFOpImpl(Op, OpSize::i128Bit, ElementSize, Src1.High, Src2.High, Shuffle >> ShiftAmount);
|
||||
Result.High = SHUFOpImpl(OpSize::i128Bit, ElementSize, Src1.High, Src2.High, Shuffle >> ShiftAmount);
|
||||
}
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
}
|
||||
@@ -1484,12 +1484,12 @@ void OpDispatchBuilder::AVX128_VBLEND(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
auto Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, !Is128Bit);
|
||||
|
||||
RefPair Result {};
|
||||
Result.Low = VectorBlend(OpSize::i128Bit, ElementSize, Src1.Low, Src2.Low, Selector);
|
||||
Result.Low = VectorBlendImpl(OpSize::i128Bit, ElementSize, Src1.Low, Src2.Low, Selector);
|
||||
|
||||
if (Is128Bit) {
|
||||
Result = AVX128_Zext(Result.Low);
|
||||
} else {
|
||||
Result.High = VectorBlend(OpSize::i128Bit, ElementSize, Src1.High, Src2.High, (Selector >> SelectorShift));
|
||||
Result.High = VectorBlendImpl(OpSize::i128Bit, ElementSize, Src1.High, Src2.High, (Selector >> SelectorShift));
|
||||
}
|
||||
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
@@ -2293,9 +2293,8 @@ void OpDispatchBuilder::AVX128_VCVTPS2PH(OpcodeArgs) {
|
||||
_PopRoundingMode(OldFPCR);
|
||||
}
|
||||
|
||||
// We need to eliminate upper junk if we're storing into a register with
|
||||
// a 256-bit source (VCVTPS2PH's destination for registers is an XMM).
|
||||
if (Op->Src[0].IsGPR() && SrcSize == OpSize::i256Bit) {
|
||||
// We need to zero the upper 128 bits if we're storing into a register
|
||||
if (Op->Dest.IsGPR()) {
|
||||
Result = AVX128_Zext(Result.Low);
|
||||
}
|
||||
|
||||
|
||||
@@ -5,9 +5,9 @@
|
||||
namespace FEXCore::IR {
|
||||
constexpr DispatchTableEntry OpDispatch_DDDTable[] = {
|
||||
{0x0C, 1, &OpDispatchBuilder::PI2FWOp},
|
||||
{0x0D, 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, false>},
|
||||
{0x0D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Int_To_Float, OpSize::i32Bit, false, false>},
|
||||
{0x1C, 1, &OpDispatchBuilder::PF2IWOp},
|
||||
{0x1D, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false>},
|
||||
{0x1D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Int, OpSize::i32Bit, false, false>},
|
||||
|
||||
{0x86, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRECPPRECISION, OpSize::i32Bit>},
|
||||
{0x87, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RSqrt3DNowOp, false>},
|
||||
@@ -15,15 +15,15 @@ constexpr DispatchTableEntry OpDispatch_DDDTable[] = {
|
||||
{0x8A, 1, &OpDispatchBuilder::PFNACCOp},
|
||||
{0x8E, 1, &OpDispatchBuilder::PFPNACCOp},
|
||||
|
||||
{0x90, 1, &OpDispatchBuilder::VPFCMPOp<1>},
|
||||
{0x90, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPFCMPOp, 1>},
|
||||
{0x94, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMIN, OpSize::i32Bit>},
|
||||
{0x96, 1, &OpDispatchBuilder::VectorUnaryDuplicateOp<IR::OP_VFRECPPRECISION, OpSize::i32Bit>},
|
||||
{0x96, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryDuplicateOp, IR::OP_VFRECPPRECISION, OpSize::i32Bit>},
|
||||
{0x97, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RSqrt3DNowOp, true>},
|
||||
|
||||
{0x9A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFSUB, OpSize::i32Bit>},
|
||||
{0x9E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADD, OpSize::i32Bit>},
|
||||
|
||||
{0xA0, 1, &OpDispatchBuilder::VPFCMPOp<2>},
|
||||
{0xA0, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPFCMPOp, 2>},
|
||||
{0xA4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMAX, OpSize::i32Bit>},
|
||||
// Can be treated as a move
|
||||
{0xA6, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
@@ -32,7 +32,7 @@ constexpr DispatchTableEntry OpDispatch_DDDTable[] = {
|
||||
{0xAA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUROp, IR::OP_VFSUB, OpSize::i32Bit>},
|
||||
{0xAE, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, OpSize::i32Bit>},
|
||||
|
||||
{0xB0, 1, &OpDispatchBuilder::VPFCMPOp<0>},
|
||||
{0xB0, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPFCMPOp, 0>},
|
||||
{0xB4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMUL, OpSize::i32Bit>},
|
||||
// Can be treated as a move
|
||||
{0xB6, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
|
||||
@@ -20,18 +20,18 @@ constexpr DispatchTableEntry OpDispatch_H0F38Table[] = {
|
||||
{OPD(PF_38_66, 0x03), 1, &OpDispatchBuilder::PHADDS},
|
||||
{OPD(PF_38_NONE, 0x04), 1, &OpDispatchBuilder::PMADDUBSW},
|
||||
{OPD(PF_38_66, 0x04), 1, &OpDispatchBuilder::PMADDUBSW},
|
||||
{OPD(PF_38_NONE, 0x05), 1, &OpDispatchBuilder::PHSUB<OpSize::i16Bit>},
|
||||
{OPD(PF_38_66, 0x05), 1, &OpDispatchBuilder::PHSUB<OpSize::i16Bit>},
|
||||
{OPD(PF_38_NONE, 0x06), 1, &OpDispatchBuilder::PHSUB<OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x06), 1, &OpDispatchBuilder::PHSUB<OpSize::i32Bit>},
|
||||
{OPD(PF_38_NONE, 0x05), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PHSUB, OpSize::i16Bit>},
|
||||
{OPD(PF_38_66, 0x05), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PHSUB, OpSize::i16Bit>},
|
||||
{OPD(PF_38_NONE, 0x06), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PHSUB, OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x06), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PHSUB, OpSize::i32Bit>},
|
||||
{OPD(PF_38_NONE, 0x07), 1, &OpDispatchBuilder::PHSUBS},
|
||||
{OPD(PF_38_66, 0x07), 1, &OpDispatchBuilder::PHSUBS},
|
||||
{OPD(PF_38_NONE, 0x08), 1, &OpDispatchBuilder::PSIGN<OpSize::i8Bit>},
|
||||
{OPD(PF_38_66, 0x08), 1, &OpDispatchBuilder::PSIGN<OpSize::i8Bit>},
|
||||
{OPD(PF_38_NONE, 0x09), 1, &OpDispatchBuilder::PSIGN<OpSize::i16Bit>},
|
||||
{OPD(PF_38_66, 0x09), 1, &OpDispatchBuilder::PSIGN<OpSize::i16Bit>},
|
||||
{OPD(PF_38_NONE, 0x0A), 1, &OpDispatchBuilder::PSIGN<OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x0A), 1, &OpDispatchBuilder::PSIGN<OpSize::i32Bit>},
|
||||
{OPD(PF_38_NONE, 0x08), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSIGN, OpSize::i8Bit>},
|
||||
{OPD(PF_38_66, 0x08), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSIGN, OpSize::i8Bit>},
|
||||
{OPD(PF_38_NONE, 0x09), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSIGN, OpSize::i16Bit>},
|
||||
{OPD(PF_38_66, 0x09), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSIGN, OpSize::i16Bit>},
|
||||
{OPD(PF_38_NONE, 0x0A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSIGN, OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x0A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSIGN, OpSize::i32Bit>},
|
||||
{OPD(PF_38_NONE, 0x0B), 1, &OpDispatchBuilder::PMULHRSW},
|
||||
{OPD(PF_38_66, 0x0B), 1, &OpDispatchBuilder::PMULHRSW},
|
||||
{OPD(PF_38_66, 0x10), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorVariableBlend, OpSize::i8Bit>},
|
||||
@@ -44,22 +44,22 @@ constexpr DispatchTableEntry OpDispatch_H0F38Table[] = {
|
||||
{OPD(PF_38_66, 0x1D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, OpSize::i16Bit>},
|
||||
{OPD(PF_38_NONE, 0x1E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x1E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VABS, OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x20), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i16Bit, true>},
|
||||
{OPD(PF_38_66, 0x21), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i32Bit, true>},
|
||||
{OPD(PF_38_66, 0x22), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i64Bit, true>},
|
||||
{OPD(PF_38_66, 0x23), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i16Bit, OpSize::i32Bit, true>},
|
||||
{OPD(PF_38_66, 0x24), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i16Bit, OpSize::i64Bit, true>},
|
||||
{OPD(PF_38_66, 0x25), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i32Bit, OpSize::i64Bit, true>},
|
||||
{OPD(PF_38_66, 0x28), 1, &OpDispatchBuilder::PMULLOp<OpSize::i32Bit, true>},
|
||||
{OPD(PF_38_66, 0x20), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ExtendVectorElements, OpSize::i8Bit, OpSize::i16Bit, true>},
|
||||
{OPD(PF_38_66, 0x21), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ExtendVectorElements, OpSize::i8Bit, OpSize::i32Bit, true>},
|
||||
{OPD(PF_38_66, 0x22), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ExtendVectorElements, OpSize::i8Bit, OpSize::i64Bit, true>},
|
||||
{OPD(PF_38_66, 0x23), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ExtendVectorElements, OpSize::i16Bit, OpSize::i32Bit, true>},
|
||||
{OPD(PF_38_66, 0x24), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ExtendVectorElements, OpSize::i16Bit, OpSize::i64Bit, true>},
|
||||
{OPD(PF_38_66, 0x25), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ExtendVectorElements, OpSize::i32Bit, OpSize::i64Bit, true>},
|
||||
{OPD(PF_38_66, 0x28), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PMULLOp, OpSize::i32Bit, true>},
|
||||
{OPD(PF_38_66, 0x29), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i64Bit>},
|
||||
{OPD(PF_38_66, 0x2A), 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{OPD(PF_38_66, 0x2B), 1, &OpDispatchBuilder::PACKUSOp<OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x30), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i16Bit, false>},
|
||||
{OPD(PF_38_66, 0x31), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i32Bit, false>},
|
||||
{OPD(PF_38_66, 0x32), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i64Bit, false>},
|
||||
{OPD(PF_38_66, 0x33), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i16Bit, OpSize::i32Bit, false>},
|
||||
{OPD(PF_38_66, 0x34), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i16Bit, OpSize::i64Bit, false>},
|
||||
{OPD(PF_38_66, 0x35), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i32Bit, OpSize::i64Bit, false>},
|
||||
{OPD(PF_38_66, 0x2A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVVectorNTOp, false>},
|
||||
{OPD(PF_38_66, 0x2B), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PACKUSOp, OpSize::i32Bit>},
|
||||
{OPD(PF_38_66, 0x30), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ExtendVectorElements, OpSize::i8Bit, OpSize::i16Bit, false>},
|
||||
{OPD(PF_38_66, 0x31), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ExtendVectorElements, OpSize::i8Bit, OpSize::i32Bit, false>},
|
||||
{OPD(PF_38_66, 0x32), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ExtendVectorElements, OpSize::i8Bit, OpSize::i64Bit, false>},
|
||||
{OPD(PF_38_66, 0x33), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ExtendVectorElements, OpSize::i16Bit, OpSize::i32Bit, false>},
|
||||
{OPD(PF_38_66, 0x34), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ExtendVectorElements, OpSize::i16Bit, OpSize::i64Bit, false>},
|
||||
{OPD(PF_38_66, 0x35), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ExtendVectorElements, OpSize::i32Bit, OpSize::i64Bit, false>},
|
||||
{OPD(PF_38_66, 0x37), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, OpSize::i64Bit>},
|
||||
{OPD(PF_38_66, 0x38), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMIN, OpSize::i8Bit>},
|
||||
{OPD(PF_38_66, 0x39), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMIN, OpSize::i32Bit>},
|
||||
|
||||
@@ -9,13 +9,13 @@ namespace FEXCore::IR {
|
||||
constexpr auto OpDispatchTableGenH0F3A = []() consteval {
|
||||
constexpr auto OpDispatchTableGenH0F3AREX = []<uint16_t REX>() consteval {
|
||||
constexpr DispatchTableEntry Table[] = {
|
||||
{OPD(REX, PF_3A_66, 0x08), 1, &OpDispatchBuilder::VectorRound<OpSize::i32Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x09), 1, &OpDispatchBuilder::VectorRound<OpSize::i64Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x0A), 1, &OpDispatchBuilder::InsertScalarRound<OpSize::i32Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x0B), 1, &OpDispatchBuilder::InsertScalarRound<OpSize::i64Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x0C), 1, &OpDispatchBuilder::VectorBlend<OpSize::i32Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x0D), 1, &OpDispatchBuilder::VectorBlend<OpSize::i64Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x0E), 1, &OpDispatchBuilder::VectorBlend<OpSize::i16Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x08), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorRound, OpSize::i32Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x09), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorRound, OpSize::i64Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x0A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::InsertScalarRound, OpSize::i32Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x0B), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::InsertScalarRound, OpSize::i64Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x0C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorBlend, OpSize::i32Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x0D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorBlend, OpSize::i64Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x0E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorBlend, OpSize::i16Bit>},
|
||||
|
||||
{OPD(REX, PF_3A_NONE, 0x0F), 1, &OpDispatchBuilder::PAlignrOp},
|
||||
{OPD(REX, PF_3A_66, 0x0F), 1, &OpDispatchBuilder::PAlignrOp},
|
||||
@@ -24,17 +24,17 @@ constexpr auto OpDispatchTableGenH0F3A = []() consteval {
|
||||
{OPD(REX, PF_3A_66, 0x15), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i16Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x17), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i32Bit>},
|
||||
|
||||
{OPD(REX, PF_3A_66, 0x20), 1, &OpDispatchBuilder::PINSROp<OpSize::i8Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x20), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PINSROp, OpSize::i8Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x21), 1, &OpDispatchBuilder::InsertPSOp},
|
||||
{OPD(REX, PF_3A_66, 0x40), 1, &OpDispatchBuilder::DPPOp<OpSize::i32Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x41), 1, &OpDispatchBuilder::DPPOp<OpSize::i64Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x40), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::DPPOp, OpSize::i32Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x41), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::DPPOp, OpSize::i64Bit>},
|
||||
{OPD(REX, PF_3A_66, 0x42), 1, &OpDispatchBuilder::MPSADBWOp},
|
||||
{OPD(REX, PF_3A_66, 0x44), 1, &OpDispatchBuilder::PCLMULQDQOp},
|
||||
|
||||
{OPD(REX, PF_3A_66, 0x60), 1, &OpDispatchBuilder::VPCMPESTRMOp},
|
||||
{OPD(REX, PF_3A_66, 0x61), 1, &OpDispatchBuilder::VPCMPESTRIOp},
|
||||
{OPD(REX, PF_3A_66, 0x62), 1, &OpDispatchBuilder::VPCMPISTRMOp},
|
||||
{OPD(REX, PF_3A_66, 0x63), 1, &OpDispatchBuilder::VPCMPISTRIOp},
|
||||
{OPD(REX, PF_3A_66, 0x60), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPCMPESTRMOp, false>},
|
||||
{OPD(REX, PF_3A_66, 0x61), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPCMPESTRIOp, false>},
|
||||
{OPD(REX, PF_3A_66, 0x62), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPCMPISTRMOp, false>},
|
||||
{OPD(REX, PF_3A_66, 0x63), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPCMPISTRIOp, false>},
|
||||
|
||||
{OPD(REX, PF_3A_NONE, 0xCC), 1, &OpDispatchBuilder::SHA1RNDS4Op},
|
||||
{OPD(REX, PF_3A_66, 0xDF), 1, &OpDispatchBuilder::AESKeyGenAssist},
|
||||
@@ -65,7 +65,7 @@ constexpr auto OpDispatch_H0F3ATableIgnoreREX = OpDispatchTableGenH0F3A();
|
||||
|
||||
constexpr DispatchTableEntry OpDispatch_H0F3ATableNeedsREX0[] = {
|
||||
{OPD(0, PF_3A_66, 0x16), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i32Bit>},
|
||||
{OPD(0, PF_3A_66, 0x22), 1, &OpDispatchBuilder::PINSROp<OpSize::i32Bit>},
|
||||
{OPD(0, PF_3A_66, 0x22), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PINSROp, OpSize::i32Bit>},
|
||||
};
|
||||
|
||||
#undef PF_3A_NONE
|
||||
|
||||
@@ -69,12 +69,12 @@ constexpr DispatchTableEntry OpDispatch_SecondaryGroupTables[] = {
|
||||
|
||||
// GROUP 9
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_NONE, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_NONE, 6), 1, &OpDispatchBuilder::RDRANDOp<false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_NONE, 7), 1, &OpDispatchBuilder::RDRANDOp<true>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_NONE, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RDRANDOp, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_NONE, 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RDRANDOp, true>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_66, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_66, 6), 1, &OpDispatchBuilder::RDRANDOp<false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_66, 7), 1, &OpDispatchBuilder::RDRANDOp<true>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_66, 6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RDRANDOp, false>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_66, 7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::RDRANDOp, true>},
|
||||
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_9, PF_F2, 1), 1, &OpDispatchBuilder::CMPXCHGPairOp},
|
||||
|
||||
|
||||
@@ -6,8 +6,7 @@ namespace FEXCore::IR {
|
||||
constexpr DispatchTableEntry OpDispatch_TwoByteOpTable[] = {
|
||||
// Instructions
|
||||
{0x03, 1, &OpDispatchBuilder::LSLOp},
|
||||
{0x06, 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{0x07, 1, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{0x06, 4, &OpDispatchBuilder::PermissionRestrictedOp},
|
||||
{0x0B, 1, &OpDispatchBuilder::INTOp},
|
||||
{0x0E, 1, &OpDispatchBuilder::X87EMMS},
|
||||
|
||||
@@ -44,7 +43,7 @@ constexpr DispatchTableEntry OpDispatch_TwoByteOpTable[] = {
|
||||
{0xBE, 2, &OpDispatchBuilder::MOVSXOp},
|
||||
{0xC0, 2, &OpDispatchBuilder::XADDOp},
|
||||
{0xC3, 1, &OpDispatchBuilder::MOVGPRNTOp},
|
||||
{0xC4, 1, &OpDispatchBuilder::PINSROp<OpSize::i16Bit>},
|
||||
{0xC4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PINSROp, OpSize::i16Bit>},
|
||||
{0xC5, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i16Bit>},
|
||||
{0xC8, 8, &OpDispatchBuilder::BSWAPOp},
|
||||
|
||||
@@ -56,10 +55,10 @@ constexpr DispatchTableEntry OpDispatch_TwoByteOpTable[] = {
|
||||
{0x16, 2, &OpDispatchBuilder::MOVHPDOp},
|
||||
{0x28, 2, &OpDispatchBuilder::MOVVectorAlignedOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::InsertMMX_To_XMM_Vector_CVT_Int_To_Float},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i32Bit, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i32Bit, true>},
|
||||
{0x2E, 2, &OpDispatchBuilder::UCOMISxOp<OpSize::i32Bit>},
|
||||
{0x2B, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVVectorNTOp, false>},
|
||||
{0x2C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int, OpSize::i32Bit, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int, OpSize::i32Bit, true>},
|
||||
{0x2E, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::UCOMISxOp, OpSize::i32Bit>},
|
||||
{0x50, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVMSKOp, OpSize::i32Bit>},
|
||||
{0x51, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFSQRT, OpSize::i32Bit>},
|
||||
{0x52, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFRSQRT, OpSize::i32Bit>},
|
||||
@@ -71,7 +70,7 @@ constexpr DispatchTableEntry OpDispatch_TwoByteOpTable[] = {
|
||||
{0x58, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADD, OpSize::i32Bit>},
|
||||
{0x59, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMUL, OpSize::i32Bit>},
|
||||
{0x5A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Float, OpSize::i64Bit, OpSize::i32Bit, false>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, false>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Int_To_Float, OpSize::i32Bit, false, false>},
|
||||
{0x5C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFSUB, OpSize::i32Bit>},
|
||||
{0x5D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMIN, OpSize::i32Bit>},
|
||||
{0x5E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFDIV, OpSize::i32Bit>},
|
||||
@@ -79,15 +78,15 @@ constexpr DispatchTableEntry OpDispatch_TwoByteOpTable[] = {
|
||||
{0x60, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i8Bit>},
|
||||
{0x61, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i16Bit>},
|
||||
{0x62, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i32Bit>},
|
||||
{0x63, 1, &OpDispatchBuilder::PACKSSOp<OpSize::i16Bit>},
|
||||
{0x63, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PACKSSOp, OpSize::i16Bit>},
|
||||
{0x64, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, OpSize::i8Bit>},
|
||||
{0x65, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, OpSize::i16Bit>},
|
||||
{0x66, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, OpSize::i32Bit>},
|
||||
{0x67, 1, &OpDispatchBuilder::PACKUSOp<OpSize::i16Bit>},
|
||||
{0x67, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PACKUSOp, OpSize::i16Bit>},
|
||||
{0x68, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i8Bit>},
|
||||
{0x69, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i16Bit>},
|
||||
{0x6A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i32Bit>},
|
||||
{0x6B, 1, &OpDispatchBuilder::PACKSSOp<OpSize::i32Bit>},
|
||||
{0x6B, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PACKSSOp, OpSize::i32Bit>},
|
||||
{0x70, 1, &OpDispatchBuilder::PSHUFW8ByteOp},
|
||||
|
||||
{0x74, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i8Bit>},
|
||||
@@ -95,7 +94,7 @@ constexpr DispatchTableEntry OpDispatch_TwoByteOpTable[] = {
|
||||
{0x76, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i32Bit>},
|
||||
{0x77, 1, &OpDispatchBuilder::X87EMMS},
|
||||
|
||||
{0xC2, 1, &OpDispatchBuilder::VFCMPOp<OpSize::i32Bit>},
|
||||
{0xC2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VFCMPOp, OpSize::i32Bit>},
|
||||
{0xC6, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHUFOp, OpSize::i32Bit>},
|
||||
|
||||
{0xD1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, OpSize::i16Bit>},
|
||||
@@ -116,9 +115,9 @@ constexpr DispatchTableEntry OpDispatch_TwoByteOpTable[] = {
|
||||
{0xE1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAOp, OpSize::i16Bit>},
|
||||
{0xE2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAOp, OpSize::i32Bit>},
|
||||
{0xE3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VURAVG, OpSize::i16Bit>},
|
||||
{0xE4, 1, &OpDispatchBuilder::PMULHW<false>},
|
||||
{0xE5, 1, &OpDispatchBuilder::PMULHW<true>},
|
||||
{0xE7, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0xE4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PMULHW, false>},
|
||||
{0xE5, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PMULHW, true>},
|
||||
{0xE7, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVVectorNTOp, false>},
|
||||
{0xE8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQSUB, OpSize::i8Bit>},
|
||||
{0xE9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQSUB, OpSize::i16Bit>},
|
||||
{0xEA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMIN, OpSize::i16Bit>},
|
||||
@@ -131,7 +130,7 @@ constexpr DispatchTableEntry OpDispatch_TwoByteOpTable[] = {
|
||||
{0xF1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, OpSize::i16Bit>},
|
||||
{0xF2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, OpSize::i32Bit>},
|
||||
{0xF3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, OpSize::i64Bit>},
|
||||
{0xF4, 1, &OpDispatchBuilder::PMULLOp<OpSize::i32Bit, false>},
|
||||
{0xF4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PMULLOp, OpSize::i32Bit, false>},
|
||||
{0xF5, 1, &OpDispatchBuilder::PMADDWD},
|
||||
{0xF6, 1, &OpDispatchBuilder::PSADBW},
|
||||
{0xF7, 1, &OpDispatchBuilder::MASKMOVOp},
|
||||
@@ -152,23 +151,23 @@ constexpr DispatchTableEntry OpDispatch_TwoByteOpTable[] = {
|
||||
|
||||
constexpr DispatchTableEntry OpDispatch_SecondaryRepModTables[] = {
|
||||
{0x10, 2, &OpDispatchBuilder::MOVSSOp},
|
||||
{0x12, 1, &OpDispatchBuilder::VMOVSLDUPOp},
|
||||
{0x16, 1, &OpDispatchBuilder::VMOVSHDUPOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::InsertCVTGPR_To_FPR<OpSize::i32Bit>},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::CVTFPR_To_GPR<OpSize::i32Bit, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::CVTFPR_To_GPR<OpSize::i32Bit, true>},
|
||||
{0x51, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x52, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFRSQRTSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x53, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFRECPSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x58, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x59, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x5A, 1, &OpDispatchBuilder::InsertScalar_CVT_Float_To_Float<OpSize::i64Bit, OpSize::i32Bit>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false>},
|
||||
{0x5C, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x5D, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x5E, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x5F, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x12, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VMOVSLDUPOp, false>},
|
||||
{0x16, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VMOVSHDUPOp, false>},
|
||||
{0x2A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::InsertCVTGPR_To_FPR, OpSize::i32Bit>},
|
||||
{0x2B, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVVectorNTOp, false>},
|
||||
{0x2C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CVTFPR_To_GPR, OpSize::i32Bit, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CVTFPR_To_GPR, OpSize::i32Bit, true>},
|
||||
{0x51, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarUnaryInsertALUOp, IR::OP_VFSQRTSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x52, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarUnaryInsertALUOp, IR::OP_VFRSQRTSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x53, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarUnaryInsertALUOp, IR::OP_VFRECPSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x58, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarInsertALUOp, IR::OP_VFADDSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x59, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarInsertALUOp, IR::OP_VFMULSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x5A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::InsertScalar_CVT_Float_To_Float, OpSize::i64Bit, OpSize::i32Bit>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Int, OpSize::i32Bit, false, false>},
|
||||
{0x5C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarInsertALUOp, IR::OP_VFSUBSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x5D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarInsertALUOp, IR::OP_VFMINSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x5E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarInsertALUOp, IR::OP_VFDIVSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x5F, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarInsertALUOp, IR::OP_VFMAXSCALARINSERT, OpSize::i32Bit>},
|
||||
{0x6F, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
{0x70, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSHUFWOp, false>},
|
||||
{0x7E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVQOp, OpDispatchBuilder::VectorOpType::SSE>},
|
||||
@@ -176,36 +175,36 @@ constexpr DispatchTableEntry OpDispatch_SecondaryRepModTables[] = {
|
||||
{0xB8, 1, &OpDispatchBuilder::PopcountOp},
|
||||
{0xBC, 1, &OpDispatchBuilder::TZCNT},
|
||||
{0xBD, 1, &OpDispatchBuilder::LZCNT},
|
||||
{0xC2, 1, &OpDispatchBuilder::InsertScalarFCMPOp<OpSize::i32Bit>},
|
||||
{0xD6, 1, &OpDispatchBuilder::MOVQ2DQ<true>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, true>},
|
||||
{0xC2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::InsertScalarFCMPOp, OpSize::i32Bit>},
|
||||
{0xD6, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVQ2DQ, true>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Int_To_Float, OpSize::i32Bit, true, false>},
|
||||
};
|
||||
|
||||
constexpr DispatchTableEntry OpDispatch_SecondaryRepNEModTables[] = {
|
||||
{0x10, 2, &OpDispatchBuilder::MOVSDOp},
|
||||
{0x12, 1, &OpDispatchBuilder::MOVDDUPOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::InsertCVTGPR_To_FPR<OpSize::i64Bit>},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::CVTFPR_To_GPR<OpSize::i64Bit, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::CVTFPR_To_GPR<OpSize::i64Bit, true>},
|
||||
{0x51, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x2A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::InsertCVTGPR_To_FPR, OpSize::i64Bit>},
|
||||
{0x2B, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVVectorNTOp, false>},
|
||||
{0x2C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CVTFPR_To_GPR, OpSize::i64Bit, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CVTFPR_To_GPR, OpSize::i64Bit, true>},
|
||||
{0x51, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarUnaryInsertALUOp, IR::OP_VFSQRTSCALARINSERT, OpSize::i64Bit>},
|
||||
// x52 = Invalid
|
||||
{0x58, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x59, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x5A, 1, &OpDispatchBuilder::InsertScalar_CVT_Float_To_Float<OpSize::i32Bit, OpSize::i64Bit>},
|
||||
{0x5C, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x5D, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x5E, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x5F, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x58, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarInsertALUOp, IR::OP_VFADDSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x59, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarInsertALUOp, IR::OP_VFMULSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x5A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::InsertScalar_CVT_Float_To_Float, OpSize::i32Bit, OpSize::i64Bit>},
|
||||
{0x5C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarInsertALUOp, IR::OP_VFSUBSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x5D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarInsertALUOp, IR::OP_VFMINSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x5E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarInsertALUOp, IR::OP_VFDIVSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x5F, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorScalarInsertALUOp, IR::OP_VFMAXSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x70, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSHUFWOp, true>},
|
||||
{0x78, 1, &OpDispatchBuilder::Insertq_imm},
|
||||
{0x79, 1, &OpDispatchBuilder::Insertq},
|
||||
{0x7C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, OpSize::i32Bit>},
|
||||
{0x7D, 1, &OpDispatchBuilder::HSUBP<OpSize::i32Bit>},
|
||||
{0xD0, 1, &OpDispatchBuilder::ADDSUBPOp<OpSize::i32Bit>},
|
||||
{0xD6, 1, &OpDispatchBuilder::MOVQ2DQ<false>},
|
||||
{0xC2, 1, &OpDispatchBuilder::InsertScalarFCMPOp<OpSize::i64Bit>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, true>},
|
||||
{0x7D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::HSUBP, OpSize::i32Bit>},
|
||||
{0xD0, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADDSUBPOp, OpSize::i32Bit>},
|
||||
{0xD6, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVQ2DQ, false>},
|
||||
{0xC2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::InsertScalarFCMPOp, OpSize::i64Bit>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Int, OpSize::i64Bit, true, false>},
|
||||
{0xF0, 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
};
|
||||
|
||||
@@ -217,10 +216,10 @@ constexpr DispatchTableEntry OpDispatch_SecondaryOpSizeModTables[] = {
|
||||
{0x16, 2, &OpDispatchBuilder::MOVHPDOp},
|
||||
{0x28, 2, &OpDispatchBuilder::MOVVectorAlignedOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i64Bit, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i64Bit, true>},
|
||||
{0x2E, 2, &OpDispatchBuilder::UCOMISxOp<OpSize::i64Bit>},
|
||||
{0x2B, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVVectorNTOp, false>},
|
||||
{0x2C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int, OpSize::i64Bit, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int, OpSize::i64Bit, true>},
|
||||
{0x2E, 2, &OpDispatchBuilder::Bind<&OpDispatchBuilder::UCOMISxOp, OpSize::i64Bit>},
|
||||
|
||||
{0x50, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVMSKOp, OpSize::i64Bit>},
|
||||
{0x51, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorUnaryOp, IR::OP_VFSQRT, OpSize::i64Bit>},
|
||||
@@ -231,7 +230,7 @@ constexpr DispatchTableEntry OpDispatch_SecondaryOpSizeModTables[] = {
|
||||
{0x58, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADD, OpSize::i64Bit>},
|
||||
{0x59, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMUL, OpSize::i64Bit>},
|
||||
{0x5A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Float, OpSize::i32Bit, OpSize::i64Bit, false>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, true>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Int, OpSize::i32Bit, true, false>},
|
||||
{0x5C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFSUB, OpSize::i64Bit>},
|
||||
{0x5D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFMIN, OpSize::i64Bit>},
|
||||
{0x5E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFDIV, OpSize::i64Bit>},
|
||||
@@ -239,15 +238,15 @@ constexpr DispatchTableEntry OpDispatch_SecondaryOpSizeModTables[] = {
|
||||
{0x60, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i8Bit>},
|
||||
{0x61, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i16Bit>},
|
||||
{0x62, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i32Bit>},
|
||||
{0x63, 1, &OpDispatchBuilder::PACKSSOp<OpSize::i16Bit>},
|
||||
{0x63, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PACKSSOp, OpSize::i16Bit>},
|
||||
{0x64, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, OpSize::i8Bit>},
|
||||
{0x65, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, OpSize::i16Bit>},
|
||||
{0x66, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPGT, OpSize::i32Bit>},
|
||||
{0x67, 1, &OpDispatchBuilder::PACKUSOp<OpSize::i16Bit>},
|
||||
{0x67, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PACKUSOp, OpSize::i16Bit>},
|
||||
{0x68, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i8Bit>},
|
||||
{0x69, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i16Bit>},
|
||||
{0x6A, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i32Bit>},
|
||||
{0x6B, 1, &OpDispatchBuilder::PACKSSOp<OpSize::i32Bit>},
|
||||
{0x6B, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PACKSSOp, OpSize::i32Bit>},
|
||||
{0x6C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKLOp, OpSize::i64Bit>},
|
||||
{0x6D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PUNPCKHOp, OpSize::i64Bit>},
|
||||
{0x6E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVBetweenGPR_FPR, OpDispatchBuilder::VectorOpType::SSE>},
|
||||
@@ -260,15 +259,15 @@ constexpr DispatchTableEntry OpDispatch_SecondaryOpSizeModTables[] = {
|
||||
{0x78, 1, nullptr}, // GROUP 17
|
||||
{0x79, 1, &OpDispatchBuilder::Extrq},
|
||||
{0x7C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, OpSize::i64Bit>},
|
||||
{0x7D, 1, &OpDispatchBuilder::HSUBP<OpSize::i64Bit>},
|
||||
{0x7D, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::HSUBP, OpSize::i64Bit>},
|
||||
{0x7E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVBetweenGPR_FPR, OpDispatchBuilder::VectorOpType::SSE>},
|
||||
{0x7F, 1, &OpDispatchBuilder::MOVVectorAlignedOp},
|
||||
{0xC2, 1, &OpDispatchBuilder::VFCMPOp<OpSize::i64Bit>},
|
||||
{0xC4, 1, &OpDispatchBuilder::PINSROp<OpSize::i16Bit>},
|
||||
{0xC2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VFCMPOp, OpSize::i64Bit>},
|
||||
{0xC4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PINSROp, OpSize::i16Bit>},
|
||||
{0xC5, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i16Bit>},
|
||||
{0xC6, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::SHUFOp, OpSize::i64Bit>},
|
||||
|
||||
{0xD0, 1, &OpDispatchBuilder::ADDSUBPOp<OpSize::i64Bit>},
|
||||
{0xD0, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::ADDSUBPOp, OpSize::i64Bit>},
|
||||
{0xD1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, OpSize::i16Bit>},
|
||||
{0xD2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, OpSize::i32Bit>},
|
||||
{0xD3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRLDOp, OpSize::i64Bit>},
|
||||
@@ -288,10 +287,10 @@ constexpr DispatchTableEntry OpDispatch_SecondaryOpSizeModTables[] = {
|
||||
{0xE1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAOp, OpSize::i16Bit>},
|
||||
{0xE2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSRAOp, OpSize::i32Bit>},
|
||||
{0xE3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VURAVG, OpSize::i16Bit>},
|
||||
{0xE4, 1, &OpDispatchBuilder::PMULHW<false>},
|
||||
{0xE5, 1, &OpDispatchBuilder::PMULHW<true>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, false>},
|
||||
{0xE7, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0xE4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PMULHW, false>},
|
||||
{0xE5, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PMULHW, true>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Int, OpSize::i64Bit, false, false>},
|
||||
{0xE7, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVVectorNTOp, false>},
|
||||
{0xE8, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQSUB, OpSize::i8Bit>},
|
||||
{0xE9, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSQSUB, OpSize::i16Bit>},
|
||||
{0xEA, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VSMIN, OpSize::i16Bit>},
|
||||
@@ -304,7 +303,7 @@ constexpr DispatchTableEntry OpDispatch_SecondaryOpSizeModTables[] = {
|
||||
{0xF1, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, OpSize::i16Bit>},
|
||||
{0xF2, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, OpSize::i32Bit>},
|
||||
{0xF3, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSLL, OpSize::i64Bit>},
|
||||
{0xF4, 1, &OpDispatchBuilder::PMULLOp<OpSize::i32Bit, false>},
|
||||
{0xF4, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PMULLOp, OpSize::i32Bit, false>},
|
||||
{0xF5, 1, &OpDispatchBuilder::PMADDWD},
|
||||
{0xF6, 1, &OpDispatchBuilder::PSADBW},
|
||||
{0xF7, 1, &OpDispatchBuilder::MASKMOVOp},
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -624,13 +624,10 @@ void OpDispatchBuilder::FXCH(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::X87FYL2X(OpcodeArgs, bool IsFYL2XP1) {
|
||||
if (IsFYL2XP1) {
|
||||
// create an add between top of stack and 1.
|
||||
Ref One = ReducedPrecisionMode ? _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, Constant(0x3FF0000000000000)) :
|
||||
LoadAndCacheNamedVectorConstant(OpSize::i128Bit, NamedVectorConstant::NAMED_VECTOR_X87_ONE);
|
||||
_F80AddValue(0, One);
|
||||
_F80FYL2XP1Stack();
|
||||
} else {
|
||||
_F80FYL2XStack();
|
||||
}
|
||||
|
||||
_F80FYL2XStack();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FCOMI(OpcodeArgs, IR::OpSize Width, bool Integer, OpDispatchBuilder::FCOMIFlags WhichFlags, bool PopTwice) {
|
||||
|
||||
@@ -200,8 +200,8 @@ const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
{0xF3, 1, X86InstInfo{"REP", TYPE_PREFIX, FLAGS_NONE, 0}},
|
||||
|
||||
// Instructions
|
||||
{0x00, 1, X86InstInfo{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x01, 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_DISPLACE_SIZE_DIV_2, 0}},
|
||||
{0x00, 1, X86InstInfo{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x01, 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x02, 1, X86InstInfo{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x03, 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x04, 1, X86InstInfo{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
@@ -210,16 +210,16 @@ const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
{0x06, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_06] }}},
|
||||
{0x07, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_07] }}},
|
||||
|
||||
{0x08, 1, X86InstInfo{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x09, 1, X86InstInfo{"OR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x08, 1, X86InstInfo{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x09, 1, X86InstInfo{"OR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x0A, 1, X86InstInfo{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x0B, 1, X86InstInfo{"OR", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x0C, 1, X86InstInfo{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0x0D, 1, X86InstInfo{"OR", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{0x0E, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_0E] }}},
|
||||
|
||||
{0x10, 1, X86InstInfo{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x11, 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_DISPLACE_SIZE_DIV_2, 0}},
|
||||
{0x10, 1, X86InstInfo{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x11, 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x12, 1, X86InstInfo{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x13, 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x14, 1, X86InstInfo{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
@@ -227,8 +227,8 @@ const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
{0x16, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_16] }}},
|
||||
{0x17, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_17] }}},
|
||||
|
||||
{0x18, 1, X86InstInfo{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x19, 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_DISPLACE_SIZE_DIV_2, 0}},
|
||||
{0x18, 1, X86InstInfo{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x19, 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x1A, 1, X86InstInfo{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x1B, 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x1C, 1, X86InstInfo{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
@@ -236,24 +236,24 @@ const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
{0x1E, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_1E] }}},
|
||||
{0x1F, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_1F] }}},
|
||||
|
||||
{0x20, 1, X86InstInfo{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x21, 1, X86InstInfo{"AND", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x20, 1, X86InstInfo{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x21, 1, X86InstInfo{"AND", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x22, 1, X86InstInfo{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x23, 1, X86InstInfo{"AND", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x24, 1, X86InstInfo{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0x25, 1, X86InstInfo{"AND", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
|
||||
{0x27, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_27] }}},
|
||||
{0x28, 1, X86InstInfo{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x29, 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x28, 1, X86InstInfo{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x29, 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x2A, 1, X86InstInfo{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x2B, 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x2C, 1, X86InstInfo{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
{0x2D, 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{0x2F, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Primary_ArchSelect_LUT[ENTRY_2F] }}},
|
||||
|
||||
{0x30, 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x31, 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x30, 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x31, 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x32, 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM, 0}},
|
||||
{0x33, 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_MODRM, 0}},
|
||||
{0x34, 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1}},
|
||||
@@ -310,8 +310,8 @@ const std::array<X86InstInfo, MAX_PRIMARY_TABLE_SIZE> BaseOps = []() consteval {
|
||||
|
||||
{0x84, 1, X86InstInfo{"TEST", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x85, 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x86, 1, X86InstInfo{"XCHG", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x87, 1, X86InstInfo{"XCHG", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x86, 1, X86InstInfo{"XCHG", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0x87, 1, X86InstInfo{"XCHG", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
|
||||
{0x88, 1, X86InstInfo{"MOV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0x89, 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
|
||||
@@ -34,7 +34,7 @@ constexpr std::array<X86InstInfo[2], ENTRY_MAX> H0F3A_ArchSelect_LUT = {{
|
||||
// ENTRY_1_3A_66_22
|
||||
{
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
{"PINSRQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1, { .OpDispatch = &IR::OpDispatchBuilder::PINSROp<IR::OpSize::i64Bit> }},
|
||||
{"PINSRQ", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_64BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_SRC_GPR, 1, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::PINSROp, IR::OpSize::i64Bit> }},
|
||||
},
|
||||
}};
|
||||
|
||||
|
||||
@@ -28,31 +28,31 @@ enum PrimaryGroup_LUT {
|
||||
|
||||
constexpr std::array<X86InstInfo[2], ENTRY_MAX> PrimaryGroup_ArchSelect_LUT = {{
|
||||
{
|
||||
{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1, { .OpDispatch = &IR::OpDispatchBuilder::SecondaryALUOp }},
|
||||
{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1, { .OpDispatch = &IR::OpDispatchBuilder::SecondaryALUOp }},
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
{
|
||||
{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1, { .OpDispatch = &IR::OpDispatchBuilder::SecondaryALUOp }},
|
||||
{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1, { .OpDispatch = &IR::OpDispatchBuilder::SecondaryALUOp }},
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
{
|
||||
{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::ADCOp, 1> }},
|
||||
{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::ADCOp, 1> }},
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
{
|
||||
{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::SBBOp, 1> }},
|
||||
{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1, { .OpDispatch = &IR::OpDispatchBuilder::Bind<&IR::OpDispatchBuilder::SBBOp, 1> }},
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
{
|
||||
{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1, { .OpDispatch = &IR::OpDispatchBuilder::SecondaryALUOp }},
|
||||
{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1, { .OpDispatch = &IR::OpDispatchBuilder::SecondaryALUOp }},
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
{
|
||||
{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1, { .OpDispatch = &IR::OpDispatchBuilder::SecondaryALUOp }},
|
||||
{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1, { .OpDispatch = &IR::OpDispatchBuilder::SecondaryALUOp }},
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
{
|
||||
{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1, { .OpDispatch = &IR::OpDispatchBuilder::SecondaryALUOp }},
|
||||
{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1, { .OpDispatch = &IR::OpDispatchBuilder::SecondaryALUOp }},
|
||||
{"", TYPE_INVALID, FLAGS_NONE, 0, { .OpDispatch = nullptr } },
|
||||
},
|
||||
{
|
||||
@@ -66,23 +66,23 @@ constexpr std::array<X86InstInfo, MAX_INST_GROUP_TABLE_SIZE> PrimaryInstGroupOps
|
||||
#define OPD(group, prefix, Reg) (((group - FEXCore::X86Tables::TYPE_GROUP_1) << 6) | (prefix) << 3 | (Reg))
|
||||
constexpr U16U8InfoStruct PrimaryGroupOpTable[] = {
|
||||
// GROUP_1 | 0x80 | reg
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 0), 1, X86InstInfo{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 1), 1, X86InstInfo{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 2), 1, X86InstInfo{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 3), 1, X86InstInfo{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 4), 1, X86InstInfo{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 5), 1, X86InstInfo{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 6), 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 7), 1, X86InstInfo{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 0), 1, X86InstInfo{"ADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 1), 1, X86InstInfo{"OR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 2), 1, X86InstInfo{"ADC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 3), 1, X86InstInfo{"SBB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 4), 1, X86InstInfo{"AND", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 5), 1, X86InstInfo{"SUB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 6), 1, X86InstInfo{"XOR", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x80), 7), 1, X86InstInfo{"CMP", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 0), 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 1), 1, X86InstInfo{"OR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 2), 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 3), 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 4), 1, X86InstInfo{"AND", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 5), 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 6), 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 7), 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 0), 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 1), 1, X86InstInfo{"OR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 2), 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 3), 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 4), 1, X86InstInfo{"AND", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 5), 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 6), 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 4}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x81), 7), 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2 | FLAGS_SUPPORTS_LOCK, 4}},
|
||||
|
||||
// Duplicates the 0x80 opcode group
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x82), 0), 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = PrimaryGroup_ArchSelect_LUT[ENTRY_1_82_0] }}},
|
||||
@@ -94,14 +94,14 @@ constexpr std::array<X86InstInfo, MAX_INST_GROUP_TABLE_SIZE> PrimaryInstGroupOps
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x82), 6), 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = PrimaryGroup_ArchSelect_LUT[ENTRY_1_82_6] }}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x82), 7), 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = PrimaryGroup_ArchSelect_LUT[ENTRY_1_82_7] }}},
|
||||
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 0), 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 1), 1, X86InstInfo{"OR", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 2), 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 3), 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 4), 1, X86InstInfo{"AND", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 5), 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 6), 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 7), 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 0), 1, X86InstInfo{"ADD", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 1), 1, X86InstInfo{"OR", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 2), 1, X86InstInfo{"ADC", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 3), 1, X86InstInfo{"SBB", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 4), 1, X86InstInfo{"AND", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 5), 1, X86InstInfo{"SUB", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 6), 1, X86InstInfo{"XOR", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
{OPD(TYPE_GROUP_1, OpToIndex(0x83), 7), 1, X86InstInfo{"CMP", TYPE_INST, FLAGS_SRC_SEXT | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
|
||||
// GROUP 2
|
||||
{OPD(TYPE_GROUP_2, OpToIndex(0xC0), 0), 1, X86InstInfo{"ROL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
@@ -161,8 +161,8 @@ constexpr std::array<X86InstInfo, MAX_INST_GROUP_TABLE_SIZE> PrimaryInstGroupOps
|
||||
// GROUP 3
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 0), 1, X86InstInfo{"TEST", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 1), 1, X86InstInfo{"TEST", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 2), 1, X86InstInfo{"NOT", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 3), 1, X86InstInfo{"NEG", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 2), 1, X86InstInfo{"NOT", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 3), 1, X86InstInfo{"NEG", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 4), 1, X86InstInfo{"MUL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 5), 1, X86InstInfo{"IMUL", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF6), 6), 1, X86InstInfo{"DIV", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
@@ -170,21 +170,21 @@ constexpr std::array<X86InstInfo, MAX_INST_GROUP_TABLE_SIZE> PrimaryInstGroupOps
|
||||
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 0), 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 1), 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SRC_SEXT64BIT | FLAGS_DISPLACE_SIZE_DIV_2, 4}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 2), 1, X86InstInfo{"NOT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 3), 1, X86InstInfo{"NEG", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 2), 1, X86InstInfo{"NOT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 3), 1, X86InstInfo{"NEG", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 4), 1, X86InstInfo{"MUL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 5), 1, X86InstInfo{"IMUL", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 6), 1, X86InstInfo{"DIV", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_3, OpToIndex(0xF7), 7), 1, X86InstInfo{"IDIV", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
|
||||
// GROUP 4
|
||||
{OPD(TYPE_GROUP_4, OpToIndex(0xFE), 0), 1, X86InstInfo{"INC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_4, OpToIndex(0xFE), 1), 1, X86InstInfo{"DEC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_4, OpToIndex(0xFE), 0), 1, X86InstInfo{"INC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{OPD(TYPE_GROUP_4, OpToIndex(0xFE), 1), 1, X86InstInfo{"DEC", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{OPD(TYPE_GROUP_4, OpToIndex(0xFE), 2), 6, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
|
||||
// GROUP 5
|
||||
{OPD(TYPE_GROUP_5, OpToIndex(0xFF), 0), 1, X86InstInfo{"INC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_5, OpToIndex(0xFF), 1), 1, X86InstInfo{"DEC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{OPD(TYPE_GROUP_5, OpToIndex(0xFF), 0), 1, X86InstInfo{"INC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{OPD(TYPE_GROUP_5, OpToIndex(0xFF), 1), 1, X86InstInfo{"DEC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{OPD(TYPE_GROUP_5, OpToIndex(0xFF), 2), 1, X86InstInfo{"CALL", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_MODRM | FLAGS_BLOCK_END | FLAGS_CALL , 0}},
|
||||
{OPD(TYPE_GROUP_5, OpToIndex(0xFF), 3), 1, X86InstInfo{"CALLF", TYPE_INST, FLAGS_SETS_RIP | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY | FLAGS_BLOCK_END, 0}},
|
||||
{OPD(TYPE_GROUP_5, OpToIndex(0xFF), 4), 1, X86InstInfo{"JMP", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_MODRM | FLAGS_BLOCK_END , 0}},
|
||||
|
||||
@@ -139,37 +139,37 @@ constexpr std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGr
|
||||
{OPD(TYPE_GROUP_8, PF_NONE, 1), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_8, PF_NONE, 2), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_8, PF_NONE, 3), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_8, PF_NONE, 4), 1, X86InstInfo{"BT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_8, PF_NONE, 5), 1, X86InstInfo{"BTS", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_8, PF_NONE, 6), 1, X86InstInfo{"BTR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_8, PF_NONE, 7), 1, X86InstInfo{"BTC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_8, PF_NONE, 4), 1, X86InstInfo{"BT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
{OPD(TYPE_GROUP_8, PF_NONE, 5), 1, X86InstInfo{"BTS", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
{OPD(TYPE_GROUP_8, PF_NONE, 6), 1, X86InstInfo{"BTR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
{OPD(TYPE_GROUP_8, PF_NONE, 7), 1, X86InstInfo{"BTC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
|
||||
{OPD(TYPE_GROUP_8, PF_F3, 0), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_8, PF_F3, 1), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_8, PF_F3, 2), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_8, PF_F3, 3), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_8, PF_F3, 4), 1, X86InstInfo{"BT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_8, PF_F3, 5), 1, X86InstInfo{"BTS", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_8, PF_F3, 6), 1, X86InstInfo{"BTR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_8, PF_F3, 7), 1, X86InstInfo{"BTC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_8, PF_F3, 4), 1, X86InstInfo{"BT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
{OPD(TYPE_GROUP_8, PF_F3, 5), 1, X86InstInfo{"BTS", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
{OPD(TYPE_GROUP_8, PF_F3, 6), 1, X86InstInfo{"BTR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
{OPD(TYPE_GROUP_8, PF_F3, 7), 1, X86InstInfo{"BTC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
|
||||
{OPD(TYPE_GROUP_8, PF_66, 0), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_8, PF_66, 1), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_8, PF_66, 2), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_8, PF_66, 3), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_8, PF_66, 4), 1, X86InstInfo{"BT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_8, PF_66, 5), 1, X86InstInfo{"BTS", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_8, PF_66, 6), 1, X86InstInfo{"BTR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_8, PF_66, 7), 1, X86InstInfo{"BTC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_8, PF_66, 4), 1, X86InstInfo{"BT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
{OPD(TYPE_GROUP_8, PF_66, 5), 1, X86InstInfo{"BTS", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
{OPD(TYPE_GROUP_8, PF_66, 6), 1, X86InstInfo{"BTR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
{OPD(TYPE_GROUP_8, PF_66, 7), 1, X86InstInfo{"BTC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
|
||||
{OPD(TYPE_GROUP_8, PF_F2, 0), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_8, PF_F2, 1), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_8, PF_F2, 2), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_8, PF_F2, 3), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_8, PF_F2, 4), 1, X86InstInfo{"BT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_8, PF_F2, 5), 1, X86InstInfo{"BTS", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_8, PF_F2, 6), 1, X86InstInfo{"BTR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_8, PF_F2, 7), 1, X86InstInfo{"BTC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 1}},
|
||||
{OPD(TYPE_GROUP_8, PF_F2, 4), 1, X86InstInfo{"BT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
{OPD(TYPE_GROUP_8, PF_F2, 5), 1, X86InstInfo{"BTS", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
{OPD(TYPE_GROUP_8, PF_F2, 6), 1, X86InstInfo{"BTR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
{OPD(TYPE_GROUP_8, PF_F2, 7), 1, X86InstInfo{"BTC", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 1}},
|
||||
|
||||
// GROUP 9
|
||||
|
||||
@@ -179,7 +179,7 @@ constexpr std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGr
|
||||
// CMPXCHG8B/16B works with all prefixes
|
||||
// Tooling fails to decode CMPXCHG with prefix
|
||||
{OPD(TYPE_GROUP_9, PF_NONE, 0), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_9, PF_NONE, 1), 1, X86InstInfo{"CMPXCHG8B/16B", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0}},
|
||||
{OPD(TYPE_GROUP_9, PF_NONE, 1), 1, X86InstInfo{"CMPXCHG8B/16B", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{OPD(TYPE_GROUP_9, PF_NONE, 2), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_9, PF_NONE, 3), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_9, PF_NONE, 4), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
@@ -188,7 +188,7 @@ constexpr std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGr
|
||||
{OPD(TYPE_GROUP_9, PF_NONE, 7), 1, X86InstInfo{"RDSEED", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_REG_ONLY, 0}},
|
||||
|
||||
{OPD(TYPE_GROUP_9, PF_F3, 0), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_9, PF_F3, 1), 1, X86InstInfo{"CMPXCHG8B/16B", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0}},
|
||||
{OPD(TYPE_GROUP_9, PF_F3, 1), 1, X86InstInfo{"CMPXCHG8B/16B", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{OPD(TYPE_GROUP_9, PF_F3, 2), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_9, PF_F3, 3), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_9, PF_F3, 4), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
@@ -197,7 +197,7 @@ constexpr std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGr
|
||||
{OPD(TYPE_GROUP_9, PF_F3, 7), 1, X86InstInfo{"RDPID", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_REG_ONLY, 0}},
|
||||
|
||||
{OPD(TYPE_GROUP_9, PF_66, 0), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_9, PF_66, 1), 1, X86InstInfo{"CMPXCHG8B/16B", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0}},
|
||||
{OPD(TYPE_GROUP_9, PF_66, 1), 1, X86InstInfo{"CMPXCHG8B/16B", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{OPD(TYPE_GROUP_9, PF_66, 2), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_9, PF_66, 3), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_9, PF_66, 4), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
@@ -206,7 +206,7 @@ constexpr std::array<X86InstInfo, MAX_INST_SECOND_GROUP_TABLE_SIZE> SecondInstGr
|
||||
{OPD(TYPE_GROUP_9, PF_66, 7), 1, X86InstInfo{"RDSEED", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_REG_ONLY, 0}},
|
||||
|
||||
{OPD(TYPE_GROUP_9, PF_F2, 0), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_9, PF_F2, 1), 1, X86InstInfo{"CMPXCHG8B/16B", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0}},
|
||||
{OPD(TYPE_GROUP_9, PF_F2, 1), 1, X86InstInfo{"CMPXCHG8B/16B", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{OPD(TYPE_GROUP_9, PF_F2, 2), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_9, PF_F2, 3), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{OPD(TYPE_GROUP_9, PF_F2, 4), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
|
||||
@@ -61,8 +61,8 @@ constexpr std::array<X86InstInfo, MAX_SECOND_TABLE_SIZE> SecondBaseOps = []() co
|
||||
{0x05, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_NONE, 0, { .Indirect = Secondary_ArchSelect_LUT[ENTRY_05] }}},
|
||||
{0x06, 1, X86InstInfo{"CLTS", TYPE_INST, FLAGS_NO_OVERLAY, 0}},
|
||||
{0x07, 1, X86InstInfo{"SYSRET", TYPE_INST, FLAGS_NO_OVERLAY, 0}},
|
||||
{0x08, 1, X86InstInfo{"INVD", TYPE_PRIV, FLAGS_NO_OVERLAY, 0}},
|
||||
{0x09, 1, X86InstInfo{"WBINVD", TYPE_PRIV, FLAGS_NO_OVERLAY, 0}},
|
||||
{0x08, 1, X86InstInfo{"INVD", TYPE_INST, FLAGS_NO_OVERLAY, 0}},
|
||||
{0x09, 1, X86InstInfo{"WBINVD", TYPE_INST, FLAGS_NO_OVERLAY, 0}},
|
||||
{0x0A, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0}},
|
||||
{0x0B, 1, X86InstInfo{"UD2", TYPE_INST, FLAGS_BLOCK_END | FLAGS_NO_OVERLAY, 0}},
|
||||
{0x0C, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0}},
|
||||
@@ -205,23 +205,23 @@ constexpr std::array<X86InstInfo, MAX_SECOND_TABLE_SIZE> SecondBaseOps = []() co
|
||||
{0xA0, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, { .Indirect = Secondary_ArchSelect_LUT[ENTRY_A0] }}},
|
||||
{0xA1, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, { .Indirect = Secondary_ArchSelect_LUT[ENTRY_A1] }}},
|
||||
{0xA2, 1, X86InstInfo{"CPUID", TYPE_INST, FLAGS_SF_SRC_RAX | FLAGS_NO_OVERLAY, 0}},
|
||||
{0xA3, 1, X86InstInfo{"BT", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0}},
|
||||
{0xA3, 1, X86InstInfo{"BT", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0xA4, 1, X86InstInfo{"SHLD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 1}},
|
||||
{0xA5, 1, X86InstInfo{"SHLD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX | FLAGS_NO_OVERLAY, 0}},
|
||||
{0xA6, 2, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0}},
|
||||
{0xA8, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, { .Indirect = Secondary_ArchSelect_LUT[ENTRY_A8] }}},
|
||||
{0xA9, 1, X86InstInfo{"", TYPE_ARCH_DISPATCHER, FLAGS_DEBUG_MEM_ACCESS | FLAGS_NO_OVERLAY, 0, { .Indirect = Secondary_ArchSelect_LUT[ENTRY_A9] }}},
|
||||
{0xAA, 1, X86InstInfo{"RSM", TYPE_PRIV, FLAGS_NO_OVERLAY, 0}},
|
||||
{0xAB, 1, X86InstInfo{"BTS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0}},
|
||||
{0xAB, 1, X86InstInfo{"BTS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0xAC, 1, X86InstInfo{"SHRD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 1}},
|
||||
{0xAD, 1, X86InstInfo{"SHRD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_SRC_RCX | FLAGS_NO_OVERLAY, 0}},
|
||||
{0xAE, 1, X86InstInfo{"", TYPE_GROUP_15, FLAGS_NO_OVERLAY, 0}},
|
||||
{0xAF, 1, X86InstInfo{"IMUL", TYPE_INST, FLAGS_MODRM | FLAGS_NO_OVERLAY, 0}},
|
||||
|
||||
{0xB0, 1, X86InstInfo{"CMPXCHG", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0}},
|
||||
{0xB1, 1, X86InstInfo{"CMPXCHG", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0}},
|
||||
{0xB0, 1, X86InstInfo{"CMPXCHG", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0xB1, 1, X86InstInfo{"CMPXCHG", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0xB2, 1, X86InstInfo{"LSS", TYPE_INVALID, FLAGS_NO_OVERLAY, 0}},
|
||||
{0xB3, 1, X86InstInfo{"BTR", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0}},
|
||||
{0xB3, 1, X86InstInfo{"BTR", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0xB4, 1, X86InstInfo{"LFS", TYPE_INVALID, FLAGS_NO_OVERLAY, 0}},
|
||||
{0xB5, 1, X86InstInfo{"LGS", TYPE_INVALID, FLAGS_NO_OVERLAY, 0}},
|
||||
{0xB6, 1, X86InstInfo{"MOVZX", TYPE_INST, GenFlagsSrcSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_NO_OVERLAY, 0}},
|
||||
@@ -229,14 +229,14 @@ constexpr std::array<X86InstInfo, MAX_SECOND_TABLE_SIZE> SecondBaseOps = []() co
|
||||
{0xB8, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{0xB9, 1, X86InstInfo{"", TYPE_GROUP_10, FLAGS_NO_OVERLAY, 0}},
|
||||
{0xBA, 1, X86InstInfo{"", TYPE_GROUP_8, FLAGS_NO_OVERLAY, 0}},
|
||||
{0xBB, 1, X86InstInfo{"BTC", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0}},
|
||||
{0xBB, 1, X86InstInfo{"BTC", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0xBC, 1, X86InstInfo{"BSF", TYPE_INST, FLAGS_MODRM | FLAGS_NO_OVERLAY66, 0}},
|
||||
{0xBD, 1, X86InstInfo{"BSR", TYPE_INST, FLAGS_MODRM | FLAGS_NO_OVERLAY66, 0}},
|
||||
{0xBE, 1, X86InstInfo{"MOVSX", TYPE_INST, GenFlagsSrcSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_NO_OVERLAY, 0}},
|
||||
{0xBF, 1, X86InstInfo{"MOVSX", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_MODRM | FLAGS_NO_OVERLAY, 0}},
|
||||
|
||||
{0xC0, 1, X86InstInfo{"XADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST, 0}},
|
||||
{0xC1, 1, X86InstInfo{"XADD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY, 0}},
|
||||
{0xC0, 1, X86InstInfo{"XADD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0xC1, 1, X86InstInfo{"XADD", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_NO_OVERLAY | FLAGS_SUPPORTS_LOCK, 0}},
|
||||
{0xC2, 1, X86InstInfo{"CMPPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1}},
|
||||
{0xC3, 1, X86InstInfo{"MOVNTI", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SF_MOD_DST, 0}},
|
||||
{0xC4, 1, X86InstInfo{"PINSRW", TYPE_INST, GenFlagsSizes(SIZE_64BIT, SIZE_16BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS | FLAGS_SF_MMX | FLAGS_SF_SRC_GPR, 1}},
|
||||
|
||||
@@ -474,7 +474,7 @@ namespace AVX256 {
|
||||
|
||||
{OPD(1, 0b00, 0x12), 1, &OpDispatchBuilder::VMOVLPOp},
|
||||
{OPD(1, 0b01, 0x12), 1, &OpDispatchBuilder::VMOVLPOp},
|
||||
{OPD(1, 0b10, 0x12), 1, &OpDispatchBuilder::VMOVSLDUPOp},
|
||||
{OPD(1, 0b10, 0x12), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VMOVSLDUPOp, true>},
|
||||
{OPD(1, 0b11, 0x12), 1, &OpDispatchBuilder::VMOVDDUPOp},
|
||||
{OPD(1, 0b00, 0x13), 1, &OpDispatchBuilder::VMOVLPOp},
|
||||
{OPD(1, 0b01, 0x13), 1, &OpDispatchBuilder::VMOVLPOp},
|
||||
@@ -487,7 +487,7 @@ namespace AVX256 {
|
||||
|
||||
{OPD(1, 0b00, 0x16), 1, &OpDispatchBuilder::VMOVHPOp},
|
||||
{OPD(1, 0b01, 0x16), 1, &OpDispatchBuilder::VMOVHPOp},
|
||||
{OPD(1, 0b10, 0x16), 1, &OpDispatchBuilder::VMOVSHDUPOp},
|
||||
{OPD(1, 0b10, 0x16), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VMOVSHDUPOp, true>},
|
||||
{OPD(1, 0b00, 0x17), 1, &OpDispatchBuilder::VMOVHPOp},
|
||||
{OPD(1, 0b01, 0x17), 1, &OpDispatchBuilder::VMOVHPOp},
|
||||
|
||||
@@ -496,36 +496,36 @@ namespace AVX256 {
|
||||
{OPD(1, 0b00, 0x29), 1, &OpDispatchBuilder::VMOVAPS_VMOVAPDOp},
|
||||
{OPD(1, 0b01, 0x29), 1, &OpDispatchBuilder::VMOVAPS_VMOVAPDOp},
|
||||
|
||||
{OPD(1, 0b10, 0x2A), 1, &OpDispatchBuilder::AVXInsertCVTGPR_To_FPR<OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x2A), 1, &OpDispatchBuilder::AVXInsertCVTGPR_To_FPR<OpSize::i64Bit>},
|
||||
{OPD(1, 0b10, 0x2A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXInsertCVTGPR_To_FPR, OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x2A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXInsertCVTGPR_To_FPR, OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x2B), 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{OPD(1, 0b01, 0x2B), 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{OPD(1, 0b00, 0x2B), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVVectorNTOp, true>},
|
||||
{OPD(1, 0b01, 0x2B), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVVectorNTOp, true>},
|
||||
|
||||
{OPD(1, 0b10, 0x2C), 1, &OpDispatchBuilder::CVTFPR_To_GPR<OpSize::i32Bit, false>},
|
||||
{OPD(1, 0b11, 0x2C), 1, &OpDispatchBuilder::CVTFPR_To_GPR<OpSize::i64Bit, false>},
|
||||
{OPD(1, 0b10, 0x2C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CVTFPR_To_GPR, OpSize::i32Bit, false>},
|
||||
{OPD(1, 0b11, 0x2C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CVTFPR_To_GPR, OpSize::i64Bit, false>},
|
||||
|
||||
{OPD(1, 0b10, 0x2D), 1, &OpDispatchBuilder::CVTFPR_To_GPR<OpSize::i32Bit, true>},
|
||||
{OPD(1, 0b11, 0x2D), 1, &OpDispatchBuilder::CVTFPR_To_GPR<OpSize::i64Bit, true>},
|
||||
{OPD(1, 0b10, 0x2D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CVTFPR_To_GPR, OpSize::i32Bit, true>},
|
||||
{OPD(1, 0b11, 0x2D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::CVTFPR_To_GPR, OpSize::i64Bit, true>},
|
||||
|
||||
{OPD(1, 0b00, 0x2E), 1, &OpDispatchBuilder::UCOMISxOp<OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x2E), 1, &OpDispatchBuilder::UCOMISxOp<OpSize::i64Bit>},
|
||||
{OPD(1, 0b00, 0x2F), 1, &OpDispatchBuilder::UCOMISxOp<OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x2F), 1, &OpDispatchBuilder::UCOMISxOp<OpSize::i64Bit>},
|
||||
{OPD(1, 0b00, 0x2E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::UCOMISxOp, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x2E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::UCOMISxOp, OpSize::i64Bit>},
|
||||
{OPD(1, 0b00, 0x2F), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::UCOMISxOp, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x2F), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::UCOMISxOp, OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x50), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVMSKOp, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x50), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVMSKOp, OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x51), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorUnaryOp, IR::OP_VFSQRT, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x51), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorUnaryOp, IR::OP_VFSQRT, OpSize::i64Bit>},
|
||||
{OPD(1, 0b10, 0x51), 1, &OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x51), 1, &OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, OpSize::i64Bit>},
|
||||
{OPD(1, 0b10, 0x51), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp, IR::OP_VFSQRTSCALARINSERT, OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x51), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp, IR::OP_VFSQRTSCALARINSERT, OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x52), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorUnaryOp, IR::OP_VFRSQRT, OpSize::i32Bit>},
|
||||
{OPD(1, 0b10, 0x52), 1, &OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp<IR::OP_VFRSQRTSCALARINSERT, OpSize::i32Bit>},
|
||||
{OPD(1, 0b10, 0x52), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp, IR::OP_VFRSQRTSCALARINSERT, OpSize::i32Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x53), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorUnaryOp, IR::OP_VFRECP, OpSize::i32Bit>},
|
||||
{OPD(1, 0b10, 0x53), 1, &OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp<IR::OP_VFRECPSCALARINSERT, OpSize::i32Bit>},
|
||||
{OPD(1, 0b10, 0x53), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp, IR::OP_VFRECPSCALARINSERT, OpSize::i32Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x54), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VAND, OpSize::i128Bit>},
|
||||
{OPD(1, 0b01, 0x54), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VAND, OpSize::i128Bit>},
|
||||
@@ -541,42 +541,42 @@ namespace AVX256 {
|
||||
|
||||
{OPD(1, 0b00, 0x58), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VFADD, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x58), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VFADD, OpSize::i64Bit>},
|
||||
{OPD(1, 0b10, 0x58), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x58), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, OpSize::i64Bit>},
|
||||
{OPD(1, 0b10, 0x58), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarInsertALUOp, IR::OP_VFADDSCALARINSERT, OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x58), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarInsertALUOp, IR::OP_VFADDSCALARINSERT, OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x59), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VFMUL, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x59), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VFMUL, OpSize::i64Bit>},
|
||||
{OPD(1, 0b10, 0x59), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x59), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, OpSize::i64Bit>},
|
||||
{OPD(1, 0b10, 0x59), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarInsertALUOp, IR::OP_VFMULSCALARINSERT, OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x59), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarInsertALUOp, IR::OP_VFMULSCALARINSERT, OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x5A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Float, OpSize::i64Bit, OpSize::i32Bit, true>},
|
||||
{OPD(1, 0b01, 0x5A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Float, OpSize::i32Bit, OpSize::i64Bit, true>},
|
||||
{OPD(1, 0b10, 0x5A), 1, &OpDispatchBuilder::AVXInsertScalar_CVT_Float_To_Float<OpSize::i64Bit, OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x5A), 1, &OpDispatchBuilder::AVXInsertScalar_CVT_Float_To_Float<OpSize::i32Bit, OpSize::i64Bit>},
|
||||
{OPD(1, 0b10, 0x5A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXInsertScalar_CVT_Float_To_Float, OpSize::i64Bit, OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x5A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXInsertScalar_CVT_Float_To_Float, OpSize::i32Bit, OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x5B), 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, false>},
|
||||
{OPD(1, 0b01, 0x5B), 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, true>},
|
||||
{OPD(1, 0b10, 0x5B), 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false>},
|
||||
{OPD(1, 0b00, 0x5B), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Int_To_Float, OpSize::i32Bit, false, true>},
|
||||
{OPD(1, 0b01, 0x5B), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Int, OpSize::i32Bit, true, true>},
|
||||
{OPD(1, 0b10, 0x5B), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Int, OpSize::i32Bit, false, true>},
|
||||
|
||||
{OPD(1, 0b00, 0x5C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VFSUB, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x5C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VFSUB, OpSize::i64Bit>},
|
||||
{OPD(1, 0b10, 0x5C), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x5C), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, OpSize::i64Bit>},
|
||||
{OPD(1, 0b10, 0x5C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarInsertALUOp, IR::OP_VFSUBSCALARINSERT, OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x5C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarInsertALUOp, IR::OP_VFSUBSCALARINSERT, OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x5D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VFMIN, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x5D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VFMIN, OpSize::i64Bit>},
|
||||
{OPD(1, 0b10, 0x5D), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x5D), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, OpSize::i64Bit>},
|
||||
{OPD(1, 0b10, 0x5D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarInsertALUOp, IR::OP_VFMINSCALARINSERT, OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x5D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarInsertALUOp, IR::OP_VFMINSCALARINSERT, OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x5E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VFDIV, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x5E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VFDIV, OpSize::i64Bit>},
|
||||
{OPD(1, 0b10, 0x5E), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x5E), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, OpSize::i64Bit>},
|
||||
{OPD(1, 0b10, 0x5E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarInsertALUOp, IR::OP_VFDIVSCALARINSERT, OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x5E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarInsertALUOp, IR::OP_VFDIVSCALARINSERT, OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0x5F), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VFMAX, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x5F), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VFMAX, OpSize::i64Bit>},
|
||||
{OPD(1, 0b10, 0x5F), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x5F), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, OpSize::i64Bit>},
|
||||
{OPD(1, 0b10, 0x5F), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarInsertALUOp, IR::OP_VFMAXSCALARINSERT, OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0x5F), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorScalarInsertALUOp, IR::OP_VFMAXSCALARINSERT, OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b01, 0x60), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPUNPCKLOp, OpSize::i8Bit>},
|
||||
{OPD(1, 0b01, 0x61), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPUNPCKLOp, OpSize::i16Bit>},
|
||||
@@ -607,8 +607,8 @@ namespace AVX256 {
|
||||
|
||||
{OPD(1, 0b00, 0x77), 1, &OpDispatchBuilder::VZEROOp},
|
||||
|
||||
{OPD(1, 0b01, 0x7C), 1, &OpDispatchBuilder::VHADDPOp<IR::OP_VFADDP, OpSize::i64Bit>},
|
||||
{OPD(1, 0b11, 0x7C), 1, &OpDispatchBuilder::VHADDPOp<IR::OP_VFADDP, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x7C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VHADDPOp, IR::OP_VFADDP, OpSize::i64Bit>},
|
||||
{OPD(1, 0b11, 0x7C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VHADDPOp, IR::OP_VFADDP, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0x7D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VHSUBPOp, OpSize::i64Bit>},
|
||||
{OPD(1, 0b11, 0x7D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VHSUBPOp, OpSize::i32Bit>},
|
||||
|
||||
@@ -618,19 +618,19 @@ namespace AVX256 {
|
||||
{OPD(1, 0b01, 0x7F), 1, &OpDispatchBuilder::VMOVAPS_VMOVAPDOp},
|
||||
{OPD(1, 0b10, 0x7F), 1, &OpDispatchBuilder::VMOVUPS_VMOVUPDOp},
|
||||
|
||||
{OPD(1, 0b00, 0xC2), 1, &OpDispatchBuilder::AVXVFCMPOp<OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0xC2), 1, &OpDispatchBuilder::AVXVFCMPOp<OpSize::i64Bit>},
|
||||
{OPD(1, 0b10, 0xC2), 1, &OpDispatchBuilder::AVXInsertScalarFCMPOp<OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0xC2), 1, &OpDispatchBuilder::AVXInsertScalarFCMPOp<OpSize::i64Bit>},
|
||||
{OPD(1, 0b00, 0xC2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVFCMPOp, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0xC2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVFCMPOp, OpSize::i64Bit>},
|
||||
{OPD(1, 0b10, 0xC2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXInsertScalarFCMPOp, OpSize::i32Bit>},
|
||||
{OPD(1, 0b11, 0xC2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXInsertScalarFCMPOp, OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b01, 0xC4), 1, &OpDispatchBuilder::VPINSRWOp},
|
||||
{OPD(1, 0b01, 0xC4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPINSRBWOp, OpSize::i16Bit>},
|
||||
{OPD(1, 0b01, 0xC5), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PExtrOp, OpSize::i16Bit>},
|
||||
|
||||
{OPD(1, 0b00, 0xC6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VSHUFOp, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0xC6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VSHUFOp, OpSize::i64Bit>},
|
||||
|
||||
{OPD(1, 0b01, 0xD0), 1, &OpDispatchBuilder::VADDSUBPOp<OpSize::i64Bit>},
|
||||
{OPD(1, 0b11, 0xD0), 1, &OpDispatchBuilder::VADDSUBPOp<OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0xD0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VADDSUBPOp, OpSize::i64Bit>},
|
||||
{OPD(1, 0b11, 0xD0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VADDSUBPOp, OpSize::i32Bit>},
|
||||
|
||||
{OPD(1, 0b01, 0xD1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPSRLDOp, OpSize::i16Bit>},
|
||||
{OPD(1, 0b01, 0xD2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPSRLDOp, OpSize::i32Bit>},
|
||||
@@ -653,14 +653,14 @@ namespace AVX256 {
|
||||
{OPD(1, 0b01, 0xE1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPSRAOp, OpSize::i16Bit>},
|
||||
{OPD(1, 0b01, 0xE2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPSRAOp, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0xE3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VURAVG, OpSize::i16Bit>},
|
||||
{OPD(1, 0b01, 0xE4), 1, &OpDispatchBuilder::VPMULHWOp<false>},
|
||||
{OPD(1, 0b01, 0xE5), 1, &OpDispatchBuilder::VPMULHWOp<true>},
|
||||
{OPD(1, 0b01, 0xE4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPMULHWOp, false>},
|
||||
{OPD(1, 0b01, 0xE5), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPMULHWOp, true>},
|
||||
|
||||
{OPD(1, 0b01, 0xE6), 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, false>},
|
||||
{OPD(1, 0b10, 0xE6), 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, true>},
|
||||
{OPD(1, 0b11, 0xE6), 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, true>},
|
||||
{OPD(1, 0b01, 0xE6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Int, OpSize::i64Bit, false, true>},
|
||||
{OPD(1, 0b10, 0xE6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Int_To_Float, OpSize::i32Bit, true, true>},
|
||||
{OPD(1, 0b11, 0xE6), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Vector_CVT_Float_To_Int, OpSize::i64Bit, true, true>},
|
||||
|
||||
{OPD(1, 0b01, 0xE7), 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{OPD(1, 0b01, 0xE7), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVVectorNTOp, true>},
|
||||
|
||||
{OPD(1, 0b01, 0xE8), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VSQSUB, OpSize::i8Bit>},
|
||||
{OPD(1, 0b01, 0xE9), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VSQSUB, OpSize::i16Bit>},
|
||||
@@ -671,11 +671,11 @@ namespace AVX256 {
|
||||
{OPD(1, 0b01, 0xEE), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VSMAX, OpSize::i16Bit>},
|
||||
{OPD(1, 0b01, 0xEF), 1, &OpDispatchBuilder::AVXVectorXOROp},
|
||||
|
||||
{OPD(1, 0b11, 0xF0), 1, &OpDispatchBuilder::MOVVectorUnalignedOp},
|
||||
{OPD(1, 0b11, 0xF0), 1, &OpDispatchBuilder::VMOVUPS_VMOVUPDOp},
|
||||
{OPD(1, 0b01, 0xF1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPSLLOp, OpSize::i16Bit>},
|
||||
{OPD(1, 0b01, 0xF2), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPSLLOp, OpSize::i32Bit>},
|
||||
{OPD(1, 0b01, 0xF3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPSLLOp, OpSize::i64Bit>},
|
||||
{OPD(1, 0b01, 0xF4), 1, &OpDispatchBuilder::VPMULLOp<OpSize::i32Bit, false>},
|
||||
{OPD(1, 0b01, 0xF4), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPMULLOp, OpSize::i32Bit, false>},
|
||||
{OPD(1, 0b01, 0xF5), 1, &OpDispatchBuilder::VPMADDWDOp},
|
||||
{OPD(1, 0b01, 0xF6), 1, &OpDispatchBuilder::VPSADBWOp},
|
||||
{OPD(1, 0b01, 0xF7), 1, &OpDispatchBuilder::MASKMOVOp},
|
||||
@@ -689,8 +689,8 @@ namespace AVX256 {
|
||||
{OPD(1, 0b01, 0xFE), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VADD, OpSize::i32Bit>},
|
||||
|
||||
{OPD(2, 0b01, 0x00), 1, &OpDispatchBuilder::VPSHUFBOp},
|
||||
{OPD(2, 0b01, 0x01), 1, &OpDispatchBuilder::VHADDPOp<IR::OP_VADDP, OpSize::i16Bit>},
|
||||
{OPD(2, 0b01, 0x02), 1, &OpDispatchBuilder::VHADDPOp<IR::OP_VADDP, OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x01), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VHADDPOp, IR::OP_VADDP, OpSize::i16Bit>},
|
||||
{OPD(2, 0b01, 0x02), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VHADDPOp, IR::OP_VADDP, OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x03), 1, &OpDispatchBuilder::VPHADDSWOp},
|
||||
{OPD(2, 0b01, 0x04), 1, &OpDispatchBuilder::VPMADDUBSWOp},
|
||||
|
||||
@@ -698,14 +698,14 @@ namespace AVX256 {
|
||||
{OPD(2, 0b01, 0x06), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPHSUBOp, OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x07), 1, &OpDispatchBuilder::VPHSUBSWOp},
|
||||
|
||||
{OPD(2, 0b01, 0x08), 1, &OpDispatchBuilder::VPSIGN<OpSize::i8Bit>},
|
||||
{OPD(2, 0b01, 0x09), 1, &OpDispatchBuilder::VPSIGN<OpSize::i16Bit>},
|
||||
{OPD(2, 0b01, 0x0A), 1, &OpDispatchBuilder::VPSIGN<OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x08), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPSIGN, OpSize::i8Bit>},
|
||||
{OPD(2, 0b01, 0x09), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPSIGN, OpSize::i16Bit>},
|
||||
{OPD(2, 0b01, 0x0A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPSIGN, OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x0B), 1, &OpDispatchBuilder::VPMULHRSWOp},
|
||||
{OPD(2, 0b01, 0x0C), 1, &OpDispatchBuilder::VPERMILRegOp<OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x0D), 1, &OpDispatchBuilder::VPERMILRegOp<OpSize::i64Bit>},
|
||||
{OPD(2, 0b01, 0x0E), 1, &OpDispatchBuilder::VTESTPOp<OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x0F), 1, &OpDispatchBuilder::VTESTPOp<OpSize::i64Bit>},
|
||||
{OPD(2, 0b01, 0x0C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPERMILRegOp, OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x0D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPERMILRegOp, OpSize::i64Bit>},
|
||||
{OPD(2, 0b01, 0x0E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VTESTPOp, OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x0F), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VTESTPOp, OpSize::i64Bit>},
|
||||
|
||||
{OPD(2, 0b01, 0x13), 1, &OpDispatchBuilder::VCVTPH2PSOp},
|
||||
{OPD(2, 0b01, 0x16), 1, &OpDispatchBuilder::VPERMDOp},
|
||||
@@ -717,28 +717,28 @@ namespace AVX256 {
|
||||
{OPD(2, 0b01, 0x1D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorUnaryOp, IR::OP_VABS, OpSize::i16Bit>},
|
||||
{OPD(2, 0b01, 0x1E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorUnaryOp, IR::OP_VABS, OpSize::i32Bit>},
|
||||
|
||||
{OPD(2, 0b01, 0x20), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i16Bit, true>},
|
||||
{OPD(2, 0b01, 0x21), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i32Bit, true>},
|
||||
{OPD(2, 0b01, 0x22), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i64Bit, true>},
|
||||
{OPD(2, 0b01, 0x23), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i16Bit, OpSize::i32Bit, true>},
|
||||
{OPD(2, 0b01, 0x24), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i16Bit, OpSize::i64Bit, true>},
|
||||
{OPD(2, 0b01, 0x25), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i32Bit, OpSize::i64Bit, true>},
|
||||
{OPD(2, 0b01, 0x20), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXExtendVectorElements, OpSize::i8Bit, OpSize::i16Bit, true>},
|
||||
{OPD(2, 0b01, 0x21), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXExtendVectorElements, OpSize::i8Bit, OpSize::i32Bit, true>},
|
||||
{OPD(2, 0b01, 0x22), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXExtendVectorElements, OpSize::i8Bit, OpSize::i64Bit, true>},
|
||||
{OPD(2, 0b01, 0x23), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXExtendVectorElements, OpSize::i16Bit, OpSize::i32Bit, true>},
|
||||
{OPD(2, 0b01, 0x24), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXExtendVectorElements, OpSize::i16Bit, OpSize::i64Bit, true>},
|
||||
{OPD(2, 0b01, 0x25), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXExtendVectorElements, OpSize::i32Bit, OpSize::i64Bit, true>},
|
||||
|
||||
{OPD(2, 0b01, 0x28), 1, &OpDispatchBuilder::VPMULLOp<OpSize::i32Bit, true>},
|
||||
{OPD(2, 0b01, 0x28), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPMULLOp, OpSize::i32Bit, true>},
|
||||
{OPD(2, 0b01, 0x29), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VCMPEQ, OpSize::i64Bit>},
|
||||
{OPD(2, 0b01, 0x2A), 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{OPD(2, 0b01, 0x2A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVVectorNTOp, true>},
|
||||
{OPD(2, 0b01, 0x2B), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPACKUSOp, OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x2C), 1, &OpDispatchBuilder::VMASKMOVOp<OpSize::i32Bit, false>},
|
||||
{OPD(2, 0b01, 0x2D), 1, &OpDispatchBuilder::VMASKMOVOp<OpSize::i64Bit, false>},
|
||||
{OPD(2, 0b01, 0x2E), 1, &OpDispatchBuilder::VMASKMOVOp<OpSize::i32Bit, true>},
|
||||
{OPD(2, 0b01, 0x2F), 1, &OpDispatchBuilder::VMASKMOVOp<OpSize::i64Bit, true>},
|
||||
{OPD(2, 0b01, 0x2C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VMASKMOVOp, OpSize::i32Bit, false>},
|
||||
{OPD(2, 0b01, 0x2D), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VMASKMOVOp, OpSize::i64Bit, false>},
|
||||
{OPD(2, 0b01, 0x2E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VMASKMOVOp, OpSize::i32Bit, true>},
|
||||
{OPD(2, 0b01, 0x2F), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VMASKMOVOp, OpSize::i64Bit, true>},
|
||||
|
||||
{OPD(2, 0b01, 0x30), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i16Bit, false>},
|
||||
{OPD(2, 0b01, 0x31), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i32Bit, false>},
|
||||
{OPD(2, 0b01, 0x32), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i8Bit, OpSize::i64Bit, false>},
|
||||
{OPD(2, 0b01, 0x33), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i16Bit, OpSize::i32Bit, false>},
|
||||
{OPD(2, 0b01, 0x34), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i16Bit, OpSize::i64Bit, false>},
|
||||
{OPD(2, 0b01, 0x35), 1, &OpDispatchBuilder::ExtendVectorElements<OpSize::i32Bit, OpSize::i64Bit, false>},
|
||||
{OPD(2, 0b01, 0x30), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXExtendVectorElements, OpSize::i8Bit, OpSize::i16Bit, false>},
|
||||
{OPD(2, 0b01, 0x31), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXExtendVectorElements, OpSize::i8Bit, OpSize::i32Bit, false>},
|
||||
{OPD(2, 0b01, 0x32), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXExtendVectorElements, OpSize::i8Bit, OpSize::i64Bit, false>},
|
||||
{OPD(2, 0b01, 0x33), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXExtendVectorElements, OpSize::i16Bit, OpSize::i32Bit, false>},
|
||||
{OPD(2, 0b01, 0x34), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXExtendVectorElements, OpSize::i16Bit, OpSize::i64Bit, false>},
|
||||
{OPD(2, 0b01, 0x35), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXExtendVectorElements, OpSize::i32Bit, OpSize::i64Bit, false>},
|
||||
{OPD(2, 0b01, 0x36), 1, &OpDispatchBuilder::VPERMDOp},
|
||||
|
||||
{OPD(2, 0b01, 0x37), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VCMPGT, OpSize::i64Bit>},
|
||||
@@ -752,7 +752,7 @@ namespace AVX256 {
|
||||
{OPD(2, 0b01, 0x3F), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VUMAX, OpSize::i32Bit>},
|
||||
|
||||
{OPD(2, 0b01, 0x40), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorALUOp, IR::OP_VMUL, OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x41), 1, &OpDispatchBuilder::PHMINPOSUWOp},
|
||||
{OPD(2, 0b01, 0x41), 1, &OpDispatchBuilder::AVXPHMINPOSUWOp},
|
||||
{OPD(2, 0b01, 0x45), 1, &OpDispatchBuilder::VPSRLVOp},
|
||||
{OPD(2, 0b01, 0x46), 1, &OpDispatchBuilder::VPSRAVDOp},
|
||||
{OPD(2, 0b01, 0x47), 1, &OpDispatchBuilder::VPSLLVOp},
|
||||
@@ -764,13 +764,13 @@ namespace AVX256 {
|
||||
{OPD(2, 0b01, 0x78), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VBROADCASTOp, OpSize::i8Bit>},
|
||||
{OPD(2, 0b01, 0x79), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VBROADCASTOp, OpSize::i16Bit>},
|
||||
|
||||
{OPD(2, 0b01, 0x8C), 1, &OpDispatchBuilder::VPMASKMOVOp<false>},
|
||||
{OPD(2, 0b01, 0x8E), 1, &OpDispatchBuilder::VPMASKMOVOp<true>},
|
||||
{OPD(2, 0b01, 0x8C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPMASKMOVOp, false>},
|
||||
{OPD(2, 0b01, 0x8E), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPMASKMOVOp, true>},
|
||||
|
||||
{OPD(2, 0b01, 0x90), 1, &OpDispatchBuilder::VPGATHER<OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x91), 1, &OpDispatchBuilder::VPGATHER<OpSize::i64Bit>},
|
||||
{OPD(2, 0b01, 0x92), 1, &OpDispatchBuilder::VPGATHER<OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x93), 1, &OpDispatchBuilder::VPGATHER<OpSize::i64Bit>},
|
||||
{OPD(2, 0b01, 0x90), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPGATHER, OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x91), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPGATHER, OpSize::i64Bit>},
|
||||
{OPD(2, 0b01, 0x92), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPGATHER, OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x93), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPGATHER, OpSize::i64Bit>},
|
||||
|
||||
{OPD(2, 0b01, 0x96), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VFMAddSubImpl, true, 1, 3, 2>}, // VFMADDSUB
|
||||
{OPD(2, 0b01, 0x97), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VFMAddSubImpl, false, 1, 3, 2>}, // VFMSUBADD
|
||||
@@ -820,10 +820,10 @@ namespace AVX256 {
|
||||
{OPD(3, 0b01, 0x04), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPERMILImmOp, OpSize::i32Bit>},
|
||||
{OPD(3, 0b01, 0x05), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPERMILImmOp, OpSize::i64Bit>},
|
||||
{OPD(3, 0b01, 0x06), 1, &OpDispatchBuilder::VPERM2Op},
|
||||
{OPD(3, 0b01, 0x08), 1, &OpDispatchBuilder::AVXVectorRound<OpSize::i32Bit>},
|
||||
{OPD(3, 0b01, 0x09), 1, &OpDispatchBuilder::AVXVectorRound<OpSize::i64Bit>},
|
||||
{OPD(3, 0b01, 0x0A), 1, &OpDispatchBuilder::AVXInsertScalarRound<OpSize::i32Bit>},
|
||||
{OPD(3, 0b01, 0x0B), 1, &OpDispatchBuilder::AVXInsertScalarRound<OpSize::i64Bit>},
|
||||
{OPD(3, 0b01, 0x08), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorRound, OpSize::i32Bit>},
|
||||
{OPD(3, 0b01, 0x09), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorRound, OpSize::i64Bit>},
|
||||
{OPD(3, 0b01, 0x0A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXInsertScalarRound, OpSize::i32Bit>},
|
||||
{OPD(3, 0b01, 0x0B), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXInsertScalarRound, OpSize::i64Bit>},
|
||||
{OPD(3, 0b01, 0x0C), 1, &OpDispatchBuilder::VPBLENDDOp},
|
||||
{OPD(3, 0b01, 0x0D), 1, &OpDispatchBuilder::VBLENDPDOp},
|
||||
{OPD(3, 0b01, 0x0E), 1, &OpDispatchBuilder::VPBLENDWOp},
|
||||
@@ -837,15 +837,15 @@ namespace AVX256 {
|
||||
{OPD(3, 0b01, 0x18), 1, &OpDispatchBuilder::VINSERTOp},
|
||||
{OPD(3, 0b01, 0x19), 1, &OpDispatchBuilder::VEXTRACT128Op},
|
||||
{OPD(3, 0b01, 0x1D), 1, &OpDispatchBuilder::VCVTPS2PHOp},
|
||||
{OPD(3, 0b01, 0x20), 1, &OpDispatchBuilder::VPINSRBOp},
|
||||
{OPD(3, 0b01, 0x20), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPINSRBWOp, OpSize::i8Bit>},
|
||||
{OPD(3, 0b01, 0x21), 1, &OpDispatchBuilder::VINSERTPSOp},
|
||||
{OPD(3, 0b01, 0x22), 1, &OpDispatchBuilder::VPINSRDQOp},
|
||||
|
||||
{OPD(3, 0b01, 0x38), 1, &OpDispatchBuilder::VINSERTOp},
|
||||
{OPD(3, 0b01, 0x39), 1, &OpDispatchBuilder::VEXTRACT128Op},
|
||||
|
||||
{OPD(3, 0b01, 0x40), 1, &OpDispatchBuilder::VDPPOp<OpSize::i32Bit>},
|
||||
{OPD(3, 0b01, 0x41), 1, &OpDispatchBuilder::VDPPOp<OpSize::i64Bit>},
|
||||
{OPD(3, 0b01, 0x40), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VDPPOp, OpSize::i32Bit>},
|
||||
{OPD(3, 0b01, 0x41), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VDPPOp, OpSize::i64Bit>},
|
||||
{OPD(3, 0b01, 0x42), 1, &OpDispatchBuilder::VMPSADBWOp},
|
||||
{OPD(3, 0b01, 0x44), 1, &OpDispatchBuilder::VPCLMULQDQOp},
|
||||
|
||||
@@ -855,10 +855,10 @@ namespace AVX256 {
|
||||
{OPD(3, 0b01, 0x4B), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorVariableBlend, OpSize::i64Bit>},
|
||||
{OPD(3, 0b01, 0x4C), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVXVectorVariableBlend, OpSize::i8Bit>},
|
||||
|
||||
{OPD(3, 0b01, 0x60), 1, &OpDispatchBuilder::VPCMPESTRMOp},
|
||||
{OPD(3, 0b01, 0x61), 1, &OpDispatchBuilder::VPCMPESTRIOp},
|
||||
{OPD(3, 0b01, 0x62), 1, &OpDispatchBuilder::VPCMPISTRMOp},
|
||||
{OPD(3, 0b01, 0x63), 1, &OpDispatchBuilder::VPCMPISTRIOp},
|
||||
{OPD(3, 0b01, 0x60), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPCMPESTRMOp, true>},
|
||||
{OPD(3, 0b01, 0x61), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPCMPESTRIOp, true>},
|
||||
{OPD(3, 0b01, 0x62), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPCMPISTRMOp, true>},
|
||||
{OPD(3, 0b01, 0x63), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPCMPISTRIOp, true>},
|
||||
|
||||
{OPD(3, 0b01, 0xDF), 1, &OpDispatchBuilder::AESKeyGenAssist},
|
||||
};
|
||||
|
||||
@@ -392,8 +392,11 @@ namespace InstFlags {
|
||||
constexpr InstFlagType FLAGS_REX_W_1 = (1ULL << 29);
|
||||
|
||||
constexpr InstFlagType FLAGS_CALL = (1ULL << 30);
|
||||
|
||||
constexpr InstFlagType FLAGS_SUPPORTS_LOCK = (1ULL << 31);
|
||||
// Flags [57..32]: Undefined
|
||||
// Flags [60..58]: Dst size
|
||||
constexpr InstFlagType FLAGS_SIZE_DST_OFF = 58;
|
||||
// Flags [63..61]: Src size
|
||||
constexpr InstFlagType FLAGS_SIZE_SRC_OFF = FLAGS_SIZE_DST_OFF + 3;
|
||||
|
||||
constexpr InstFlagType SIZE_MASK = 0b111;
|
||||
|
||||
@@ -717,7 +717,7 @@
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"CacheLineClean GPR:$Addr": {
|
||||
"Desc": ["Does a 64 byte cacheline cleanat the address specified",
|
||||
"Desc": ["Does a 64 byte cacheline clean at the address specified",
|
||||
"Only cleans the data cachelines. Doesn't do any zeroing",
|
||||
"Skips the invalidation step of the CacheLineClear operation"
|
||||
],
|
||||
@@ -2762,11 +2762,11 @@
|
||||
},
|
||||
"FPR = F64FPREM FPR:$Src1, FPR:$Src2": {
|
||||
"DestSize": "OpSize::i64Bit",
|
||||
"JITDispatch": false
|
||||
"JITDispatch": true
|
||||
},
|
||||
"FPR = F64FPREM1 FPR:$Src1, FPR:$Src2": {
|
||||
"DestSize": "OpSize::i64Bit",
|
||||
"JITDispatch": false
|
||||
"JITDispatch": true
|
||||
},
|
||||
"FPR = F64SCALE FPR:$Src1, FPR:$Src2": {
|
||||
"DestSize": "OpSize::i64Bit",
|
||||
@@ -2780,6 +2780,10 @@
|
||||
"DestSize": "OpSize::i64Bit",
|
||||
"JITDispatch": true
|
||||
},
|
||||
"FPR = F64FYL2XP1 FPR:$Src, FPR:$Src2": {
|
||||
"DestSize": "OpSize::i64Bit",
|
||||
"JITDispatch": true
|
||||
},
|
||||
"FPR = F64TAN FPR:$Src": {
|
||||
"DestSize": "OpSize::i64Bit",
|
||||
"JITDispatch": true
|
||||
@@ -3208,6 +3212,20 @@
|
||||
"DestSize": "OpSize::i128Bit",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"FPR = F80FYL2XP1Stack": {
|
||||
"Desc": [
|
||||
"Computes ST1 * log2(1 + ST0)",
|
||||
"Stores the result in ST1, and pops the top of the stack.",
|
||||
"Returns the new value at the top of the stack, i.e. the result of the operation."
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "OpSize::i128Bit",
|
||||
"X87": true
|
||||
},
|
||||
"FPR = F80FYL2XP1 FPR:$X80Src1, FPR:$X80Src2": {
|
||||
"DestSize": "OpSize::i128Bit",
|
||||
"JITDispatch": false
|
||||
},
|
||||
"F80VBSLStack OpSize:#RegisterSize, FPR:$VectorMask, u8:$SrcStack1, u8:$SrcStack2": {
|
||||
"Desc": [
|
||||
"Does a vector bitwise select.",
|
||||
|
||||
@@ -208,6 +208,10 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, NamedVectorCon
|
||||
return "movmaskb";
|
||||
case NamedVectorConstant::NAMED_VECTOR_MOVMASKB_UPPER:
|
||||
return "movmaskb_upper";
|
||||
case NamedVectorConstant::NAMED_VECTOR_256_MID_ELEMENT_SWAP:
|
||||
return "v256_mid_element_swap";
|
||||
case NamedVectorConstant::NAMED_VECTOR_256_MID_ELEMENT_SWAP_UPPER:
|
||||
return "v256_mid_element_swap_upper";
|
||||
case NamedVectorConstant::NAMED_VECTOR_ZERO:
|
||||
return "vectorzero";
|
||||
case NamedVectorConstant::NAMED_VECTOR_X87_ONE:
|
||||
|
||||
@@ -186,12 +186,12 @@ public:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
unsigned PostRA() const {
|
||||
bool PostRA() const {
|
||||
return GetHeader()->PostRA;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
unsigned SpillSlots() const {
|
||||
uint32_t SpillSlots() const {
|
||||
return GetHeader()->SpillSlots;
|
||||
}
|
||||
|
||||
|
||||
@@ -8,23 +8,18 @@ class CPUIDEmu;
|
||||
struct HostFeatures;
|
||||
} // namespace FEXCore
|
||||
|
||||
namespace FEXCore::Utils {
|
||||
class IntrusivePooledAllocator;
|
||||
}
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class Pass;
|
||||
class RegisterAllocationPass;
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadFlagCalculationEliminination();
|
||||
fextl::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass(const FEXCore::CPUIDEmu* CPUID);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateX87StackOptimizationPass(const FEXCore::HostFeatures&, OpSize GPROpSize);
|
||||
fextl::unique_ptr<Pass> CreateDeadFlagCalculationEliminination();
|
||||
fextl::unique_ptr<Pass> CreateRegisterAllocationPass(const CPUIDEmu* CPUID);
|
||||
fextl::unique_ptr<Pass> CreateX87StackOptimizationPass(const HostFeatures&, OpSize GPROpSize);
|
||||
|
||||
namespace Validation {
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateIRValidation();
|
||||
fextl::unique_ptr<Pass> CreateIRValidation();
|
||||
} // namespace Validation
|
||||
|
||||
namespace Debug {
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateIRDumper();
|
||||
fextl::unique_ptr<Pass> CreateIRDumper();
|
||||
}
|
||||
} // namespace FEXCore::IR
|
||||
@@ -67,7 +67,7 @@ void IRDumper::Run(IREmitter* IREmit) {
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateIRDumper() {
|
||||
fextl::unique_ptr<Pass> CreateIRDumper() {
|
||||
return fextl::make_unique<IRDumper>();
|
||||
}
|
||||
} // namespace FEXCore::IR::Debug
|
||||
@@ -271,7 +271,7 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateIRValidation() {
|
||||
fextl::unique_ptr<Pass> CreateIRValidation() {
|
||||
return fextl::make_unique<IRValidation>();
|
||||
}
|
||||
} // namespace FEXCore::IR::Validation
|
||||
@@ -747,7 +747,7 @@ void DeadFlagCalculationEliminination::Run(IREmitter* IREmit) {
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadFlagCalculationEliminination() {
|
||||
fextl::unique_ptr<Pass> CreateDeadFlagCalculationEliminination() {
|
||||
return fextl::make_unique<DeadFlagCalculationEliminination>();
|
||||
}
|
||||
|
||||
|
||||
@@ -781,7 +781,7 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
IR->GetHeader()->PostRA = true;
|
||||
}
|
||||
|
||||
fextl::unique_ptr<IR::RegisterAllocationPass> CreateRegisterAllocationPass(const FEXCore::CPUIDEmu* CPUID) {
|
||||
fextl::unique_ptr<IR::Pass> CreateRegisterAllocationPass(const CPUIDEmu* CPUID) {
|
||||
return fextl::make_unique<ConstrainedRAPass>(CPUID);
|
||||
}
|
||||
} // namespace FEXCore::IR
|
||||
@@ -188,7 +188,7 @@ private:
|
||||
|
||||
void Store80BitToMem(const IROp_StoreStackMem* Op, Ref StackNode, Ref AddrNode, Ref Offset, OpSize Align, MemOffsetType OffsetType,
|
||||
uint8_t OffsetScale) {
|
||||
if (Features.SupportsSVE128 || Features.SupportsSVE256) {
|
||||
if (Features.SupportsSVE()) {
|
||||
AddressMode A {.Base = AddrNode,
|
||||
.Index = Op->Offset.IsInvalid() ? nullptr : Offset,
|
||||
.IndexType = MemOffsetType::SXTX,
|
||||
@@ -785,6 +785,12 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_F80FYL2XP1STACK: {
|
||||
HandleBinopStack(OP_F64FYL2XP1, false, OP_F80FYL2XP1, 1, 0, 1);
|
||||
StackPop();
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_F80ATANSTACK: {
|
||||
HandleBinopStack(OP_F64ATAN, false, OP_F80ATAN, 1, 1, 0);
|
||||
StackPop();
|
||||
@@ -1225,7 +1231,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
return;
|
||||
}
|
||||
|
||||
fextl::unique_ptr<Pass> CreateX87StackOptimizationPass(const FEXCore::HostFeatures& Features, OpSize GPROpSize) {
|
||||
fextl::unique_ptr<Pass> CreateX87StackOptimizationPass(const HostFeatures& Features, OpSize GPROpSize) {
|
||||
return fextl::make_unique<X87StackOptimization>(Features, GPROpSize);
|
||||
}
|
||||
} // namespace FEXCore::IR
|
||||
@@ -114,27 +114,17 @@ FEX_DEFAULT_VISIBILITY size_t DetermineVASize() {
|
||||
};
|
||||
|
||||
for (auto Bits : TLBSizes) {
|
||||
uintptr_t Size = 1ULL << Bits;
|
||||
// Just try allocating
|
||||
// We can't actually determine VA size on ARM safely
|
||||
auto Find = [](uintptr_t Size) -> bool {
|
||||
for (int i = 0; i < 64; ++i) {
|
||||
// Try grabbing a some of the top pages of the range
|
||||
// x86 allocates some high pages in the top end
|
||||
void* Ptr = ::mmap(reinterpret_cast<void*>(Size - FEXCore::Utils::FEX_PAGE_SIZE * i), FEXCore::Utils::FEX_PAGE_SIZE, PROT_NONE,
|
||||
MAP_FIXED_NOREPLACE | MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
if (Ptr != (void*)~0ULL) {
|
||||
::munmap(Ptr, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
if (Ptr == (void*)(Size - FEXCore::Utils::FEX_PAGE_SIZE * i)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
return false;
|
||||
};
|
||||
|
||||
if (Find(Size)) {
|
||||
HostVASize = Bits;
|
||||
// We can't actually determine VA size on ARM safely.
|
||||
// Instead, try allocating the page at the top of the range.
|
||||
// If this succeeds OR the page is reported as already existing,
|
||||
// we know we're in valid VA space. Otherwise, we must go lower.
|
||||
void* Addr = reinterpret_cast<void*>((1ULL << Bits) - FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
void* Ptr = ::mmap(Addr, FEXCore::Utils::FEX_PAGE_SIZE, PROT_NONE, MAP_FIXED_NOREPLACE | MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
if (Ptr != (void*)~0ULL) {
|
||||
::munmap(Ptr, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
}
|
||||
if (Ptr != (void*)~0ULL || errno == EEXIST) {
|
||||
return Bits;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -343,11 +343,12 @@ static bool RunCASPAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg1, uint3
|
||||
// 32bit
|
||||
uint64_t Addr = GPRs[AddressReg];
|
||||
|
||||
// Lower register must be even, so only upper register can be 31.
|
||||
uint32_t DesiredLower = GPRs[DesiredReg1];
|
||||
uint32_t DesiredUpper = GPRs[DesiredReg2];
|
||||
uint32_t DesiredUpper = DesiredReg2 == 31 ? 0 : GPRs[DesiredReg2];
|
||||
|
||||
uint32_t ExpectedLower = GPRs[ExpectedReg1];
|
||||
uint32_t ExpectedUpper = GPRs[ExpectedReg2];
|
||||
uint32_t ExpectedUpper = ExpectedReg2 == 31 ? 0 : GPRs[ExpectedReg2];
|
||||
|
||||
// Cross-cacheline CAS doesn't work on ARM
|
||||
// It isn't even guaranteed to work on x86
|
||||
@@ -1352,7 +1353,9 @@ static std::optional<uint64_t> DoCAS(uint32_t Size, uint64_t Desired, uint64_t E
|
||||
}
|
||||
|
||||
static bool RunCASAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg, uint32_t ExpectedReg, uint32_t AddressReg, uint32_t* StrictSplitLockMutex) {
|
||||
std::optional<uint64_t> Res = DoCAS(Size, GPRs[DesiredReg], GPRs[ExpectedReg], GPRs[AddressReg], StrictSplitLockMutex);
|
||||
uint64_t Desired = DesiredReg == 31 ? 0 : GPRs[DesiredReg];
|
||||
uint64_t Expected = ExpectedReg == 31 ? 0 : GPRs[ExpectedReg];
|
||||
std::optional<uint64_t> Res = DoCAS(Size, Desired, Expected, GPRs[AddressReg], StrictSplitLockMutex);
|
||||
if (!Res.has_value()) {
|
||||
return false;
|
||||
}
|
||||
@@ -1384,6 +1387,8 @@ static bool HandleAtomicMemOp(uint32_t Instr, uint64_t* GPRs, uint32_t* StrictSp
|
||||
|
||||
uint8_t Op = (Instr >> 12) & 0xF;
|
||||
|
||||
uint64_t Source = SourceReg == 31 ? 0 : GPRs[SourceReg];
|
||||
|
||||
if (Size == 2) {
|
||||
auto NOPExpected = [](uint16_t SrcVal, uint16_t) -> uint16_t {
|
||||
return SrcVal;
|
||||
@@ -1420,7 +1425,7 @@ static bool HandleAtomicMemOp(uint32_t Instr, uint64_t* GPRs, uint32_t* StrictSp
|
||||
default: LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}", Op); return false;
|
||||
}
|
||||
|
||||
auto Res = DoCAS16<true>(GPRs[SourceReg],
|
||||
auto Res = DoCAS16<true>(Source,
|
||||
0, // Unused
|
||||
Addr, NOPExpected, DesiredFunction, StrictSplitLockMutex);
|
||||
// If we passed and our destination register is not zero
|
||||
@@ -1465,7 +1470,7 @@ static bool HandleAtomicMemOp(uint32_t Instr, uint64_t* GPRs, uint32_t* StrictSp
|
||||
default: LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}", Op); return false;
|
||||
}
|
||||
|
||||
auto Res = DoCAS32<true>(GPRs[SourceReg],
|
||||
auto Res = DoCAS32<true>(Source,
|
||||
0, // Unused
|
||||
Addr, NOPExpected, DesiredFunction, StrictSplitLockMutex);
|
||||
// If we passed and our destination register is not zero
|
||||
@@ -1510,7 +1515,7 @@ static bool HandleAtomicMemOp(uint32_t Instr, uint64_t* GPRs, uint32_t* StrictSp
|
||||
default: LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}", Op); return false;
|
||||
}
|
||||
|
||||
auto Res = DoCAS64<true>(GPRs[SourceReg],
|
||||
auto Res = DoCAS64<true>(Source,
|
||||
0, // Unused
|
||||
Addr, NOPExpected, DesiredFunction, StrictSplitLockMutex);
|
||||
// If we passed and our destination register is not zero
|
||||
@@ -1572,9 +1577,11 @@ static bool HandleAtomicStore(uint32_t Instr, uint64_t* GPRs, int64_t Offset, ui
|
||||
uint64_t Addr = GPRs[AddressReg] + Offset;
|
||||
|
||||
constexpr bool DoRetry = false;
|
||||
uint64_t Data = DataReg == 31 ? 0 : GPRs[DataReg];
|
||||
|
||||
if (Size == 2) {
|
||||
DoCAS16<DoRetry>(
|
||||
GPRs[DataReg],
|
||||
Data,
|
||||
0, // Unused
|
||||
Addr,
|
||||
[](uint16_t SrcVal, uint16_t) -> uint16_t {
|
||||
@@ -1589,7 +1596,7 @@ static bool HandleAtomicStore(uint32_t Instr, uint64_t* GPRs, int64_t Offset, ui
|
||||
return true;
|
||||
} else if (Size == 4) {
|
||||
DoCAS32<DoRetry>(
|
||||
GPRs[DataReg],
|
||||
Data,
|
||||
0, // Unused
|
||||
Addr,
|
||||
[](uint32_t SrcVal, uint32_t) -> uint32_t {
|
||||
@@ -1604,7 +1611,7 @@ static bool HandleAtomicStore(uint32_t Instr, uint64_t* GPRs, int64_t Offset, ui
|
||||
return true;
|
||||
} else if (Size == 8) {
|
||||
DoCAS64<DoRetry>(
|
||||
GPRs[DataReg],
|
||||
Data,
|
||||
0, // Unused
|
||||
Addr,
|
||||
[](uint64_t SrcVal, uint64_t) -> uint64_t {
|
||||
@@ -1834,6 +1841,7 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
return Desired;
|
||||
};
|
||||
|
||||
uint64_t Source = DataSourceReg == 31 ? 0 : GPRs[DataSourceReg];
|
||||
if (Size == 2) {
|
||||
using AtomicType = uint16_t;
|
||||
CASDesiredFn<AtomicType> DesiredFunction {};
|
||||
@@ -1852,7 +1860,7 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
default: LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}", FEXCore::ToUnderlying(AtomicOp)); return false;
|
||||
}
|
||||
|
||||
auto Res = DoCAS16<DoRetry>(GPRs[DataSourceReg],
|
||||
auto Res = DoCAS16<DoRetry>(Source,
|
||||
0, // Unused
|
||||
Addr, NOPExpected, DesiredFunction, StrictSplitLockMutex);
|
||||
|
||||
@@ -1879,7 +1887,7 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
default: LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}", FEXCore::ToUnderlying(AtomicOp)); return false;
|
||||
}
|
||||
|
||||
auto Res = DoCAS32<DoRetry>(GPRs[DataSourceReg],
|
||||
auto Res = DoCAS32<DoRetry>(Source,
|
||||
0, // Unused
|
||||
Addr, NOPExpected, DesiredFunction, StrictSplitLockMutex);
|
||||
|
||||
@@ -1906,7 +1914,7 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
default: LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}", FEXCore::ToUnderlying(AtomicOp)); return false;
|
||||
}
|
||||
|
||||
auto Res = DoCAS64<DoRetry>(GPRs[DataSourceReg],
|
||||
auto Res = DoCAS64<DoRetry>(Source,
|
||||
0, // Unused
|
||||
Addr, NOPExpected, DesiredFunction, StrictSplitLockMutex);
|
||||
if (AtomicFetch && ResultReg != 31) {
|
||||
@@ -1947,8 +1955,24 @@ std::optional<int32_t> HandleUnalignedAccess(FEXCore::Core::InternalThreadState*
|
||||
uint32_t* StrictSplitLockMutex {CTX->Config.StrictInProcessSplitLocks ? &CTX->StrictSplitLockMutex : nullptr};
|
||||
|
||||
if (!IsJIT) [[unlikely]] {
|
||||
if ((Instr & LDAXR_MASK) == LDAR_INST || // LDAR*
|
||||
(Instr & LDAXR_MASK) == LDAPR_INST) { // LDAPR*
|
||||
if ((Instr & ArchHelpers::Arm64::CASPAL_MASK) == ArchHelpers::Arm64::CASPAL_INST) { // CASPAL
|
||||
if (ArchHelpers::Arm64::HandleCASPAL(Instr, GPRs, StrictSplitLockMutex)) {
|
||||
// Skip this instruction now
|
||||
return 4;
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS CASPAL: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return std::nullopt;
|
||||
}
|
||||
} else if ((Instr & ArchHelpers::Arm64::CASAL_MASK) == ArchHelpers::Arm64::CASAL_INST) { // CASAL
|
||||
if (ArchHelpers::Arm64::HandleCASAL(GPRs, Instr, StrictSplitLockMutex)) {
|
||||
// Skip this instruction now
|
||||
return 4;
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS CASAL: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return std::nullopt;
|
||||
}
|
||||
} else if ((Instr & LDAXR_MASK) == LDAR_INST || // LDAR*
|
||||
(Instr & LDAXR_MASK) == LDAPR_INST) { // LDAPR*
|
||||
if (ArchHelpers::Arm64::HandleAtomicLoad(Instr, GPRs, 0)) {
|
||||
// Skip this instruction now
|
||||
return 4;
|
||||
@@ -1991,16 +2015,17 @@ std::optional<int32_t> HandleUnalignedAccess(FEXCore::Core::InternalThreadState*
|
||||
} else if ((Instr & ArchHelpers::Arm64::STLXR_MASK) == ArchHelpers::Arm64::STLXR_INST) { // STLXR*
|
||||
uint32_t StatusReg = Instr << 11 >> 27;
|
||||
// // Emulate exclusive store by validating the address and value against the last unaligned LDAXR*.
|
||||
if (GPRs[AddrReg] != Thread->ExclusiveStore.Addr || Size > Thread->ExclusiveStore.Size) {
|
||||
uint32_t SizeBytes = 1u << Size;
|
||||
if (GPRs[AddrReg] != Thread->ExclusiveStore.Addr || SizeBytes > Thread->ExclusiveStore.Size) {
|
||||
if (StatusReg != 31) {
|
||||
GPRs[StatusReg] = 1;
|
||||
}
|
||||
return 4;
|
||||
}
|
||||
if (std::optional<uint64_t> Prev =
|
||||
DoCAS(Size, DataReg == 31 ? 0 : GPRs[DataReg], Thread->ExclusiveStore.Store, GPRs[AddrReg], StrictSplitLockMutex)) {
|
||||
DoCAS(SizeBytes, DataReg == 31 ? 0 : GPRs[DataReg], Thread->ExclusiveStore.Store, GPRs[AddrReg], StrictSplitLockMutex)) {
|
||||
if (StatusReg != 31) {
|
||||
GPRs[StatusReg] = !!memcmp(&Thread->ExclusiveStore.Store, &*Prev, Size);
|
||||
GPRs[StatusReg] = !!memcmp(&Thread->ExclusiveStore.Store, &*Prev, SizeBytes);
|
||||
}
|
||||
Thread->ExclusiveStore.Size = 0;
|
||||
return 4;
|
||||
|
||||
@@ -8,6 +8,7 @@
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXCore/fextl/robin_map.h>
|
||||
#include <FEXCore/Utils/TypeDefines.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cstdint>
|
||||
@@ -178,7 +179,55 @@ private:
|
||||
CodeMapOpener& FileOpener;
|
||||
};
|
||||
|
||||
class AbstractCodeCache;
|
||||
|
||||
/**
|
||||
* Manages runtime state associated with a mapped code cache file.
|
||||
*
|
||||
* The mapped file pointer is managed by the frontend and must be valid
|
||||
* throughout the lifetime of this object.
|
||||
*/
|
||||
struct MappedCodeCacheFile {
|
||||
// Calls UnregisterMappedCodeBuffer internally, see its docstring about synchronization requirements
|
||||
~MappedCodeCacheFile();
|
||||
|
||||
// If not nullptr, the MappedCodeCacheFile will be unregistered from this on destruction
|
||||
AbstractCodeCache* CacheManager;
|
||||
|
||||
std::span<std::byte> MappedFile; // Mapped data of the whole cache file
|
||||
std::span<std::byte> CodeBufferInFile; // Subspan of cached ARM64 data within MappedFile (pre-relocation)
|
||||
std::span<std::byte> CodeBuffer; // Cached ARM64 data used for execution (post-relocation; owned by MappedCodeCacheFile)
|
||||
std::byte* BlockListInFile; // Pointer to BlockListEntry data within MappedFile
|
||||
uint32_t NumBlocks; // Number of BlockListEntry objects
|
||||
uint32_t NumCodePages; // Number of code page entrypoint mappings
|
||||
|
||||
struct PageRelocationRange {
|
||||
uint32_t Offset; // In bytes from start of file
|
||||
uint32_t Length; // Number of relocations
|
||||
};
|
||||
|
||||
// List of relocation ranges in the mapped cache file, grouped by the code page they apply to.
|
||||
// This vector is indexed by the relative page offset from the start of the ARM64 code data.
|
||||
//
|
||||
// For example PageRelocationRanges[1] == { 0x100, 0x20 } means:
|
||||
// - there are 0x20 bytes of relocation data at offset 0x100 in the cache file
|
||||
// - these 0x20 bytes of relocation data will patch data at CodeBuffer[0x1000..0x2000]
|
||||
fextl::vector<PageRelocationRange> PageRelocationRanges;
|
||||
fextl::vector<bool> LoadedPages;
|
||||
|
||||
uint64_t GuestBase {}; // Guest base address for relocation application
|
||||
|
||||
// Helper member to prevent moving/copying without disallowing aggregate-construction
|
||||
std::atomic<int> disallow_copy_or_move;
|
||||
|
||||
size_t NumPages() const {
|
||||
return CodeBuffer.size_bytes() / FEXCore::Utils::FEX_PAGE_SIZE;
|
||||
}
|
||||
};
|
||||
|
||||
class AbstractCodeCache {
|
||||
fextl::vector<std::span<std::byte>> MappedCodeBuffers;
|
||||
|
||||
public:
|
||||
virtual ~AbstractCodeCache() = default;
|
||||
|
||||
@@ -190,13 +239,6 @@ public:
|
||||
*/
|
||||
virtual uint64_t ComputeCodeMapId(std::string_view Filename, int FD) = 0;
|
||||
|
||||
/**
|
||||
* Loads a code cache from mapped memory and appends it to the current Core state.
|
||||
* TODO: Optionally recompiles all contained code blocks at runtime for validation.
|
||||
* Returns false if the provided cache file is invalid, and true otherwise.
|
||||
*/
|
||||
virtual bool LoadData(Core::InternalThreadState*, std::byte* MappedCacheFile, const ExecutableFileSectionInfo&) = 0;
|
||||
|
||||
/**
|
||||
* Bundles the current Core state (CodeBuffer, GuestToHostMapping, ...) to a code cache and writes it to the given file descriptor.
|
||||
* Returns true on success.
|
||||
@@ -207,6 +249,42 @@ public:
|
||||
* Function to be called before compiling any code for caching purposes
|
||||
*/
|
||||
virtual void InitiateCacheGeneration() = 0;
|
||||
|
||||
/**
|
||||
* Loads a code cache from mapped memory.
|
||||
*
|
||||
* Code sections must be enabled in a second step (see EnableLoadedSection).
|
||||
* Afterwards, individual code pages must be finalized using FinalizeCodePages.
|
||||
*
|
||||
* On success, this returns a MappedCodeCacheFile that must be kept alive
|
||||
* as long the cache is in use.
|
||||
*/
|
||||
virtual fextl::unique_ptr<MappedCodeCacheFile> LoadCache(std::span<std::byte> CacheFile, const ExecutableFileInfo&, uint64_t FileStartVA) = 0;
|
||||
|
||||
/**
|
||||
* Registers cached blocks for the given file section to the LookupCache.
|
||||
*
|
||||
* Also runs extended cache validation if enabled.
|
||||
*/
|
||||
virtual bool EnableLoadedSection(Core::InternalThreadState*, MappedCodeCacheFile&, const ExecutableFileSectionInfo&) = 0;
|
||||
|
||||
/**
|
||||
* Extend the given code range so that it can be safely finalized.
|
||||
*
|
||||
* This is required for example to avoid dangling page-crossing FEX relocations on the edges
|
||||
*
|
||||
* StartPage and EndPage a 0-based relative page offsets into the cached code.
|
||||
*/
|
||||
static std::span<std::byte> SelectCodeRangeToFinalize(MappedCodeCacheFile&, size_t StartPage, size_t EndPage);
|
||||
|
||||
/**
|
||||
* Finalize code pages in the given range (see SelectCodePagesToFinalize) for execution.
|
||||
*/
|
||||
virtual void FinalizeCodePages(MappedCodeCacheFile&, std::span<std::byte> CodeRange) = 0;
|
||||
|
||||
void RegisterMappedCodeBuffer(MappedCodeCacheFile&);
|
||||
void UnregisterMappedCodeBuffer(MappedCodeCacheFile&);
|
||||
bool IsAddressInMappedCodeBuffer(uintptr_t Address) const;
|
||||
};
|
||||
|
||||
} // namespace FEXCore
|
||||
@@ -302,6 +302,7 @@ enum FallbackHandlerIndex {
|
||||
OPINDEX_F80MUL,
|
||||
OPINDEX_F80DIV,
|
||||
OPINDEX_F80FYL2X,
|
||||
OPINDEX_F80FYL2XP1,
|
||||
OPINDEX_F80ATAN,
|
||||
OPINDEX_F80FPREM1,
|
||||
OPINDEX_F80FPREM,
|
||||
@@ -315,6 +316,7 @@ enum FallbackHandlerIndex {
|
||||
OPINDEX_F64ATAN,
|
||||
OPINDEX_F64F2XM1,
|
||||
OPINDEX_F64FYL2X,
|
||||
OPINDEX_F64FYL2XP1,
|
||||
OPINDEX_F64FPREM,
|
||||
OPINDEX_F64FPREM1,
|
||||
OPINDEX_F64SCALE,
|
||||
@@ -383,6 +385,9 @@ struct JITPointers {
|
||||
uint64_t F64ScaleHandler {};
|
||||
uint64_t F64AtanHandler {};
|
||||
uint64_t F64FYL2XHandler {};
|
||||
uint64_t F64FYL2XP1Handler {};
|
||||
uint64_t F64FPREMHandler {};
|
||||
uint64_t F64FPREM1Handler {};
|
||||
/** @} */
|
||||
|
||||
// Copy of process-wide named vector constants data.
|
||||
|
||||
@@ -5,13 +5,20 @@
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
/**
|
||||
* @brief Backend features that change how codegen is generated from IR
|
||||
*
|
||||
* Specifically things that affect the IR->Codegen process
|
||||
* Not the x86->IR process
|
||||
*/
|
||||
struct HostFeatures {
|
||||
/**
|
||||
* @brief Backend features that change how codegen is generated from IR
|
||||
*
|
||||
* Specifically things that affect the IR->Codegen process
|
||||
* Not the x86->IR process
|
||||
*/
|
||||
// Whether or not the host supports any kind of SVE implementation.
|
||||
[[nodiscard]]
|
||||
bool SupportsSVE() const {
|
||||
return SupportsSVE128 || SupportsSVE256;
|
||||
}
|
||||
|
||||
uint32_t DCacheLineSize {};
|
||||
uint32_t ICacheLineSize {};
|
||||
bool SupportsCacheMaintenanceOps {};
|
||||
@@ -42,11 +49,21 @@ struct HostFeatures {
|
||||
bool Supports3DNow {};
|
||||
bool SupportsSSE4a {};
|
||||
bool SupportsMOPS {};
|
||||
bool PreferZVAForVZero {};
|
||||
|
||||
// Float exception behaviour
|
||||
bool SupportsAFP {};
|
||||
bool SupportsFloatExceptions {};
|
||||
|
||||
// Changes code generation slightly.
|
||||
enum class HostTypeEnum {
|
||||
Unknown,
|
||||
Linux,
|
||||
Wow64,
|
||||
Arm64ec,
|
||||
};
|
||||
HostTypeEnum HostType {};
|
||||
|
||||
// Flag if this is InstCountCI
|
||||
bool IsInstCountCI {};
|
||||
|
||||
|
||||
@@ -125,9 +125,9 @@ struct alignas(FEXCore::Utils::FEX_PAGE_SIZE) InternalThreadState : public FEXCo
|
||||
alignas(FEXCore::Utils::FEX_PAGE_SIZE) uint8_t InterruptFaultPage[FEXCore::Utils::FEX_PAGE_SIZE];
|
||||
};
|
||||
static_assert(std::is_standard_layout_v<FEXCore::Core::InternalThreadState>);
|
||||
static_assert((offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState)) <
|
||||
FEXCore::Utils::FEX_PAGE_SIZE,
|
||||
"Fault page is outside of immediate range from CPU state");
|
||||
static_assert(sizeof(FEXCore::Core::InternalThreadState) == (FEXCore::Utils::FEX_PAGE_SIZE * 2));
|
||||
// Maximum unsigned-offset store range for fault page.
|
||||
static_assert(
|
||||
(offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState)) <= 65520,
|
||||
"Fault page is outside of immediate range from CPU state");
|
||||
|
||||
} // namespace FEXCore::Core
|
||||
@@ -34,6 +34,12 @@ enum NamedVectorConstant : uint8_t {
|
||||
NAMED_VECTOR_MOVMASKB,
|
||||
NAMED_VECTOR_MOVMASKB_UPPER,
|
||||
|
||||
// Used to swap [0, 1, 2, 3] into [0, 2, 1, 3] in lieu
|
||||
// of Q operations introduced in SVE2.1. Can be removed when
|
||||
// such operations become available.
|
||||
NAMED_VECTOR_256_MID_ELEMENT_SWAP,
|
||||
NAMED_VECTOR_256_MID_ELEMENT_SWAP_UPPER,
|
||||
|
||||
NAMED_VECTOR_X87_ONE,
|
||||
NAMED_VECTOR_X87_LOG2_10,
|
||||
NAMED_VECTOR_X87_LOG2_E,
|
||||
|
||||
@@ -35,6 +35,9 @@ public:
|
||||
const auto Result = pthread_mutex_lock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to lock with {}", __func__, Result);
|
||||
}
|
||||
bool try_lock() {
|
||||
return pthread_mutex_trylock(&Mutex) == 0;
|
||||
}
|
||||
void unlock() {
|
||||
const auto Result = pthread_mutex_unlock(&Mutex);
|
||||
LOGMAN_THROW_A_FMT(Result == 0, "{} failed to unlock with {}", __func__, Result);
|
||||
|
||||
@@ -656,7 +656,7 @@ fextl::string GetDataDirectory(bool Global, const PortableInformation& PortableI
|
||||
fextl::string GetConfigDirectory(bool Global, const PortableInformation& PortableInfo) {
|
||||
const char* ConfigOverride = getenv("FEX_APP_CONFIG_LOCATION");
|
||||
if (PortableInfo.IsPortable && Global) {
|
||||
return fextl::fmt::format("{}/fex-emu/", PortableInfo.InterpreterPath);
|
||||
return fextl::fmt::format("{}/../share/fex-emu/", PortableInfo.InterpreterPath);
|
||||
} else if (ConfigOverride && !Global) {
|
||||
fextl::string AppConfigStr = ConfigOverride;
|
||||
if (FHU::Filesystem::IsRelative(AppConfigStr)) {
|
||||
|
||||
@@ -137,21 +137,31 @@ fextl::string GetServerSocketName() {
|
||||
return ServerSocketPath;
|
||||
}
|
||||
|
||||
fextl::string GetServerSocketPath() {
|
||||
fextl::string GetServerSocketPath(bool ForceTmp) {
|
||||
fextl::string name {};
|
||||
fextl::string Folder {};
|
||||
#ifndef FEX_STEAM_SUPPORT
|
||||
FEX_CONFIG_OPT(ServerSocketPath, SERVERSOCKETPATH);
|
||||
|
||||
name = ServerSocketPath();
|
||||
if (!ForceTmp) {
|
||||
name = ServerSocketPath();
|
||||
|
||||
if (name.starts_with("/")) {
|
||||
return name;
|
||||
if (name.starts_with("/")) {
|
||||
return name;
|
||||
}
|
||||
}
|
||||
|
||||
auto Folder = GetTempFolder();
|
||||
Folder = GetTempFolder();
|
||||
#else
|
||||
// Under Steam the FEXServer's socket is a game-specific directory.
|
||||
auto Folder = GetServerLockFolder();
|
||||
if (ForceTmp) {
|
||||
// If we're forcing temporary directory usage then the server socket path has exceeded sun_path 108 byte limit.
|
||||
// Let's be a bit nice and put some more metadata in the server socket path.
|
||||
const auto SteamID = getenv("SteamAppId") ?: "";
|
||||
return fextl::fmt::format("{}/{}.FEXServer.Socket", GetTempFolder(), SteamID);
|
||||
} else {
|
||||
// Under Steam the FEXServer's socket is a game-specific directory.
|
||||
Folder = GetServerLockFolder();
|
||||
}
|
||||
#endif
|
||||
|
||||
if (name.empty()) {
|
||||
@@ -203,7 +213,11 @@ int ConnectToServer(ConnectionOption ConnectionOption) {
|
||||
|
||||
// Try again with a path-based socket, since abstract sockets will fail if we have been
|
||||
// placed in a new netns as part of a sandbox.
|
||||
auto ServerSocketPath = GetServerSocketPath();
|
||||
auto ServerSocketPath = GetServerSocketPath(false);
|
||||
if (ServerSocketPath.size() > sizeof(sockaddr_un::sun_path) - 1) {
|
||||
LogMan::Msg::EFmt("Socket path '{}' too large for Unix domain sockets. Moving to tmp", ServerSocketPath);
|
||||
ServerSocketPath = FEXServerClient::GetServerSocketPath(true);
|
||||
}
|
||||
|
||||
addr.sun_family = AF_UNIX;
|
||||
SizeOfSocketString = std::min(ServerSocketPath.size(), sizeof(addr.sun_path) - 1);
|
||||
|
||||
@@ -64,7 +64,7 @@ fextl::string GetServerRootFSLockFile();
|
||||
fextl::string GetTempFolder();
|
||||
fextl::string GetServerMountFolder();
|
||||
fextl::string GetServerSocketName();
|
||||
fextl::string GetServerSocketPath();
|
||||
fextl::string GetServerSocketPath(bool ForceTmp);
|
||||
int GetServerFD();
|
||||
|
||||
bool SetupClient(std::string_view InterpreterPath);
|
||||
|
||||
@@ -541,6 +541,8 @@ static void HandleErrata(FEXCore::HostFeatures* HostFeatures, uint64_t MIDR) {
|
||||
constexpr uint32_t PartNum_Oryon1 = 0x001;
|
||||
constexpr uint32_t PartNum_Oryon3 = 0x002;
|
||||
|
||||
constexpr uint32_t Implementer_Ampere = 0xc0;
|
||||
|
||||
auto GetMIDRImplementer = [](uint32_t MIDR) -> uint32_t {
|
||||
return (MIDR >> 24) & 0xFF;
|
||||
};
|
||||
@@ -593,6 +595,15 @@ static void HandleErrata(FEXCore::HostFeatures* HostFeatures, uint64_t MIDR) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (MIDR_Implementer == Implementer_Ampere) {
|
||||
// Ampere Computing CPUs that support CLZero should prefer using `dc zva` for vzero{upper,all} as its faster there.
|
||||
// For Cortex CPUs it doesn't matter one way or the other.
|
||||
// For Oryon CPUs, it is dramatically faster to avoid `dc zva` as it has dramatic stalls around barriers and overlapping `dc zva`.
|
||||
//
|
||||
// Because the `dc zva` optimization was implemented for Ampere, only use that path on the hardware.
|
||||
HostFeatures->PreferZVAForVZero = HostFeatures->SupportsCLZERO;
|
||||
}
|
||||
}
|
||||
|
||||
void FetchHostFeatures(FEX::CPUFeatures& Features, FEXCore::HostFeatures& HostFeatures, bool SupportsCacheMaintenanceOps, uint64_t CTR,
|
||||
@@ -658,6 +669,7 @@ void FetchHostFeatures(FEX::CPUFeatures& Features, FEXCore::HostFeatures& HostFe
|
||||
HostFeatures.SupportsAVX = true;
|
||||
HostFeatures.SupportsAES256 = HostFeatures.SupportsAVX && HostFeatures.SupportsAES;
|
||||
HostFeatures.SupportsPreserveAllABI = FEX_HAS_PRESERVE_ALL_ATTR;
|
||||
HostFeatures.PreferZVAForVZero = false;
|
||||
|
||||
if (CTR) {
|
||||
HostFeatures.DCacheLineSize = 4 << ((CTR >> 16) & 0xF);
|
||||
@@ -742,7 +754,7 @@ FEXCore::HostFeatures FetchHostFeatures() {
|
||||
|
||||
uint64_t CTR = 0;
|
||||
uint64_t MIDR = 0;
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
#if defined(ARCHITECTURE_arm64) && !defined(VIXL_SIMULATOR)
|
||||
// We need to get the CPU's cache line size
|
||||
// We expect sane targets that have correct cacheline sizes across clusters
|
||||
__asm volatile("mrs %[ctr], ctr_el0" : [ctr] "=r"(CTR));
|
||||
@@ -754,6 +766,7 @@ FEXCore::HostFeatures FetchHostFeatures() {
|
||||
FetchHostFeatures(Features, HostFeatures, true, CTR, MIDR);
|
||||
|
||||
HostFeatures.SupportsCPUIndexInTPIDRRO = false;
|
||||
HostFeatures.HostType = FEXCore::HostFeatures::HostTypeEnum::Linux;
|
||||
return HostFeatures;
|
||||
}
|
||||
} // namespace FEX
|
||||
@@ -2,7 +2,7 @@
|
||||
/*
|
||||
$info$
|
||||
tags: Bin|FEXBash
|
||||
desc: Launches bash under FEX and passes arguments via -c to it
|
||||
desc: Wrapper for invoking x86 bash from the rootfs using FEX
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
@@ -13,58 +13,69 @@ $end_info$
|
||||
#include <unistd.h>
|
||||
#include <vector>
|
||||
|
||||
static std::string EnchantedPS1(const char* PS1Env) {
|
||||
using namespace std::string_view_literals;
|
||||
const char* ColorTerm = getenv("COLORTERM");
|
||||
const bool SupportsTrueColor = isatty(STDIN_FILENO) && ColorTerm && (ColorTerm == "truecolor"sv || ColorTerm == "24bit"sv);
|
||||
|
||||
std::string PS1 = "PS1=FEXBash ";
|
||||
if (SupportsTrueColor) {
|
||||
// Rainbow FEXBash text matching the FEX logo
|
||||
PS1 = "PS1="
|
||||
R"(\[\e[38;2;251;0;145m\]F)"
|
||||
R"(\[\e[38;2;199;0;198m\]E)"
|
||||
R"(\[\e[38;2;141;68;253m\]X)"
|
||||
R"(\[\e[38;2;31;159;248m\]B)"
|
||||
R"(\[\e[38;2;0;218;181m\]a)"
|
||||
R"(\[\e[38;2;129;229;93m\]s)"
|
||||
R"(\[\e[38;2;240;220;10m\]h)"
|
||||
R"(\[\e[0m\] )";
|
||||
}
|
||||
|
||||
if (PS1Env) {
|
||||
PS1 += &PS1Env[strlen("PS1=")];
|
||||
} else {
|
||||
PS1 += R"(\u@\h:\w> )";
|
||||
}
|
||||
|
||||
return PS1;
|
||||
}
|
||||
|
||||
int main(int argc, char** argv, char** const envp) {
|
||||
// Skip argv[0].
|
||||
const int ArgCount = argc - 1;
|
||||
const bool EmptyArgs = ArgCount == 0;
|
||||
// Skip argv[0]
|
||||
const bool EmptyArgs = argc == 1;
|
||||
|
||||
std::vector<const char*> Argv;
|
||||
// FEX will handle finding bash in the rootfs
|
||||
// Use /bin/sh for -c commands and /bin/bash for interactive mode
|
||||
// Use /bin/bash for interactive mode and /bin/sh when running a script or when using -c
|
||||
const char* BashPath = EmptyArgs ? "/bin/bash" : "/bin/sh";
|
||||
|
||||
std::string FEXPath = std::filesystem::path(argv[0]).parent_path().string() + "/FEX";
|
||||
auto FEXPath = std::filesystem::path(argv[0]).replace_filename("FEX");
|
||||
|
||||
// Check if a local FEX to FEXBash exists
|
||||
// If it does then it takes priority over the installed one
|
||||
if (!std::filesystem::exists(FEXPath)) {
|
||||
char FEXBashPath[PATH_MAX];
|
||||
auto Result = readlink("/proc/self/exe", FEXBashPath, PATH_MAX);
|
||||
if (Result != -1) {
|
||||
FEXPath = std::filesystem::path(&FEXBashPath[0], &FEXBashPath[Result]).parent_path().string() + "/FEX";
|
||||
if (!std::filesystem::is_regular_file(FEXPath)) {
|
||||
std::error_code ec;
|
||||
auto FEXBashPath = std::filesystem::read_symlink("/proc/self/exe", ec);
|
||||
if (!ec) {
|
||||
FEXPath = FEXBashPath.replace_filename("FEX");
|
||||
}
|
||||
|
||||
if (!std::filesystem::exists(FEXPath)) {
|
||||
if (!std::filesystem::is_regular_file(FEXPath)) {
|
||||
fmt::print(stderr, "Could not locate FEX executable\n");
|
||||
std::abort();
|
||||
}
|
||||
}
|
||||
const char* FEXArgs[] = {
|
||||
FEXPath.c_str(),
|
||||
BashPath,
|
||||
"-c",
|
||||
};
|
||||
|
||||
// Remove -c argument if arguments are empty
|
||||
// Lets us start an emulated bash instance
|
||||
const size_t FEXArgsCount = std::size(FEXArgs) - (EmptyArgs ? 1 : 0);
|
||||
|
||||
Argv.resize(ArgCount + FEXArgsCount);
|
||||
|
||||
// Pass in the FEX arguments
|
||||
for (size_t i = 0; i < FEXArgsCount; ++i) {
|
||||
Argv[i] = FEXArgs[i];
|
||||
}
|
||||
|
||||
// Bring in passed in arguments
|
||||
for (size_t i = 0; i < ArgCount; ++i) {
|
||||
Argv[i + FEXArgsCount] = argv[i + 1];
|
||||
std::vector<const char*> Argv;
|
||||
Argv.emplace_back(FEXPath.c_str());
|
||||
Argv.emplace_back(BashPath);
|
||||
for (int i = 1; i < argc; ++i) {
|
||||
Argv.emplace_back(argv[i]);
|
||||
}
|
||||
|
||||
// Set --norc when no arguments are passed so PS1 doesn't get overwritten
|
||||
const char* NoRC = "--norc";
|
||||
if (EmptyArgs) {
|
||||
Argv.emplace_back(NoRC);
|
||||
Argv.emplace_back("--norc");
|
||||
}
|
||||
|
||||
Argv.emplace_back(nullptr);
|
||||
@@ -86,11 +97,8 @@ int main(int argc, char** argv, char** const envp) {
|
||||
Envp.emplace_back(envp[i]);
|
||||
}
|
||||
}
|
||||
|
||||
std::string PS1 = "PS1=FEXBash-\\u@\\h:\\w> ";
|
||||
if (PS1Env) {
|
||||
PS1 += &PS1Env[strlen("PS1=")];
|
||||
}
|
||||
// Keep the string alive until after execve.
|
||||
const std::string PS1 = EnchantedPS1(PS1Env);
|
||||
Envp.emplace_back(PS1.c_str());
|
||||
Envp.emplace_back(nullptr);
|
||||
|
||||
|
||||
@@ -17,6 +17,7 @@
|
||||
#include <signal.h>
|
||||
#include <ucontext.h>
|
||||
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
namespace {
|
||||
struct TSOEmulationFacts {
|
||||
bool LSE {}, LSE2 {};
|
||||
@@ -24,7 +25,6 @@ struct TSOEmulationFacts {
|
||||
bool LRCPC1 {}, LRCPC2 {}, LRCPC3 {};
|
||||
};
|
||||
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
bool CheckForHardwareTSO() {
|
||||
// Check to see if this is supported.
|
||||
auto Result = prctl(PR_GET_MEM_MODEL, 0, 0, 0, 0);
|
||||
@@ -89,14 +89,8 @@ TSOEmulationFacts GetTSOEmulationFacts() {
|
||||
.LRCPC3 = ((ISAR1 >> ISAR1_FIELDS::LRCPC) & IDFIELDMASK) >= 0b0011,
|
||||
};
|
||||
}
|
||||
#else
|
||||
TSOEmulationFacts GetTSOEmulationFacts() {
|
||||
return {};
|
||||
}
|
||||
#endif
|
||||
} // namespace
|
||||
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
namespace SIGBUSTest {
|
||||
static bool* FaultArray {};
|
||||
|
||||
@@ -131,6 +125,41 @@ __attribute__((naked)) void atomic_load_u128(std::byte* Data) {
|
||||
: "x1", "x2", "x3", "memory");
|
||||
}
|
||||
|
||||
__attribute__((naked)) void atomic_set_u16(std::byte* Data, uint16_t value) {
|
||||
asm volatile(R"(
|
||||
.word 0x78e13002; // ldsetalh w1, w2, [x0];
|
||||
ret;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
|
||||
__attribute__((naked)) void atomic_set_u32(std::byte* Data, uint32_t value) {
|
||||
asm volatile(R"(
|
||||
.word 0xb8e13002; // ldsetal w1, w2, [x0];
|
||||
ret;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
|
||||
__attribute__((naked)) void atomic_set_u64(std::byte* Data, uint64_t value) {
|
||||
asm volatile(R"(
|
||||
.word 0xf8e13002; // ldsetal x1, x2, [x0];
|
||||
ret;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
__attribute__((naked)) void atomic_set_u128_impl(__uint128_t expected, __uint128_t desired, std::byte* Data) {
|
||||
asm volatile(R"(
|
||||
.word 0x4860fc82; // caspal x0, x1, x2, x3, [x4];
|
||||
ret;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
|
||||
static inline void atomic_set_u128(std::byte* Data, __uint128_t value) {
|
||||
atomic_set_u128_impl(*reinterpret_cast<__uint128_t*>(Data), value, Data);
|
||||
}
|
||||
|
||||
static void HandleSIGBUS(int, siginfo_t* info, void* context) {
|
||||
FaultArray[reinterpret_cast<uintptr_t>(info->si_addr) & 63] = true;
|
||||
|
||||
@@ -140,7 +169,22 @@ static void HandleSIGBUS(int, siginfo_t* info, void* context) {
|
||||
mcontext->pc += 4;
|
||||
}
|
||||
|
||||
void TestSIGBUS() {
|
||||
static bool CalculatedFaultOffsets {};
|
||||
static bool FaultOffset_16bit[64] {};
|
||||
static bool FaultOffset_32bit[64] {};
|
||||
static bool FaultOffset_64bit[64] {};
|
||||
static bool FaultOffset_128bit[64] {};
|
||||
|
||||
static bool FaultOffset_RMW_16bit[64] {};
|
||||
static bool FaultOffset_RMW_32bit[64] {};
|
||||
static bool FaultOffset_RMW_64bit[64] {};
|
||||
static bool FaultOffset_RMW_128bit[64] {};
|
||||
|
||||
void RunFaultTests() {
|
||||
if (CalculatedFaultOffsets) {
|
||||
return;
|
||||
}
|
||||
|
||||
struct sigaction act {};
|
||||
act.sa_sigaction = HandleSIGBUS;
|
||||
act.sa_flags = SA_SIGINFO;
|
||||
@@ -154,6 +198,36 @@ void TestSIGBUS() {
|
||||
}
|
||||
};
|
||||
|
||||
auto test_rmw_fault = [](bool* FaultOffsets, auto AccessFunction, std::byte* AccessArray) {
|
||||
FaultArray = FaultOffsets;
|
||||
for (size_t i = 0; i < 64; ++i) {
|
||||
AccessFunction(AccessArray + i, 1);
|
||||
}
|
||||
};
|
||||
|
||||
test_fault(FaultOffset_16bit, atomic_load_u16, ptr);
|
||||
test_fault(FaultOffset_32bit, atomic_load_u32, ptr);
|
||||
test_fault(FaultOffset_64bit, atomic_load_u64, ptr);
|
||||
test_fault(FaultOffset_128bit, atomic_load_u128, ptr);
|
||||
|
||||
auto TSOFacts = GetTSOEmulationFacts();
|
||||
|
||||
if (TSOFacts.LSE) {
|
||||
test_rmw_fault(FaultOffset_RMW_16bit, atomic_set_u16, ptr);
|
||||
test_rmw_fault(FaultOffset_RMW_32bit, atomic_set_u32, ptr);
|
||||
test_rmw_fault(FaultOffset_RMW_64bit, atomic_set_u64, ptr);
|
||||
test_rmw_fault(FaultOffset_RMW_128bit, atomic_set_u128, ptr);
|
||||
}
|
||||
|
||||
munmap(ptr, 4096);
|
||||
sigaction(SIGBUS, &act, nullptr);
|
||||
|
||||
CalculatedFaultOffsets = true;
|
||||
}
|
||||
|
||||
void PrintSIGBUSInfo() {
|
||||
RunFaultTests();
|
||||
|
||||
auto print_granule = [](const char* size, bool* FaultArray) {
|
||||
std::string output {};
|
||||
for (size_t i = 0; i < 64; ++i) {
|
||||
@@ -171,25 +245,54 @@ void TestSIGBUS() {
|
||||
fprintf(stdout, "%s: %s\n", size, output.c_str());
|
||||
};
|
||||
|
||||
bool FaultOffset_16bit[64] {};
|
||||
bool FaultOffset_32bit[64] {};
|
||||
bool FaultOffset_64bit[64] {};
|
||||
bool FaultOffset_128bit[64] {};
|
||||
auto TSOFacts = GetTSOEmulationFacts();
|
||||
const bool RMWIsDifferent = TSOFacts.LSE && (memcmp(FaultOffset_16bit, FaultOffset_RMW_16bit, sizeof(FaultOffset_16bit)) != 0 ||
|
||||
memcmp(FaultOffset_32bit, FaultOffset_RMW_32bit, sizeof(FaultOffset_32bit)) != 0 ||
|
||||
memcmp(FaultOffset_64bit, FaultOffset_RMW_64bit, sizeof(FaultOffset_64bit)) != 0 ||
|
||||
memcmp(FaultOffset_128bit, FaultOffset_RMW_128bit, sizeof(FaultOffset_128bit)) != 0);
|
||||
|
||||
test_fault(FaultOffset_16bit, atomic_load_u16, ptr);
|
||||
test_fault(FaultOffset_32bit, atomic_load_u32, ptr);
|
||||
test_fault(FaultOffset_64bit, atomic_load_u64, ptr);
|
||||
test_fault(FaultOffset_128bit, atomic_load_u128, ptr);
|
||||
|
||||
munmap(ptr, 4096);
|
||||
sigaction(SIGBUS, &act, nullptr);
|
||||
|
||||
fprintf(stdout, "Fault Granularity: Split every 16 bytes\n");
|
||||
if (!RMWIsDifferent) {
|
||||
fprintf(stdout, "Fault Granularity: Split every 16 bytes\n");
|
||||
} else {
|
||||
fprintf(stdout, "Load/Store Fault Granularity: Split every 16 bytes\n");
|
||||
}
|
||||
print_granule(" 16-bit", FaultOffset_16bit);
|
||||
print_granule(" 32-bit", FaultOffset_32bit);
|
||||
print_granule(" 64-bit", FaultOffset_64bit);
|
||||
print_granule("128-bit", FaultOffset_128bit);
|
||||
|
||||
if (RMWIsDifferent) {
|
||||
fprintf(stdout, "RMW Atomic Fault Granularity: Split every 16 bytes\n");
|
||||
print_granule(" 16-bit", FaultOffset_RMW_16bit);
|
||||
print_granule(" 32-bit", FaultOffset_RMW_32bit);
|
||||
print_granule(" 64-bit", FaultOffset_RMW_64bit);
|
||||
print_granule("128-bit", FaultOffset_RMW_128bit);
|
||||
}
|
||||
}
|
||||
|
||||
struct FirstFaultInformation {
|
||||
int32_t LoadStoreFaultAlignment {};
|
||||
int32_t RMWFaultAlignment {};
|
||||
};
|
||||
|
||||
FirstFaultInformation CalculateFirstFaultInformation() {
|
||||
RunFaultTests();
|
||||
FirstFaultInformation Info {};
|
||||
auto FindFirstFaultOffset = [](bool FaultOffsets[64]) -> int32_t {
|
||||
for (int32_t i = 0; i < 64; ++i) {
|
||||
if (FaultOffsets[i]) {
|
||||
return i;
|
||||
}
|
||||
}
|
||||
|
||||
return -1;
|
||||
};
|
||||
|
||||
Info.LoadStoreFaultAlignment = FindFirstFaultOffset(FaultOffset_16bit) + 1;
|
||||
Info.RMWFaultAlignment = FindFirstFaultOffset(FaultOffset_RMW_16bit) + 1;
|
||||
return Info;
|
||||
}
|
||||
|
||||
} // namespace SIGBUSTest
|
||||
#endif
|
||||
|
||||
@@ -210,9 +313,8 @@ int main(int argc, char** argv, char** envp) {
|
||||
|
||||
Parser.add_option("--current-rootfs").action("store_true").help("Print the directory that contains the FEX rootfs. Mounted in the case of squashfs");
|
||||
|
||||
Parser.add_option("--tso-emulation-info").action("store_true").help("Print how FEX is emulating the x86-TSO memory model.");
|
||||
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
Parser.add_option("--tso-emulation-info").action("store_true").help("Print how FEX is emulating the x86-TSO memory model.");
|
||||
Parser.add_option("--test-fault-granularity").action("store_true").help("Show SIGBUS fault granularity");
|
||||
Parser.add_option("--identification-reg-info").action("store_true").help("Print identification registers");
|
||||
#endif
|
||||
@@ -248,7 +350,7 @@ int main(int argc, char** argv, char** envp) {
|
||||
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
if (Options.is_set_by_user("test_fault_granularity")) {
|
||||
SIGBUSTest::TestSIGBUS();
|
||||
SIGBUSTest::PrintSIGBUSInfo();
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -272,12 +374,23 @@ int main(int argc, char** argv, char** envp) {
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
if (Options.is_set_by_user("tso_emulation_info")) {
|
||||
auto TSOFacts = GetTSOEmulationFacts();
|
||||
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
|
||||
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
|
||||
FEX_CONFIG_OPT(VectorTSOEnabled, VECTORTSOENABLED);
|
||||
FEX_CONFIG_OPT(HalfBarrierTSOEnabled, HALFBARRIERTSOENABLED);
|
||||
FEX_CONFIG_OPT(StrictInProcessSplitLocks, STRICTINPROCESSSPLITLOCKS);
|
||||
|
||||
const char* GPRMemoryTSOEmulation {};
|
||||
const char* MemcpyMemoryTSOEmulation {};
|
||||
const char* VectorMemoryTSOEmulation {};
|
||||
const char* UnalignedMemoryLoadStoreTSOEmulation {};
|
||||
const char* SplitLock16BEmulationType {};
|
||||
const char* SplitLock16BConfigurationType {};
|
||||
std::string UnalignedMemoryLoadStoreAlignmentGranularity {};
|
||||
std::string UnalignedRMWAlignmentGranularity {};
|
||||
|
||||
if (TSOFacts.HardwareTSO) {
|
||||
GPRMemoryTSOEmulation = "\e[32mHardware TSO\e[0m";
|
||||
@@ -314,34 +427,52 @@ int main(int argc, char** argv, char** envp) {
|
||||
UnalignedMemoryLoadStoreTSOEmulation = "\e[31mHalf-Barriers\e[0m";
|
||||
}
|
||||
|
||||
const auto FFInfo = SIGBUSTest::CalculateFirstFaultInformation();
|
||||
|
||||
if (FFInfo.RMWFaultAlignment >= 64) {
|
||||
SplitLock16BEmulationType = "\e[32mHardware cacheline unaligned atomics\e[0m";
|
||||
SplitLock16BConfigurationType = "\e[32mTear-free\e[0m";
|
||||
} else {
|
||||
SplitLock16BEmulationType = TSOFacts.LSE ? "\e[31mTearing CAS loops\e[0m" : "\e[31mTearing LL/SC loops\e[0m";
|
||||
SplitLock16BConfigurationType = StrictInProcessSplitLocks() ? "In-process mutex" : "Tearing";
|
||||
}
|
||||
|
||||
if (FFInfo.LoadStoreFaultAlignment != 1) {
|
||||
UnalignedMemoryLoadStoreAlignmentGranularity = fmt::format("\e[32m{}-byte\e[0m", FFInfo.LoadStoreFaultAlignment);
|
||||
} else {
|
||||
UnalignedMemoryLoadStoreAlignmentGranularity = TSOFacts.LSE2 ? "\e[32m16-byte\e[0m" : "\e[31mNatural alignment\e[0m";
|
||||
}
|
||||
|
||||
if (FFInfo.LoadStoreFaultAlignment != FFInfo.RMWFaultAlignment) {
|
||||
if (FFInfo.LoadStoreFaultAlignment != 1) {
|
||||
UnalignedRMWAlignmentGranularity = fmt::format("\e[32m{}-byte\e[0m", FFInfo.RMWFaultAlignment);
|
||||
} else {
|
||||
UnalignedRMWAlignmentGranularity = TSOFacts.LSE2 ? "\e[32m16-byte\e[0m" : "\e[31mNatural alignment\e[0m";
|
||||
}
|
||||
}
|
||||
|
||||
fprintf(stdout, "Hardware Features:\n");
|
||||
fprintf(stdout, "\tMemory atomics emulation method: %s\n", TSOFacts.LSE ? "\e[32mLSE\e[0m" : "\e[31mLL/SC\e[0m");
|
||||
fprintf(stdout, "\tUnaligned atomic memory granularity: %s\n", TSOFacts.LSE2 ? "\e[32m16-byte\e[0m" : "\e[31mNatural alignment\e[0m");
|
||||
///< TODO: Once TME is supported by hardware this can change.
|
||||
fprintf(stdout, "\tUnaligned atomic memory granularity: %s\n", UnalignedMemoryLoadStoreAlignmentGranularity.c_str());
|
||||
if (FFInfo.LoadStoreFaultAlignment != FFInfo.RMWFaultAlignment) {
|
||||
fprintf(stdout, "\tUnaligned atomic RMW granularity: %s\n", UnalignedRMWAlignmentGranularity.c_str());
|
||||
}
|
||||
fprintf(stdout, "\tUnaligned memory loadstore emulation: %s\n", UnalignedMemoryLoadStoreTSOEmulation);
|
||||
fprintf(stdout, "\t16-Byte split-lock atomic emulation: %s\n", TSOFacts.LSE ? "\e[31mTearing CAS loops\e[0m" : "\e[31mTearing LL/SC loops\e[0m");
|
||||
fprintf(stdout, "\t16-Byte split-lock atomic emulation: %s\n", SplitLock16BEmulationType);
|
||||
fprintf(stdout, "\t64-Byte split-lock atomic emulation: %s\n", TSOFacts.LSE ? "\e[31mTearing CAS loops\e[0m" : "\e[31mTearing LL/SC loops\e[0m");
|
||||
fprintf(stdout, "\tGPR memory model emulation: %s\n", GPRMemoryTSOEmulation);
|
||||
fprintf(stdout, "\tMemcpy memory model emulation: %s\n", MemcpyMemoryTSOEmulation);
|
||||
fprintf(stdout, "\tVector memory model emulation: %s\n", VectorMemoryTSOEmulation);
|
||||
|
||||
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
|
||||
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
|
||||
FEX_CONFIG_OPT(VectorTSOEnabled, VECTORTSOENABLED);
|
||||
FEX_CONFIG_OPT(HalfBarrierTSOEnabled, HALFBARRIERTSOENABLED);
|
||||
FEX_CONFIG_OPT(StrictInProcessSplitLocks, STRICTINPROCESSSPLITLOCKS);
|
||||
fprintf(stderr, "Strict: %d\n", StrictInProcessSplitLocks());
|
||||
|
||||
fprintf(stdout, "\nConfiguration:\n");
|
||||
fprintf(stdout, "\tTSO Emulation: %s\n", TSOEnabled() ? "Enabled" : "Disabled");
|
||||
fprintf(stdout, "\tMemcpy TSO Emulation: %s\n", TSOEnabled() && MemcpySetTSOEnabled() ? "Enabled" : "Disabled");
|
||||
fprintf(stdout, "\tVector TSO Emulation: %s\n", TSOEnabled() && VectorTSOEnabled() ? "Enabled" : "Disabled");
|
||||
fprintf(stdout, "\tHalf-barrier unaligned TSO emulation: %s\n", TSOEnabled() && HalfBarrierTSOEnabled() ? "Enabled" : "Disabled");
|
||||
fprintf(stdout, "\t16-Byte strict split-lock emulation: %s\n", StrictInProcessSplitLocks() ? "In-process mutex" : "Tearing");
|
||||
fprintf(stdout, "\t16-Byte strict split-lock emulation: %s\n", SplitLock16BConfigurationType);
|
||||
fprintf(stdout, "\t64-Byte strict split-lock emulation: %s\n", StrictInProcessSplitLocks() ? "In-process mutex" : "Tearing");
|
||||
}
|
||||
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
if (Options.is_set_by_user("identification_reg_info")) {
|
||||
auto Features = FEX::GetCPUFeaturesFromIDRegisters();
|
||||
fextl::string features {};
|
||||
|
||||
@@ -645,6 +645,11 @@ public:
|
||||
|
||||
if (Is64BitMode()) {
|
||||
AuxVariables.emplace_back(auxv_t {4, 0x38}); // AT_PHENT
|
||||
|
||||
// 64-bit vsyscall entry points are hardcoded to a single page at 0xffffffffff600000.
|
||||
// FEX can't actually map anything there so it is a hardcoded quirk, similar to how the kernel traps these executions.
|
||||
// Just track it as a mapped anonymous executable page.
|
||||
Handler->AddVirtualPage(Thread, 0xFFFFFFFFFF600000ULL, FEXCore::Utils::FEX_PAGE_SIZE, PROT_READ | PROT_EXEC);
|
||||
} else {
|
||||
AuxVariables.emplace_back(auxv_t {4, 0x20}); // AT_PHENT
|
||||
|
||||
|
||||
@@ -6,10 +6,20 @@ target_link_libraries(FEXOfflineCompiler PRIVATE
|
||||
cpp-optparse
|
||||
FEXCore
|
||||
JemallocLibs
|
||||
LinuxEmulation
|
||||
${PTHREAD_LIB}
|
||||
fmt::fmt)
|
||||
|
||||
if (MINGW)
|
||||
patch_library_wine(FEXOfflineCompiler)
|
||||
target_include_directories(FEXOfflineCompiler PRIVATE
|
||||
"${CMAKE_SOURCE_DIR}/Source/Windows/include/"
|
||||
"${CMAKE_SOURCE_DIR}/Source/"
|
||||
"${CMAKE_SOURCE_DIR}/Source/Windows/"
|
||||
)
|
||||
target_link_libraries(FEXOfflineCompiler PRIVATE CommonWindows ntdll_ex)
|
||||
else()
|
||||
target_link_libraries(FEXOfflineCompiler PRIVATE ${PTHREAD_LIB} LinuxEmulation)
|
||||
endif()
|
||||
|
||||
LinkerGC(FEXOfflineCompiler)
|
||||
|
||||
install(TARGETS FEXOfflineCompiler RUNTIME
|
||||
|
||||
@@ -1,13 +1,20 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#ifndef _WIN32
|
||||
#include "../FEXInterpreter/ELFCodeLoader.h"
|
||||
#endif
|
||||
#include <DummyHandlers.h>
|
||||
#ifndef _WIN32
|
||||
#include <PortabilityInfo.h>
|
||||
#include <Thunks.h>
|
||||
#endif
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CodeCache.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/HLE/SourcecodeResolver.h>
|
||||
|
||||
#include <Common/ArgumentLoader.h>
|
||||
#include <Common/Config.h>
|
||||
@@ -16,15 +23,42 @@
|
||||
|
||||
#include <OptionParser.h>
|
||||
|
||||
#ifndef _WIN32
|
||||
#include <elf.h>
|
||||
#endif
|
||||
#include <fmt/printf.h>
|
||||
#include <libgen.h>
|
||||
|
||||
#include <fcntl.h>
|
||||
#include <fstream>
|
||||
#include <optional>
|
||||
#include <ranges>
|
||||
|
||||
#ifndef _WIN32
|
||||
#include <sys/mman.h>
|
||||
#else
|
||||
#include <Common/CPUFeatures.h>
|
||||
#include <Common/Handle.h>
|
||||
#include <Common/ImageTracker.h>
|
||||
#include <Common/InvalidationTracker.h>
|
||||
#include <Common/JITGuardPage.h>
|
||||
#include <Common/Logging.h>
|
||||
#include <Common/Module.h>
|
||||
#include <Common/OvercommitTracker.h>
|
||||
#include <Common/PortabilityInfo.h>
|
||||
|
||||
static std::unique_ptr<FEX::Windows::OvercommitTracker> OvercommitTracker;
|
||||
#endif
|
||||
static FEXCore::Core::InternalThreadState* Thread = nullptr;
|
||||
|
||||
#ifdef _WIN32
|
||||
class AOTSyscallHandler : public FEXCore::HLE::SyscallHandler {
|
||||
#else
|
||||
class AOTSyscallHandler : public FEXCore::HLE::SyscallHandler, public FEX::HLE::SyscallMmapInterface {
|
||||
#endif
|
||||
public:
|
||||
AOTSyscallHandler(FEXCore::HLE::SyscallOSABI SyscallOSABI) {
|
||||
AOTSyscallHandler(FEXCore::Context::Context& CTX, FEXCore::HLE::SyscallOSABI SyscallOSABI)
|
||||
: CTX(CTX) {
|
||||
OSABI = SyscallOSABI;
|
||||
}
|
||||
|
||||
@@ -33,24 +67,39 @@ public:
|
||||
return 0;
|
||||
}
|
||||
|
||||
FEXCore::Context::Context& CTX;
|
||||
#ifdef _WIN32
|
||||
FEX::Windows::ImageTracker ImageTracker {CTX, true};
|
||||
const std::unordered_map<DWORD, FEXCore::Core::InternalThreadState*> ThreadsUnused;
|
||||
FEX::Windows::InvalidationTracker InvalidationTracker {CTX, ThreadsUnused};
|
||||
#else
|
||||
FEXCore::ExecutableFileInfo FileInfo;
|
||||
std::map<uint64_t, uint64_t> FileRanges;
|
||||
|
||||
#endif
|
||||
uintptr_t VAFileStart = 0;
|
||||
|
||||
// These are no-ops implementations of the SyscallHandler API
|
||||
std::optional<FEXCore::ExecutableFileSectionInfo> LookupExecutableFileSection(FEXCore::Core::InternalThreadState*, uint64_t Address) override {
|
||||
#ifndef _WIN32
|
||||
auto It = FileRanges.upper_bound(Address - VAFileStart);
|
||||
LOGMAN_THROW_A_FMT(It != FileRanges.begin(), "Could not find associated file mapping");
|
||||
--It;
|
||||
LOGMAN_THROW_A_FMT(VAFileStart + It->first + It->second > Address, "Could not find associated file mapping for {:#x}", Address);
|
||||
return FEXCore::ExecutableFileSectionInfo {FileInfo, VAFileStart, VAFileStart + It->first, VAFileStart + It->first + It->second};
|
||||
#else
|
||||
return ImageTracker.LookupExecutableFileSection(Address);
|
||||
#endif
|
||||
}
|
||||
|
||||
FEXCore::HLE::ExecutableRangeInfo QueryGuestExecutableRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Address) override {
|
||||
#ifndef _WIN32
|
||||
return {0, UINT64_MAX, true};
|
||||
#else
|
||||
return InvalidationTracker.QueryExecutableRange(Address);
|
||||
#endif
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
void* GuestMmap(FEXCore::Core::InternalThreadState*, void* addr, size_t Size, int prot, int Flags, int fd, off_t offset) override {
|
||||
// Force writeable to allow applying relocations
|
||||
auto Ret = mmap(addr, Size, prot | PROT_WRITE, Flags, fd, offset);
|
||||
@@ -64,6 +113,19 @@ public:
|
||||
uint64_t GuestMunmap(FEXCore::Core::InternalThreadState*, void* addr, uint64_t length) override {
|
||||
return munmap(addr, length);
|
||||
}
|
||||
|
||||
void AddVirtualPage(FEXCore::Core::InternalThreadState* Thread, uint64_t addr, size_t length, int prot) override {
|
||||
LogMan::Msg::AFmt("Can't Track mmap through here");
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
#else
|
||||
void MarkOvercommitRange(uint64_t Start, uint64_t Length) override {
|
||||
OvercommitTracker->MarkRange(Start, Length);
|
||||
}
|
||||
void UnmarkOvercommitRange(uint64_t Start, uint64_t Length) override {
|
||||
OvercommitTracker->UnmarkRange(Start, Length);
|
||||
}
|
||||
#endif
|
||||
};
|
||||
|
||||
static void MsgHandler(LogMan::DebugLevels Level, const char* Message) {
|
||||
@@ -87,8 +149,18 @@ struct std::hash<FEXCore::ExecutableFileInfo> {
|
||||
}
|
||||
};
|
||||
|
||||
// Windows requires O_BINARY, whereas on Linux it's implicit
|
||||
#ifndef O_BINARY
|
||||
#define O_BINARY 0
|
||||
#endif
|
||||
|
||||
// Placeholder data to ensure the compile thread doesn't de-reference nullptr data
|
||||
static FEXCore::Core::CPUState::gdt_segment gdt[32] {};
|
||||
#if !defined(_WIN32) || defined(_M_ARM64EC)
|
||||
static constexpr size_t DefaultCS {FEXCore::Core::CPUState::DEFAULT_USER_CS};
|
||||
#else
|
||||
static constexpr size_t DefaultCS {4};
|
||||
#endif
|
||||
|
||||
static FEXCore::Core::InternalThreadState* SetupCompileThread(FEXCore::Context::Context& CTX, bool Is64Bit) {
|
||||
auto Thread = CTX.CreateThread(0, 0);
|
||||
@@ -96,13 +168,11 @@ static FEXCore::Core::InternalThreadState* SetupCompileThread(FEXCore::Context::
|
||||
auto Frame = Thread->CurrentFrame;
|
||||
Frame->State.segment_arrays[FEXCore::Core::CPUState::SEGMENT_ARRAY_INDEX_GDT] = &gdt[0];
|
||||
Frame->State.segment_arrays[FEXCore::Core::CPUState::SEGMENT_ARRAY_INDEX_LDT] = &gdt[0];
|
||||
|
||||
Frame->State.cs_idx = FEXCore::Core::CPUState::DEFAULT_USER_CS << 3;
|
||||
Frame->State.cs_idx = DefaultCS << 3;
|
||||
auto GDT = FEXCore::Core::CPUState::GetSegmentFromIndex(Frame->State, Frame->State.cs_idx);
|
||||
FEXCore::Core::CPUState::SetGDTBase(GDT, 0);
|
||||
FEXCore::Core::CPUState::SetGDTLimit(GDT, 0xFFFFFU);
|
||||
Frame->State.cs_cached =
|
||||
FEXCore::Core::CPUState::CalculateGDTBase(*FEXCore::Core::CPUState::GetSegmentFromIndex(Frame->State, Frame->State.cs_idx));
|
||||
Frame->State.cs_cached = FEXCore::Core::CPUState::CalculateGDTBase(*GDT);
|
||||
|
||||
if (Is64Bit) {
|
||||
GDT->L = 1; // L = Long Mode = 64-bit
|
||||
@@ -115,20 +185,256 @@ static FEXCore::Core::InternalThreadState* SetupCompileThread(FEXCore::Context::
|
||||
return Thread;
|
||||
}
|
||||
|
||||
// Returns filename of generated cache on success
|
||||
static std::optional<std::string> GenerateSingleCache(FEXCore::ExecutableFileInfo& Binary, fextl::set<uintptr_t> BlockList, std::string_view OutDir) {
|
||||
uint64_t CodeCacheConfigId = 0; // TODO: Make unique to active configuration
|
||||
#ifdef _WIN32
|
||||
static bool RelocateMappedImage(HMODULE Module) {
|
||||
const auto* NtHeaders = reinterpret_cast<FEX::Windows::ArchImageNtHeaders*>(RtlImageNtHeader(Module));
|
||||
if (!NtHeaders) {
|
||||
return false;
|
||||
}
|
||||
|
||||
const auto BaseAddress = reinterpret_cast<uintptr_t>(Module);
|
||||
const auto PreferredBase = NtHeaders->OptionalHeader.ImageBase;
|
||||
const auto Delta = static_cast<intptr_t>(BaseAddress - PreferredBase);
|
||||
|
||||
// Wine will automatically relocate all DLLs to their mapped address, but PE relocations must still be applied so
|
||||
// FEXCore can correctly transform them into FEX relocations
|
||||
if (Delta == 0) {
|
||||
return true;
|
||||
}
|
||||
|
||||
ULONG RelocSize = 0;
|
||||
auto* RelocBlock =
|
||||
reinterpret_cast<IMAGE_BASE_RELOCATION*>(RtlImageDirectoryEntryToData(Module, true, IMAGE_DIRECTORY_ENTRY_BASERELOC, &RelocSize));
|
||||
|
||||
if (!RelocBlock || RelocSize == 0) {
|
||||
return true;
|
||||
}
|
||||
|
||||
// Reprotect all sections as RW to apply relocations, saving their prior protections
|
||||
struct SectionPatchState {
|
||||
void* Address;
|
||||
SIZE_T Size;
|
||||
DWORD PreviousProtection;
|
||||
};
|
||||
std::vector<SectionPatchState> SectionStates;
|
||||
SectionStates.reserve(NtHeaders->FileHeader.NumberOfSections);
|
||||
|
||||
auto* SectionHeader = IMAGE_FIRST_SECTION(NtHeaders);
|
||||
const auto* SectionHeaderEnd = SectionHeader + NtHeaders->FileHeader.NumberOfSections;
|
||||
for (; SectionHeader != SectionHeaderEnd; ++SectionHeader) {
|
||||
if (SectionHeader->SizeOfRawData == 0) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const auto SecAddr = reinterpret_cast<void*>(BaseAddress + SectionHeader->VirtualAddress);
|
||||
const SIZE_T SecSize = SectionHeader->Misc.VirtualSize;
|
||||
|
||||
DWORD OldProt = 0;
|
||||
if (!VirtualProtect(SecAddr, SecSize, PAGE_READWRITE, &OldProt)) {
|
||||
for (const auto& State : SectionStates) {
|
||||
DWORD Ignored;
|
||||
VirtualProtect(State.Address, State.Size, State.PreviousProtection, &Ignored);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
SectionStates.push_back({SecAddr, SecSize, OldProt});
|
||||
}
|
||||
|
||||
// Apply relocations to all sections
|
||||
bool RelocSuccess = true;
|
||||
const uintptr_t RelocEnd = reinterpret_cast<uintptr_t>(RelocBlock) + RelocSize;
|
||||
const uint32_t ImageSize = NtHeaders->OptionalHeader.SizeOfImage;
|
||||
|
||||
while (reinterpret_cast<uintptr_t>(RelocBlock) < RelocEnd && RelocBlock->SizeOfBlock) {
|
||||
if (RelocBlock->VirtualAddress >= ImageSize) {
|
||||
RelocSuccess = false;
|
||||
break;
|
||||
}
|
||||
|
||||
const auto Count = (RelocBlock->SizeOfBlock - sizeof(IMAGE_BASE_RELOCATION)) / sizeof(USHORT);
|
||||
const auto PageAddress = BaseAddress + RelocBlock->VirtualAddress;
|
||||
|
||||
RelocBlock = LdrProcessRelocationBlock(PageAddress, Count, reinterpret_cast<USHORT*>(RelocBlock + 1), Delta);
|
||||
|
||||
if (!RelocBlock) {
|
||||
RelocSuccess = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Restore sections to previous protection states
|
||||
for (const auto& State : SectionStates) {
|
||||
DWORD Ignored;
|
||||
VirtualProtect(State.Address, State.Size, State.PreviousProtection, &Ignored);
|
||||
}
|
||||
|
||||
if (!RelocSuccess) {
|
||||
return false;
|
||||
}
|
||||
|
||||
LogMan::Msg::IFmt("Relocated image {:X} -> {:X}", PreferredBase, BaseAddress);
|
||||
return true;
|
||||
}
|
||||
|
||||
#ifdef ARCHITECTURE_arm64ec
|
||||
static void* MapView(HANDLE SectionHandle) {
|
||||
return MapViewOfFile(SectionHandle, FILE_MAP_EXECUTE | FILE_MAP_READ, 0, 0, 0);
|
||||
}
|
||||
#else
|
||||
static void* MapView(HANDLE SectionHandle) {
|
||||
void* BaseAddress = nullptr;
|
||||
SIZE_T ViewSize = 0;
|
||||
LARGE_INTEGER Offset {};
|
||||
|
||||
// Map images in the lower 32-bits for WOW64 so relocations can be correctly applied
|
||||
const ULONG_PTR ZeroBits = 0x7fffffff;
|
||||
|
||||
NTSTATUS Status =
|
||||
NtMapViewOfSection(SectionHandle, GetCurrentProcess(), &BaseAddress, ZeroBits, 0, &Offset, &ViewSize, ViewShare, 0, PAGE_EXECUTE_READ);
|
||||
|
||||
if (Status < 0) {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
return BaseAddress;
|
||||
}
|
||||
#endif
|
||||
|
||||
// Returns the base address of the mapped image
|
||||
static std::optional<uint64_t> TryMapImage(FEX::Windows::InvalidationTracker& InvalidationTracker, FEX::Windows::ImageTracker& ImageTracker,
|
||||
const FEXCore::CodeMapFileId& ID, FEXCore::ExecutableFileInfo& Info) {
|
||||
{
|
||||
FEX::Windows::ScopedHandle File {CreateFileA(Info.Filename.c_str(), GENERIC_READ | SYNCHRONIZE, FILE_SHARE_READ | FILE_SHARE_DELETE,
|
||||
nullptr, OPEN_EXISTING, FILE_ATTRIBUTE_NORMAL, nullptr)};
|
||||
if (!File) {
|
||||
LogMan::Msg::EFmt("Couldn't find image: {}", Info.Filename);
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
FEX::Windows::ScopedHandle Section {CreateFileMappingA(*File, nullptr, SEC_IMAGE | PAGE_EXECUTE_READ, 0, 0, nullptr)};
|
||||
if (!Section) {
|
||||
LogMan::Msg::EFmt("Couldn't create section for image: {}", Info.Filename);
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
void* Mapping = MapView(*Section);
|
||||
if (!Mapping) {
|
||||
LogMan::Msg::EFmt("Couldn't map section for image: {}", Info.Filename);
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
if (!RelocateMappedImage(reinterpret_cast<HMODULE>(Mapping))) {
|
||||
LogMan::Msg::EFmt("Failed to apply image relocations");
|
||||
UnmapViewOfFile(Mapping);
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
uint64_t BaseAddress = reinterpret_cast<uint64_t>(Mapping);
|
||||
LogMan::Msg::IFmt("Mapped image: {} @ {:X}", Info.Filename, BaseAddress);
|
||||
|
||||
InvalidationTracker.HandleImageMap(FEX::Windows::BaseName(Info.Filename), BaseAddress);
|
||||
ImageTracker.HandleImageMap(Info.Filename, BaseAddress, false /* unused during cache generation */);
|
||||
|
||||
return BaseAddress;
|
||||
}
|
||||
}
|
||||
|
||||
static LONG ExceptionHandler(_EXCEPTION_POINTERS* ExceptionInfo) {
|
||||
if (ExceptionInfo->ExceptionRecord->ExceptionCode == EXCEPTION_ACCESS_VIOLATION) {
|
||||
const auto FaultAddress = static_cast<uint64_t>(ExceptionInfo->ExceptionRecord->ExceptionInformation[1]);
|
||||
if (OvercommitTracker->HandleAccessViolation(FaultAddress)) {
|
||||
return EXCEPTION_CONTINUE_EXECUTION;
|
||||
}
|
||||
|
||||
#ifdef ARCHITECTURE_arm64ec
|
||||
ARM64_NT_CONTEXT ArmContext {};
|
||||
auto* Context = &ArmContext;
|
||||
#else
|
||||
auto* Context = ExceptionInfo->ContextRecord;
|
||||
#endif
|
||||
if (FEX::Windows::JITGuardPage::HandleJITGuardPage(Thread, reinterpret_cast<void*>(FaultAddress), Context->X,
|
||||
reinterpret_cast<__uint128_t*>(Context->V), &Context->Pc)) {
|
||||
#ifdef ARCHITECTURE_arm64ec
|
||||
auto* ECContext = reinterpret_cast<ARM64EC_NT_CONTEXT*>(ExceptionInfo->ContextRecord);
|
||||
ECContext->X0 = Context->X0;
|
||||
ECContext->X19 = Context->X19;
|
||||
ECContext->X20 = Context->X20;
|
||||
ECContext->X21 = Context->X21;
|
||||
ECContext->X22 = Context->X22;
|
||||
ECContext->X25 = Context->X25;
|
||||
ECContext->X26 = Context->X26;
|
||||
ECContext->X27 = Context->X27;
|
||||
ECContext->Fp = Context->Fp;
|
||||
ECContext->Lr = Context->Lr;
|
||||
ECContext->Sp = Context->Sp;
|
||||
ECContext->Pc = Context->Pc;
|
||||
|
||||
for (size_t i = 0; i < 8; ++i) {
|
||||
memcpy(&reinterpret_cast<__uint128_t*>(ECContext->V)[8 + i], &reinterpret_cast<__uint128_t*>(Context->V)[8 + i], sizeof(uint64_t));
|
||||
}
|
||||
#endif
|
||||
return EXCEPTION_CONTINUE_EXECUTION;
|
||||
}
|
||||
}
|
||||
return EXCEPTION_CONTINUE_SEARCH;
|
||||
}
|
||||
|
||||
struct winsize {
|
||||
int ws_col;
|
||||
};
|
||||
#endif
|
||||
|
||||
// Returns filename of generated cache on success
|
||||
static std::optional<std::string> GenerateSingleCache(FEXCore::ExecutableFileInfo& Binary, uint64_t CodeCacheConfigId,
|
||||
fextl::set<uintptr_t> BlockList, std::string_view OutDir) {
|
||||
#ifndef _WIN32
|
||||
ELFCodeLoader Loader(Binary.Filename.c_str(), -1, "", fextl::vector<fextl::string> {Binary.Filename.c_str()},
|
||||
fextl::vector<fextl::string> {}, nullptr, nullptr, true /* skip interpreter */);
|
||||
if (!Loader.ELFWasLoaded()) {
|
||||
fmt::print("Invalid or unsupported ELF file.\n");
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
const bool Is64Bit = Loader.Is64BitMode();
|
||||
#elif defined(_M_ARM64EC)
|
||||
const bool Is64Bit = true;
|
||||
#else
|
||||
const bool Is64Bit = false;
|
||||
#endif
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_IS64BIT_MODE, Is64Bit ? "1" : "0");
|
||||
|
||||
// Load HostFeatures
|
||||
#ifndef _WIN32
|
||||
auto HostFeatures = FEX::FetchHostFeatures();
|
||||
#else
|
||||
const auto NtDll = GetModuleHandle("ntdll.dll");
|
||||
const bool IsWine = !!GetProcAddress(NtDll, "wine_get_version");
|
||||
auto HostFeatures = FEX::Windows::CPUFeatures::FetchHostFeatures(
|
||||
IsWine, Is64Bit ? FEXCore::HostFeatures::HostTypeEnum::Arm64ec : FEXCore::HostFeatures::HostTypeEnum::Wow64);
|
||||
#endif
|
||||
|
||||
auto CTX = FEXCore::Context::Context::CreateNewContext(HostFeatures);
|
||||
CTX->GetCodeCache().InitiateCacheGeneration();
|
||||
|
||||
#ifdef _WIN32
|
||||
OvercommitTracker = std::make_unique<FEX::Windows::OvercommitTracker>(IsWine);
|
||||
|
||||
auto SyscallOSABI = Is64Bit ? FEXCore::HLE::SyscallOSABI::OS_LINUX64 : FEXCore::HLE::SyscallOSABI::OS_LINUX32;
|
||||
auto SyscallHandler = std::make_unique<AOTSyscallHandler>(SyscallOSABI);
|
||||
auto SyscallHandler = std::make_unique<AOTSyscallHandler>(*CTX, SyscallOSABI);
|
||||
|
||||
SyscallHandler->VAFileStart =
|
||||
TryMapImage(SyscallHandler->InvalidationTracker, SyscallHandler->ImageTracker, Binary.FileId, Binary).value_or(0);
|
||||
if (!SyscallHandler->VAFileStart) {
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
// Register exception handler for OvercommitTracker
|
||||
AddVectoredExceptionHandler(1, ExceptionHandler);
|
||||
#else
|
||||
Loader.CalculateHWCaps(CTX.get());
|
||||
|
||||
auto SyscallOSABI = Is64Bit ? FEXCore::HLE::SyscallOSABI::OS_LINUX64 : FEXCore::HLE::SyscallOSABI::OS_LINUX32;
|
||||
auto SyscallHandler = std::make_unique<AOTSyscallHandler>(*CTX, SyscallOSABI);
|
||||
|
||||
// Populate relocations from ELF file
|
||||
{
|
||||
@@ -138,10 +444,12 @@ static std::optional<std::string> GenerateSingleCache(FEXCore::ExecutableFileInf
|
||||
SyscallHandler->FileInfo.Relocations = Binary.Relocations;
|
||||
}
|
||||
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_IS64BIT_MODE, Is64Bit ? "1" : "0");
|
||||
|
||||
// Load HostFeatures
|
||||
auto HostFeatures = FEX::FetchHostFeatures();
|
||||
if (!Is64Bit) {
|
||||
const auto PageSize = sysconf(_SC_PAGESIZE);
|
||||
// Block upper address space
|
||||
FEXCore::Allocator::SetupHooks(PageSize > 0 ? PageSize : FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
}
|
||||
#endif
|
||||
|
||||
if (!std::filesystem::exists(Binary.Filename)) {
|
||||
fmt::print("File {} does not exist\n", Binary.Filename);
|
||||
@@ -149,28 +457,21 @@ static std::optional<std::string> GenerateSingleCache(FEXCore::ExecutableFileInf
|
||||
return /*EXIT_FAILURE*/ std::nullopt;
|
||||
}
|
||||
|
||||
auto CTX = FEXCore::Context::Context::CreateNewContext(HostFeatures);
|
||||
|
||||
Loader.CalculateHWCaps(CTX.get());
|
||||
|
||||
auto SignalDelegation = std::make_unique<FEX::DummyHandlers::DummySignalDelegator>();
|
||||
CTX->SetSignalDelegator(SignalDelegation.get());
|
||||
CTX->SetSyscallHandler(SyscallHandler.get());
|
||||
#ifndef _WIN32
|
||||
auto ThunkHandler = FEX::HLE::CreateThunkHandler();
|
||||
CTX->SetThunkHandler(ThunkHandler.get());
|
||||
#endif
|
||||
|
||||
if (!CTX->InitCore()) {
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
if (!Is64Bit) {
|
||||
const auto PageSize = sysconf(_SC_PAGESIZE);
|
||||
// Block upper address space
|
||||
FEXCore::Allocator::SetupHooks(PageSize > 0 ? PageSize : FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
}
|
||||
|
||||
auto Thread = SetupCompileThread(*CTX, Is64Bit);
|
||||
Thread = SetupCompileThread(*CTX, Is64Bit);
|
||||
|
||||
#ifndef _WIN32
|
||||
{
|
||||
auto ElfBase = Loader.LoadMainElfFile(nullptr, SyscallHandler.get(), Thread);
|
||||
if (!ElfBase.has_value()) {
|
||||
@@ -196,12 +497,9 @@ static std::optional<std::string> GenerateSingleCache(FEXCore::ExecutableFileInf
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
CTX->GetCodeCache().InitiateCacheGeneration();
|
||||
#endif
|
||||
|
||||
{
|
||||
std::vector<std::unique_ptr<ELFCodeLoader>> LoaderMem;
|
||||
|
||||
// Refuse to continue if the block list contains any out-of-bounds blocks.
|
||||
// This often indicates a corrupted code map.
|
||||
{
|
||||
@@ -225,13 +523,17 @@ static std::optional<std::string> GenerateSingleCache(FEXCore::ExecutableFileInf
|
||||
|
||||
auto Filename = fmt::format("{}{}-{:016x}", OutDir, FEXCore::CodeMap::GetBaseFilename(Binary, false), CodeCacheConfigId);
|
||||
auto FilenameNew = Filename + ".new";
|
||||
int fd = open(FilenameNew.c_str(), O_CREAT | O_WRONLY, 0644);
|
||||
int fd = open(FilenameNew.c_str(), O_CREAT | O_WRONLY | O_BINARY, 0644);
|
||||
{
|
||||
auto Entry = SyscallHandler->LookupExecutableFileSection(Thread, SyscallHandler->VAFileStart).value();
|
||||
#ifndef _WIN32
|
||||
CTX->GetCodeCache().SaveData(*Thread, fd, Entry, 0 /* TODO: Use static base address information if available */);
|
||||
#else
|
||||
CTX->GetCodeCache().SaveData(*Thread, fd, Entry, SyscallHandler->VAFileStart);
|
||||
#endif
|
||||
}
|
||||
std::filesystem::rename(FilenameNew.c_str(), Filename.c_str());
|
||||
close(fd);
|
||||
std::filesystem::rename(FilenameNew.c_str(), Filename.c_str());
|
||||
return Filename;
|
||||
}
|
||||
}
|
||||
@@ -303,10 +605,11 @@ static int GenerateCache(int argc, const char** argv) {
|
||||
|
||||
const auto PortableInfo = FEX::ReadPortabilityInformation();
|
||||
char* envp[] = {nullptr};
|
||||
FEXCore::Config::Shutdown();
|
||||
FEX::Config::LoadConfig("", envp, PortableInfo);
|
||||
|
||||
auto NumBlocks = Data.at(ProgramName).size();
|
||||
auto GeneratedCache = GenerateSingleCache(ProgramName, Data.at(ProgramName), OutDir);
|
||||
auto GeneratedCache = GenerateSingleCache(ProgramName, 0 /* TODO: Config id */, Data.at(ProgramName), OutDir);
|
||||
if (GeneratedCache) {
|
||||
fmt::print("Successfully populated cache {} ({} blocks) via {}\n\n", GeneratedCache.value(), NumBlocks,
|
||||
std::filesystem::path {CodeMapPath}.filename().string());
|
||||
@@ -314,20 +617,229 @@ static int GenerateCache(int argc, const char** argv) {
|
||||
return GeneratedCache ? 0 : 1;
|
||||
}
|
||||
|
||||
/**
|
||||
* Writes aggregated code map data into a single code map file that is ready to be used for cache generation
|
||||
*/
|
||||
static void WriteNewCodeMap(const FEXCore::ExecutableFileInfo& File, const std::string& OutputName, const fextl::set<uintptr_t>& Blocks,
|
||||
bool IsExecutable, const std::set<FEXCore::ExecutableFileInfo>& Dependencies) {
|
||||
fmt::print("Writing {} blocks to {}\n", Blocks.size(), OutputName);
|
||||
|
||||
struct CodeMapOpener : FEXCore::CodeMapOpener {
|
||||
CodeMapOpener(const std::string& Filename) {
|
||||
FD = open(Filename.c_str(), O_CREAT | O_TRUNC | O_WRONLY | O_BINARY, 0644);
|
||||
}
|
||||
|
||||
int OpenCodeMapFile() override {
|
||||
return FD;
|
||||
}
|
||||
|
||||
int FD;
|
||||
};
|
||||
|
||||
CodeMapOpener CodeMapOpener(OutputName);
|
||||
FEXCore::CodeMapWriter OutputCodeMap(CodeMapOpener, true);
|
||||
if (IsExecutable) {
|
||||
// List the main executable and all used libraries
|
||||
OutputCodeMap.AppendSetMainExecutable(File);
|
||||
|
||||
for (auto& Dependency : Dependencies) {
|
||||
OutputCodeMap.AppendLibraryLoad(Dependency);
|
||||
}
|
||||
} else {
|
||||
// List only the library itself
|
||||
OutputCodeMap.AppendLibraryLoad(File);
|
||||
}
|
||||
|
||||
for (auto& Block : Blocks) {
|
||||
OutputCodeMap.AppendBlock(FEXCore::ExecutableFileSectionInfo {File, 0}, Block);
|
||||
}
|
||||
}
|
||||
|
||||
struct ParsedContentsAndDependencies {
|
||||
fextl::string Filename;
|
||||
fextl::set<uint64_t> Blocks;
|
||||
bool IsExecutable = false;
|
||||
std::set<FEXCore::CodeMapFileId> Dependencies;
|
||||
};
|
||||
|
||||
/**
|
||||
* Discovers any pending code maps, parses their contents into a runtime data structure, and deletes them
|
||||
*/
|
||||
static std::map<FEXCore::CodeMapFileId, ParsedContentsAndDependencies> ImportPendingCodeMaps(const std::string& NewCodeMapDirectory) {
|
||||
// TODO: Handle nomb code maps
|
||||
std::map<FEXCore::CodeMapFileId, ParsedContentsAndDependencies> Result;
|
||||
for (auto& Entry : std::filesystem::directory_iterator(NewCodeMapDirectory)) {
|
||||
if (!Entry.is_regular_file()) {
|
||||
continue;
|
||||
}
|
||||
const auto Name = Entry.path().filename().string();
|
||||
if (!Name.ends_with(".bin")) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if (std::filesystem::file_size(Entry.path()) == 0) {
|
||||
fmt::println("Found zero-size code map {}, deleting", Name);
|
||||
std::filesystem::remove(Entry.path());
|
||||
continue;
|
||||
}
|
||||
|
||||
fmt::print("Importing new code map {}\n", Name);
|
||||
std::ifstream Incoming(Entry.path(), std::ios_base::binary);
|
||||
std::set<FEXCore::CodeMapFileId> Dependencies;
|
||||
std::optional<FEXCore::CodeMapFileId> ExecutableFileId;
|
||||
for (auto& [FileId, Contents] : FEXCore::CodeMap::ParseCodeMap(Incoming)) {
|
||||
auto& [Filename, Blocks, IsExecutable, _] =
|
||||
Result.emplace(std::piecewise_construct, std::forward_as_tuple(FileId), std::tuple {}).first->second;
|
||||
Filename = std::move(Contents.Filename);
|
||||
Blocks.merge(std::move(Contents.Blocks));
|
||||
IsExecutable = Contents.IsExecutable;
|
||||
if (IsExecutable) {
|
||||
LOGMAN_THROW_A_FMT(!ExecutableFileId, "Expected a unique executable identifiers per code map");
|
||||
ExecutableFileId = FileId;
|
||||
} else {
|
||||
Dependencies.insert(FileId);
|
||||
}
|
||||
}
|
||||
|
||||
// Every imported code map should have had exactly one executable marker
|
||||
LOGMAN_THROW_A_FMT(ExecutableFileId, "Could not find an executable identifer in the code map");
|
||||
Result.at(*ExecutableFileId).Dependencies = std::move(Dependencies);
|
||||
|
||||
// Delete imported code map
|
||||
Incoming.close();
|
||||
std::filesystem::remove(Entry.path());
|
||||
}
|
||||
|
||||
return Result;
|
||||
}
|
||||
|
||||
/**
|
||||
* Checks and processes new code maps generated by FEX
|
||||
*
|
||||
* Processed code maps are merged into the reference ("ready") code maps
|
||||
*/
|
||||
static void AggregateCodeMaps(const std::string& NewCodeMapDirectory, const std::string& ReadyCodeMapDirectory) {
|
||||
auto IncomingCodeMap = ImportPendingCodeMaps(NewCodeMapDirectory);
|
||||
|
||||
for (auto& [FileId, Contents] : IncomingCodeMap) {
|
||||
// For each referenced binary, add the newly referenced offsets to that binary's reference code map
|
||||
const FEXCore::ExecutableFileInfo File {nullptr, FileId, Contents.Filename};
|
||||
const auto BinaryName = std::string {FEXCore::CodeMap::GetBaseFilename(File, false)};
|
||||
auto OutputName = fmt::format("{}/{}", ReadyCodeMapDirectory, BinaryName);
|
||||
|
||||
if (auto ReferenceCodeMap = std::ifstream(OutputName, std::ios_base::binary)) {
|
||||
auto PreviousBlocks = FEXCore::CodeMap::ParseCodeMap(ReferenceCodeMap).at(File.FileId).Blocks;
|
||||
auto NumPreviousBlocks = PreviousBlocks.size();
|
||||
Contents.Blocks.merge(std::move(PreviousBlocks));
|
||||
if (Contents.Blocks.size() == NumPreviousBlocks) {
|
||||
// No new blocks => skip updating
|
||||
continue;
|
||||
} else {
|
||||
fmt::println(" Found {} new blocks ({} total) in code map {} for {}", Contents.Blocks.size() - NumPreviousBlocks,
|
||||
Contents.Blocks.size(), BinaryName, File.Filename);
|
||||
}
|
||||
}
|
||||
|
||||
// Update code map
|
||||
std::set<FEXCore::ExecutableFileInfo> Dependencies;
|
||||
for (auto& Dependency : Contents.Dependencies) {
|
||||
Dependencies.emplace(nullptr, Dependency, IncomingCodeMap.at(Dependency).Filename);
|
||||
}
|
||||
WriteNewCodeMap(File, OutputName, Contents.Blocks, Contents.IsExecutable, Dependencies);
|
||||
}
|
||||
}
|
||||
|
||||
static int ProcessAll() {
|
||||
const auto CacheDirectory = FEX::Config::GetCacheDirectory();
|
||||
const std::string NewCodeMapDirectory = fmt::format("{}codemap/new", CacheDirectory);
|
||||
const std::string ReadyCodeMapDirectory = fmt::format("{}codemap/ready", CacheDirectory);
|
||||
|
||||
// Import new code maps and aggregate them into ready code maps
|
||||
std::filesystem::create_directories(ReadyCodeMapDirectory);
|
||||
AggregateCodeMaps(NewCodeMapDirectory, ReadyCodeMapDirectory);
|
||||
|
||||
// Generate caches
|
||||
fextl::string OutDir = CacheDirectory + "cache/";
|
||||
std::filesystem::create_directories(OutDir);
|
||||
|
||||
// Iterate over all executables (.exe).
|
||||
// These determine the emulator configuration to use when compiling dependencies.
|
||||
for (auto& Entry : std::filesystem::directory_iterator(ReadyCodeMapDirectory)) {
|
||||
std::ifstream CodeMap(Entry.path(), std::ios_base::binary);
|
||||
auto Parsed = FEXCore::CodeMap::ParseCodeMap(CodeMap);
|
||||
auto ExecutableIt = std::ranges::find_if(Parsed, [](const auto& Entry) { return Entry.second.IsExecutable; });
|
||||
if (ExecutableIt == Parsed.end()) {
|
||||
// Skip libraries; they're only processed as dependencies of a main executable
|
||||
continue;
|
||||
}
|
||||
|
||||
fmt::println("\nChecking caches for executable {}", ExecutableIt->second.Filename);
|
||||
|
||||
// TODO: Compute the cache config id from the active FEX configuration
|
||||
uint64_t CodeCacheConfigId = 0;
|
||||
|
||||
auto GetCacheFilename = [&](const FEXCore::ExecutableFileInfo& File) {
|
||||
return fmt::format("{}{}-{:016x}", OutDir, FEXCore::CodeMap::GetBaseFilename(File, false), CodeCacheConfigId);
|
||||
};
|
||||
|
||||
// Check the main binary and all of its dependencies
|
||||
for (auto& [FileId, Contents] : Parsed) {
|
||||
const FEXCore::ExecutableFileInfo File {nullptr, FileId, Contents.Filename};
|
||||
std::error_code ec;
|
||||
const auto BinaryName = FEXCore::CodeMap::GetBaseFilename(File, false);
|
||||
const auto MergedCodeMapFilename = fmt::format("{}/{}", ReadyCodeMapDirectory, BinaryName);
|
||||
const auto LastCodeMapUpdate = std::filesystem::last_write_time(MergedCodeMapFilename, ec);
|
||||
if (ec) {
|
||||
// No reference code map exists for this dependency yet, so there's nothing to generate a cache from
|
||||
continue;
|
||||
}
|
||||
|
||||
if (std::filesystem::last_write_time(GetCacheFilename(File), ec) > LastCodeMapUpdate && !ec) {
|
||||
fmt::println(" Cache up to date: {}", BinaryName);
|
||||
continue;
|
||||
}
|
||||
|
||||
// TODO: Also check for matching FEX version from cache header
|
||||
|
||||
fmt::println(" {} cache: {}", ec ? "Generating" : "Updating outdated", BinaryName);
|
||||
|
||||
// Defer to GenerateCache
|
||||
const auto FileIdArg = fmt::format("{:016x}", FileId);
|
||||
std::vector<const char*> GenerateArgs {
|
||||
"generate", "--fileid", FileIdArg.c_str(), "--outdir", OutDir.c_str(), MergedCodeMapFilename.c_str(),
|
||||
};
|
||||
if (GenerateCache(GenerateArgs.size(), GenerateArgs.data()) != 0) {
|
||||
fmt::println("ERROR: Cache generation failed for {}", BinaryName);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
int main(int argc, char** argv) {
|
||||
#ifndef _WIN32
|
||||
LogMan::Throw::InstallHandler(AssertHandler);
|
||||
LogMan::Msg::InstallHandler(MsgHandler);
|
||||
#else
|
||||
FEX::Windows::Logging::Init();
|
||||
#endif
|
||||
|
||||
std::vector<const char*> Args {argv + 1, argv + argc};
|
||||
auto CommandName = std::string {basename(argv[0])} + " " + (argc > 1 ? argv[1] : "");
|
||||
Args[0] = CommandName.c_str();
|
||||
if (!Args.empty()) {
|
||||
Args[0] = CommandName.c_str();
|
||||
}
|
||||
|
||||
if (argc >= 2 && argv[1] == std::string_view {"generate"}) {
|
||||
return GenerateCache(argc - 1, Args.data());
|
||||
} else if (argc >= 2 && argv[1] == std::string_view {"process-all"}) {
|
||||
return ProcessAll();
|
||||
} else {
|
||||
fmt::print("Usage: {} <command>\n\n", basename(argv[0]));
|
||||
fmt::print("Commands:\n");
|
||||
fmt::print(" generate\tTrigger cache generation from combined code map\n");
|
||||
fmt::print(" process-all\tProcess all new code maps and update all caches\n");
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
}
|
||||
@@ -228,7 +228,12 @@ bool InitializeServerSocket(bool abstract) {
|
||||
if (abstract) {
|
||||
ServerSocketName = FEXServerClient::GetServerSocketName();
|
||||
} else {
|
||||
ServerSocketName = FEXServerClient::GetServerSocketPath();
|
||||
ServerSocketName = FEXServerClient::GetServerSocketPath(false);
|
||||
if (ServerSocketName.size() > sizeof(sockaddr_un::sun_path) - 1) {
|
||||
LogMan::Msg::EFmt("Socket path '{}' too large for Unix domain sockets. Moving to tmp", ServerSocketName);
|
||||
ServerSocketName = FEXServerClient::GetServerSocketPath(true);
|
||||
}
|
||||
|
||||
// Unlink the socket file if it exists
|
||||
// We are being asked to create a daemon, not error check
|
||||
// We don't care if this failed or not
|
||||
|
||||
@@ -60,6 +60,7 @@ $end_info$
|
||||
#include <syscall.h>
|
||||
#include <sys/mman.h>
|
||||
#include <sys/utsname.h>
|
||||
#include <thread>
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEX::HLE {
|
||||
@@ -899,9 +900,19 @@ uint64_t UnimplementedSyscallSafe(FEXCore::Core::CpuStateFrame* Frame, uint64_t
|
||||
}
|
||||
|
||||
void SyscallHandler::LockBeforeFork(FEXCore::Core::InternalThreadState* Thread) {
|
||||
TM.LockBeforeFork();
|
||||
Thread->CTX->LockBeforeFork(Thread);
|
||||
VMATracking.Mutex.lock();
|
||||
while (true) {
|
||||
TM.LockBeforeFork();
|
||||
Thread->CTX->LockBeforeFork(Thread);
|
||||
if (std::try_lock(CodeCachePatchingMutex, VMATracking.Mutex) == -1) {
|
||||
break;
|
||||
}
|
||||
|
||||
// Lock failed: Another thread has temporarily acquired these mutexes.
|
||||
// Release them to a void a deadlock and retry later
|
||||
CTX->UnlockAfterFork(Thread, false);
|
||||
TM.UnlockAfterFork(Thread, false);
|
||||
std::this_thread::sleep_for(std::chrono::milliseconds {10});
|
||||
};
|
||||
}
|
||||
|
||||
void SyscallHandler::UnlockAfterFork(FEXCore::Core::InternalThreadState* LiveThread, bool Child) {
|
||||
@@ -910,8 +921,10 @@ void SyscallHandler::UnlockAfterFork(FEXCore::Core::InternalThreadState* LiveThr
|
||||
FM.SetProtectedCodeMapFD(-1);
|
||||
|
||||
VMATracking.Mutex.StealAndDropActiveLocks();
|
||||
CodeCachePatchingMutex.StealAndDropActiveLocks();
|
||||
} else {
|
||||
VMATracking.Mutex.unlock();
|
||||
CodeCachePatchingMutex.unlock();
|
||||
}
|
||||
|
||||
CTX->UnlockAfterFork(LiveThread, Child);
|
||||
|
||||
@@ -108,6 +108,8 @@ public:
|
||||
|
||||
// does a guest munmap as if done via a guest syscall
|
||||
virtual uint64_t GuestMunmap(FEXCore::Core::InternalThreadState* Thread, void* addr, uint64_t length) = 0;
|
||||
|
||||
virtual void AddVirtualPage(FEXCore::Core::InternalThreadState* Thread, uint64_t addr, size_t length, int prot) = 0;
|
||||
};
|
||||
|
||||
class SyscallHandler : public FEXCore::HLE::SyscallHandler,
|
||||
@@ -254,6 +256,10 @@ public:
|
||||
std::optional<LateApplyExtendedVolatileMetadata> TrackMmap(FEXCore::Core::InternalThreadState* Thread, uint64_t addr, size_t length,
|
||||
int prot, int flags, int fd, off_t offset,
|
||||
std::optional<FEXCore::ExecutableFileSectionInfo>& CachedSection);
|
||||
|
||||
void AddVirtualPage(FEXCore::Core::InternalThreadState* Thread, uint64_t addr, size_t length, int prot) override;
|
||||
using SyscallMmapInterface::AddVirtualPage;
|
||||
|
||||
void TrackMunmap(FEXCore::Core::InternalThreadState* Thread, void* addr, size_t length);
|
||||
void TrackMremap(FEXCore::Core::InternalThreadState* Thread, uint64_t OldAddress, size_t OldSize, size_t NewSize, int flags, uint64_t NewAddress);
|
||||
void TrackShmat(FEXCore::Core::InternalThreadState* Thread, int shmid, uint64_t shmaddr, int shmflg, uint64_t Length);
|
||||
@@ -261,10 +267,14 @@ public:
|
||||
void TrackMprotect(FEXCore::Core::InternalThreadState* Thread, void* addr, size_t len, int prot);
|
||||
void TrackMadvise(FEXCore::Core::InternalThreadState* Thread, uintptr_t Base, uintptr_t Size, int advice);
|
||||
|
||||
void InvalidateCodeRangeIfNecessary(FEXCore::Core::InternalThreadState* Thread, uint64_t Base, uint64_t Length) {
|
||||
void InvalidateCodeRangeIfNecessary(FEXCore::Core::InternalThreadState* Thread, uint64_t Base, uint64_t Length, bool CheckPendingVMAResources) {
|
||||
if (SMCChecks != FEXCore::Config::CONFIG_SMC_NONE) {
|
||||
TM.InvalidateGuestCodeRange(Thread, Base, Length);
|
||||
}
|
||||
if (CheckPendingVMAResources && Thread) {
|
||||
auto lk = FEXCore::GuardSignalDeferringSection(VMATracking.Mutex, Thread);
|
||||
VMATracking.FlushPendingResourceDeletions();
|
||||
}
|
||||
}
|
||||
|
||||
void InvalidateCodeRangeIfNecessaryOnRemap(FEXCore::Core::InternalThreadState* Thread, uint64_t OldAddress, uint64_t NewAddress,
|
||||
@@ -359,6 +369,9 @@ private:
|
||||
|
||||
std::mutex FutexMutex;
|
||||
std::mutex SyscallMutex;
|
||||
// std::mutex CodeCachePatchingMutex;
|
||||
FEXCore::ForkableUniqueMutex CodeCachePatchingMutex;
|
||||
|
||||
FEX::CodeLoader* LocalLoader {};
|
||||
bool NeedToCheckXID {true};
|
||||
|
||||
|
||||
@@ -437,7 +437,6 @@ namespace x64 {
|
||||
REGISTER_SYSCALL_IMPL_X64(getsockopt, SyscallPassthrough5<SYSCALL_DEF(getsockopt)>);
|
||||
REGISTER_SYSCALL_IMPL_X64(wait4, SyscallPassthrough4<SYSCALL_DEF(wait4)>);
|
||||
REGISTER_SYSCALL_IMPL_X64(semop, SyscallPassthrough3<SYSCALL_DEF(semop)>);
|
||||
REGISTER_SYSCALL_IMPL_X64(gettimeofday, SyscallPassthrough2<SYSCALL_DEF(gettimeofday)>);
|
||||
REGISTER_SYSCALL_IMPL_X64(getrlimit, SyscallPassthrough2<SYSCALL_DEF(getrlimit)>);
|
||||
REGISTER_SYSCALL_IMPL_X64(getrusage, SyscallPassthrough2<SYSCALL_DEF(getrusage)>);
|
||||
REGISTER_SYSCALL_IMPL_X64(sysinfo, SyscallPassthrough1<SYSCALL_DEF(sysinfo)>);
|
||||
|
||||
@@ -30,7 +30,21 @@ $end_info$
|
||||
#include <Linux/Utils/ELFParser.h>
|
||||
|
||||
namespace FEX::HLE {
|
||||
// SMC interactions
|
||||
static void HandleSegfaultForCodeCacheFinalization(FEXCore::Core::InternalThreadState& Thread, FEXCore::MappedCodeCacheFile& Code,
|
||||
uintptr_t FaultAddress) {
|
||||
FEXCORE_PROFILE_SCOPED("Load code cache page");
|
||||
size_t PageIdx = (reinterpret_cast<std::byte*>(FaultAddress) - Code.CodeBuffer.data()) / FEXCore::Utils::FEX_PAGE_SIZE;
|
||||
|
||||
auto RangeToFinalize = Thread.CTX->GetCodeCache().SelectCodeRangeToFinalize(Code, PageIdx, PageIdx + 1);
|
||||
if (!RangeToFinalize.empty()) {
|
||||
Thread.CTX->GetCodeCache().FinalizeCodePages(Code, RangeToFinalize);
|
||||
}
|
||||
}
|
||||
|
||||
// Handles segfaults from:
|
||||
// - call-ret shadow stack overflow
|
||||
// - guest-side self-modifying code (SMC)
|
||||
// - lazy loading of mapped code cache pages
|
||||
bool SyscallHandler::HandleSegfault(FEXCore::Core::InternalThreadState* Thread, int Signal, void* info, void* ucontext) {
|
||||
const auto FaultAddress = (uintptr_t)((siginfo_t*)info)->si_addr;
|
||||
|
||||
@@ -46,13 +60,25 @@ bool SyscallHandler::HandleSegfault(FEXCore::Core::InternalThreadState* Thread,
|
||||
// Can't use the deferred signal lock in the SIGSEGV handler.
|
||||
auto lk = FEXCore::MaskSignalsAndLockMutex<std::shared_lock>(_SyscallHandler->VMATracking.Mutex);
|
||||
|
||||
auto VMATracking = &_SyscallHandler->VMATracking;
|
||||
auto& VMATracking = _SyscallHandler->VMATracking;
|
||||
|
||||
// If the write spans two pages, they will be flushed one at a time (generating two faults)
|
||||
auto Entry = VMATracking->FindVMAEntry(FaultAddress);
|
||||
auto Entry = VMATracking.FindVMAEntry(FaultAddress);
|
||||
|
||||
// If an untracked address, or the mapping wasn't writable, it can't be handled here
|
||||
if (Entry == VMATracking->VMAs.end() || !Entry->second.Prot.Writable) {
|
||||
if (Entry == VMATracking.VMAs.end()) {
|
||||
// Not a guest page; check mapped code cache pages
|
||||
auto* Code = VMATracking.FindMappedCodeCacheByHostAddress(FaultAddress);
|
||||
if (!Code) {
|
||||
// Untracked address; not handled here
|
||||
return false;
|
||||
}
|
||||
std::lock_guard lk(_SyscallHandler->CodeCachePatchingMutex);
|
||||
HandleSegfaultForCodeCacheFinalization(*Thread, *Code, FaultAddress);
|
||||
return true;
|
||||
}
|
||||
|
||||
// If the mapping wasn't writable, it can't be handled here
|
||||
if (!Entry->second.Prot.Writable) {
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -170,7 +196,7 @@ void SyscallHandler::MarkGuestExecutableRange(FEXCore::Core::InternalThreadState
|
||||
}
|
||||
|
||||
void SyscallHandler::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) {
|
||||
InvalidateCodeRangeIfNecessary(Thread, Start, Length);
|
||||
InvalidateCodeRangeIfNecessary(Thread, Start, Length, false);
|
||||
}
|
||||
|
||||
static FEXCore::ExecutableFileSectionInfo BuildSectionInfo(const VMATracking::MappedResource& Resource, uint64_t Base, uint64_t Size) {
|
||||
@@ -233,41 +259,56 @@ static ReadELFHeadersResult ReadELFHeaders(int FD, std::span<std::byte> HeaderDa
|
||||
return ReadELFHeadersResult {std::move(Parser.phdrs), std::move(Relocations), HasCodeRelocations};
|
||||
}
|
||||
|
||||
static void LoadCodeCache(FEXCore::Core::InternalThreadState& Thread, FEXCore::ExecutableFileSectionInfo& Section, uint64_t CodeCacheConfigId) {
|
||||
static fextl::unique_ptr<FEXCore::MappedCodeCacheFile>
|
||||
LoadCodeCache(FEXCore::Core::InternalThreadState& Thread, VMATracking::VMATracking& VMATracking,
|
||||
const FEXCore::ExecutableFileInfo& FileInfo, uint64_t CodeCacheConfigId, uint64_t FileStartVA) {
|
||||
auto& CodeCache = Thread.CTX->GetCodeCache();
|
||||
|
||||
auto CacheFilename = fextl::fmt::format("{}cache/{}-{:016x}", FEX::Config::GetCacheDirectory(),
|
||||
FEXCore::CodeMap::GetBaseFilename(Section.FileInfo, false), CodeCacheConfigId);
|
||||
FEXCore::CodeMap::GetBaseFilename(FileInfo, false), CodeCacheConfigId);
|
||||
int CacheFD = open(CacheFilename.c_str(), O_RDONLY);
|
||||
if (CacheFD == -1) {
|
||||
LogMan::Msg::IFmt("Cache file does not exist: {}", CacheFilename);
|
||||
return;
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
struct stat buf;
|
||||
if (fstat(CacheFD, &buf) != 0) {
|
||||
LogMan::Msg::EFmt("Invalid cache file: {}", CacheFilename);
|
||||
close(CacheFD);
|
||||
return;
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
auto CacheFileSize = buf.st_size;
|
||||
auto CacheFileSize = static_cast<std::size_t>(buf.st_size);
|
||||
auto MappedCache = (std::byte*)FEXCore::Allocator::mmap(nullptr, CacheFileSize, PROT_READ, MAP_PRIVATE, CacheFD, 0);
|
||||
LOGMAN_THROW_A_FMT(MappedCache, "Failed to map code cache into memory");
|
||||
if (!Thread.CTX->GetCodeCache().LoadData(&Thread, MappedCache, Section)) {
|
||||
// TODO: Delete this cache file
|
||||
}
|
||||
FEXCore::Allocator::munmap(MappedCache, CacheFileSize);
|
||||
close(CacheFD);
|
||||
if (!MappedCache || MappedCache == MAP_FAILED) {
|
||||
LogMan::Msg::EFmt("Failed to map code cache into memory");
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
auto Result = CodeCache.LoadCache(std::span {MappedCache, CacheFileSize}, FileInfo, FileStartVA);
|
||||
if (!Result) {
|
||||
FEXCore::Allocator::munmap(MappedCache, CacheFileSize);
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
// NOTE: This is synchronized by acquiring VMATracking.Mutex at call site
|
||||
CodeCache.RegisterMappedCodeBuffer(*Result);
|
||||
|
||||
return Result;
|
||||
}
|
||||
|
||||
void* SyscallHandler::GuestMmap(bool Is64Bit, FEXCore::Core::InternalThreadState* Thread, void* addr, size_t length, int prot, int flags,
|
||||
int fd, off_t offset) {
|
||||
LOGMAN_THROW_A_FMT(Is64Bit || (length >> 32) == 0, "values must fit to 32 bits");
|
||||
|
||||
uint64_t Result {};
|
||||
uint64_t Result;
|
||||
size_t Size = FEXCore::AlignUp(length, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
std::optional<LateApplyExtendedVolatileMetadata> LateMetadata = std::nullopt;
|
||||
|
||||
std::optional<FEXCore::ExecutableFileSectionInfo> CachedSection;
|
||||
bool PendingResourceDeletion;
|
||||
|
||||
{
|
||||
// NOTE: Frontend calls this with a nullptr Thread during initialization, but
|
||||
@@ -290,9 +331,10 @@ void* SyscallHandler::GuestMmap(bool Is64Bit, FEXCore::Core::InternalThreadState
|
||||
}
|
||||
|
||||
LateMetadata = TrackMmap(Thread, Result, length, prot, flags, fd, offset, CachedSection);
|
||||
PendingResourceDeletion = VMATracking.HasPendingResourceDeletions();
|
||||
}
|
||||
|
||||
InvalidateCodeRangeIfNecessary(Thread, Result, Size);
|
||||
InvalidateCodeRangeIfNecessary(Thread, Result, Size, PendingResourceDeletion);
|
||||
|
||||
if (LateMetadata) {
|
||||
auto CodeInvalidationlk = FEXCore::GuardSignalDeferringSectionWithFallback(CTX->GetCodeInvalidationMutex(), Thread);
|
||||
@@ -300,7 +342,8 @@ void* SyscallHandler::GuestMmap(bool Is64Bit, FEXCore::Core::InternalThreadState
|
||||
}
|
||||
|
||||
if (EnableCodeCaching && CachedSection) {
|
||||
LoadCodeCache(*Thread, *CachedSection, CodeCacheConfigId);
|
||||
Thread->CTX->GetCodeCache().EnableLoadedSection(
|
||||
Thread, *static_cast<const VMATracking::ExecutableFileState&>(CachedSection->FileInfo).MappedCache, *CachedSection);
|
||||
}
|
||||
|
||||
return reinterpret_cast<void*>(Result);
|
||||
@@ -310,8 +353,9 @@ uint64_t SyscallHandler::GuestMunmap(bool Is64Bit, FEXCore::Core::InternalThread
|
||||
LOGMAN_THROW_A_FMT(Is64Bit || (reinterpret_cast<uintptr_t>(addr) >> 32) == 0, "values must fit to 32 bits: {}", fmt::ptr(addr));
|
||||
LOGMAN_THROW_A_FMT(Is64Bit || (length >> 32) == 0, "values must fit to 32 bits");
|
||||
|
||||
uint64_t Result {};
|
||||
uint64_t Result;
|
||||
uint64_t Size = FEXCore::AlignUp(length, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
bool PendingResourceDeletion;
|
||||
|
||||
{
|
||||
// Frontend calls this with nullptr Thread during initialization.
|
||||
@@ -331,8 +375,9 @@ uint64_t SyscallHandler::GuestMunmap(bool Is64Bit, FEXCore::Core::InternalThread
|
||||
}
|
||||
}
|
||||
TrackMunmap(Thread, addr, length);
|
||||
PendingResourceDeletion = VMATracking.HasPendingResourceDeletions();
|
||||
}
|
||||
InvalidateCodeRangeIfNecessary(Thread, reinterpret_cast<uint64_t>(addr), Size);
|
||||
InvalidateCodeRangeIfNecessary(Thread, reinterpret_cast<uint64_t>(addr), Size, PendingResourceDeletion);
|
||||
|
||||
if (length) {
|
||||
auto CodeInvalidationlk = FEXCore::GuardSignalDeferringSectionWithFallback(CTX->GetCodeInvalidationMutex(), Thread);
|
||||
@@ -382,7 +427,7 @@ void SyscallHandler::TriggerGuestLibWrapperCodeCacheLoad(FEXCore::Core::Internal
|
||||
}
|
||||
|
||||
auto SectionInfo = BuildSectionInfo(*VMAEntry->second.Resource, VMA->Base, VMA->Length);
|
||||
LoadCodeCache(Thread, SectionInfo, CodeCacheConfigId);
|
||||
LoadCodeCache(Thread, VMATracking, SectionInfo.FileInfo, CodeCacheConfigId, SectionInfo.FileStartVA);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -433,7 +478,7 @@ uint64_t SyscallHandler::GuestMprotect(FEXCore::Core::InternalThreadState* Threa
|
||||
TrackMprotect(Thread, addr, len, prot);
|
||||
}
|
||||
|
||||
InvalidateCodeRangeIfNecessary(Thread, reinterpret_cast<uint64_t>(addr), len);
|
||||
InvalidateCodeRangeIfNecessary(Thread, reinterpret_cast<uint64_t>(addr), len, false);
|
||||
|
||||
// Prepare for delayed code cache load after ld/Wine is done applying relocations.
|
||||
// Hooking into mprotect is a reliable heuristic that matches behavior of ld (for ELF) and Wine (for PE).
|
||||
@@ -457,17 +502,21 @@ uint64_t SyscallHandler::GuestMprotect(FEXCore::Core::InternalThreadState* Threa
|
||||
}
|
||||
|
||||
// Trigger delayed cache load. This must be done separately since
|
||||
// LoadCodeCache will call interfaces that acquire the VMATracking mutex.
|
||||
// EnableLoadedSection will call interfaces that acquire the VMATracking mutex.
|
||||
for (auto& CachedSection : CachedSections) {
|
||||
LoadCodeCache(*Thread, CachedSection, CodeCacheConfigId);
|
||||
auto Cache = static_cast<const VMATracking::ExecutableFileState&>(CachedSection.FileInfo).MappedCache.get();
|
||||
if (Cache) {
|
||||
Thread->CTX->GetCodeCache().EnableLoadedSection(Thread, *Cache, CachedSection);
|
||||
}
|
||||
}
|
||||
|
||||
return Result;
|
||||
}
|
||||
|
||||
uint64_t SyscallHandler::GuestShmat(bool Is64Bit, FEXCore::Core::InternalThreadState* Thread, int shmid, const void* shmaddr, int shmflg) {
|
||||
uint64_t Result {};
|
||||
uint64_t Length {};
|
||||
uint64_t Result;
|
||||
uint64_t Length;
|
||||
bool PendingResourceDeletion;
|
||||
|
||||
{
|
||||
auto lk = FEXCore::GuardSignalDeferringSection(VMATracking.Mutex, Thread);
|
||||
@@ -492,15 +541,17 @@ uint64_t SyscallHandler::GuestShmat(bool Is64Bit, FEXCore::Core::InternalThreadS
|
||||
|
||||
Length = stat.shm_segsz;
|
||||
TrackShmat(Thread, shmid, Result, shmflg, Length);
|
||||
PendingResourceDeletion = VMATracking.HasPendingResourceDeletions();
|
||||
}
|
||||
|
||||
InvalidateCodeRangeIfNecessary(Thread, Result, Length);
|
||||
InvalidateCodeRangeIfNecessary(Thread, Result, Length, PendingResourceDeletion);
|
||||
return Result;
|
||||
}
|
||||
|
||||
uint64_t SyscallHandler::GuestShmdt(bool Is64Bit, FEXCore::Core::InternalThreadState* Thread, const void* shmaddr) {
|
||||
uint64_t Result {};
|
||||
uint64_t Length {};
|
||||
uint64_t Result;
|
||||
uint64_t Length;
|
||||
bool PendingResourceDeletion;
|
||||
{
|
||||
auto lk = FEXCore::GuardSignalDeferringSection(VMATracking.Mutex, Thread);
|
||||
if (Is64Bit) {
|
||||
@@ -516,9 +567,10 @@ uint64_t SyscallHandler::GuestShmdt(bool Is64Bit, FEXCore::Core::InternalThreadS
|
||||
}
|
||||
|
||||
Length = TrackShmdt(Thread, reinterpret_cast<uintptr_t>(shmaddr));
|
||||
PendingResourceDeletion = VMATracking.HasPendingResourceDeletions();
|
||||
}
|
||||
|
||||
InvalidateCodeRangeIfNecessary(Thread, reinterpret_cast<uintptr_t>(shmaddr), Length);
|
||||
InvalidateCodeRangeIfNecessary(Thread, reinterpret_cast<uintptr_t>(shmaddr), Length, PendingResourceDeletion);
|
||||
return Result;
|
||||
}
|
||||
|
||||
@@ -557,7 +609,7 @@ SyscallHandler::TrackMmap(FEXCore::Core::InternalThreadState* Thread, uint64_t a
|
||||
if (PathLength != -1 && S_ISREG(buf.st_mode) && (buf.st_mode & S_IXUSR)) {
|
||||
// ELF files that are mapped multiple times get a separate MappedResource for each base virtual address
|
||||
if ((prot & PROT_READ) && Inserted) {
|
||||
Resource->MappedFile = fextl::make_unique<FEXCore::ExecutableFileInfo>();
|
||||
Resource->MappedFile = fextl::make_unique<VMATracking::ExecutableFileState>();
|
||||
Resource->MappedFile->Filename = fextl::string(Tmp, PathLength);
|
||||
Resource->MappedFile->FileId = CTX->GetCodeCache().ComputeCodeMapId(Resource->MappedFile->Filename, fd);
|
||||
|
||||
@@ -643,7 +695,14 @@ SyscallHandler::TrackMmap(FEXCore::Core::InternalThreadState* Thread, uint64_t a
|
||||
// Load code cache if present.
|
||||
// FEXServer was requested to generate library caches on program launch.
|
||||
if (EnableCodeCaching && Resource && Resource->MappedFile && VMATracking::VMAProt::fromProt(prot).Executable) {
|
||||
if (Resource->MappedFile->Filename.ends_with("-guest.so")) {
|
||||
if (!Resource->MappedFile->AttemptedCacheLoad) {
|
||||
Resource->MappedFile->MappedCache = LoadCodeCache(*Thread, VMATracking, *Resource->MappedFile, CodeCacheConfigId, Resource->FirstVMA->Base);
|
||||
Resource->MappedFile->AttemptedCacheLoad = true;
|
||||
}
|
||||
|
||||
if (!Resource->MappedFile->MappedCache) {
|
||||
// No cache present
|
||||
} else if (Resource->MappedFile->Filename.ends_with("-guest.so")) {
|
||||
// For guest library wrappers, cache loading must be delayed until LoadLib is called.
|
||||
// Before that, we can't patch up the SHA256 function identifiers.
|
||||
LogMan::Msg::IFmt("Delaying code cache load for {}", Resource->MappedFile->Filename);
|
||||
@@ -662,6 +721,12 @@ SyscallHandler::TrackMmap(FEXCore::Core::InternalThreadState* Thread, uint64_t a
|
||||
return VolatileMetadata;
|
||||
}
|
||||
|
||||
void SyscallHandler::AddVirtualPage(FEXCore::Core::InternalThreadState* Thread, uint64_t addr, size_t length, int prot) {
|
||||
auto lk = FEXCore::GuardSignalDeferringSectionWithFallback(VMATracking.Mutex, Thread);
|
||||
VMATracking.TrackVMARange(CTX, nullptr, addr, 0, length, VMATracking::VMAFlags::fromFlags(MAP_ANONYMOUS | MAP_PRIVATE),
|
||||
VMATracking::VMAProt::fromProt(prot));
|
||||
}
|
||||
|
||||
void SyscallHandler::TrackMunmap(FEXCore::Core::InternalThreadState* Thread, void* addr, size_t length) {
|
||||
uint64_t Size = FEXCore::AlignUp(length, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
VMATracking.DeleteVMARange(CTX, reinterpret_cast<uintptr_t>(addr), Size);
|
||||
|
||||
@@ -7,10 +7,20 @@ desc: VMA Tracking
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "FEXCore/Utils/MathUtils.h"
|
||||
#include "LinuxSyscalls/Syscalls.h"
|
||||
#include <sys/shm.h>
|
||||
|
||||
namespace FEX::HLE::VMATracking {
|
||||
|
||||
ExecutableFileState::~ExecutableFileState() {
|
||||
if (MappedCache && MappedCache->MappedFile.data()) {
|
||||
auto ret = FEXCore::Allocator::munmap(MappedCache->MappedFile.data(),
|
||||
FEXCore::AlignUp(MappedCache->MappedFile.size_bytes(), FEXCore::Utils::FEX_PAGE_SIZE));
|
||||
LOGMAN_THROW_A_FMT(ret == 0, "Error unmapping cache for {}: {} {}", Filename, errno, strerror(errno));
|
||||
}
|
||||
}
|
||||
|
||||
/// Helpers ///
|
||||
auto VMAProt::fromProt(int Prot) -> VMAProt {
|
||||
return VMAProt {
|
||||
@@ -235,7 +245,13 @@ void VMATracking::DeleteVMARange(FEXCore::Context::Context* CTX, uintptr_t Base,
|
||||
// If linked to a Mapped Resource, remove from linked list and possibly delete the Mapped Resource
|
||||
if (Current->Resource) {
|
||||
if (ListRemove(Current) && Current->Resource != PreservedMappedResource) {
|
||||
MappedResources.erase(Current->Resource->Iterator);
|
||||
auto Iter = Current->Resource->Iterator;
|
||||
// Defer deletion if the resource has mapped code cache data, so its code buffer
|
||||
// outlives code cache invalidation (which runs after the VMA lock is released).
|
||||
if (Current->Resource->MappedFile && Current->Resource->MappedFile->MappedCache) {
|
||||
PendingResourceDeletions.push_back(std::move(*Current->Resource));
|
||||
}
|
||||
MappedResources.erase(Iter);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -277,6 +293,11 @@ void VMATracking::DeleteVMARange(FEXCore::Context::Context* CTX, uintptr_t Base,
|
||||
}
|
||||
}
|
||||
|
||||
void VMATracking::FlushPendingResourceDeletions() {
|
||||
Mutex.check_lock_owned_by_self_as_write();
|
||||
PendingResourceDeletions.clear();
|
||||
}
|
||||
|
||||
// Change flags of mappings in a range and split the mappings if needed
|
||||
void VMATracking::ChangeProtectionFlags(uintptr_t Base, uintptr_t Length, VMAProt NewProt) {
|
||||
Mutex.check_lock_owned_by_self_as_write();
|
||||
@@ -542,7 +563,11 @@ uintptr_t VMATracking::DeleteSHMRegion(FEXCore::Context::Context* CTX, uintptr_t
|
||||
do {
|
||||
if (Entry->second.Resource == Resource) {
|
||||
if (ListRemove(&Entry->second)) {
|
||||
MappedResources.erase(Entry->second.Resource->Iterator);
|
||||
auto Iter = Entry->second.Resource->Iterator;
|
||||
if (Entry->second.Resource->MappedFile && Entry->second.Resource->MappedFile->MappedCache) {
|
||||
PendingResourceDeletions.push_back(std::move(*Entry->second.Resource));
|
||||
}
|
||||
MappedResources.erase(Iter);
|
||||
}
|
||||
Entry = VMAs.erase(Entry);
|
||||
} else {
|
||||
@@ -552,4 +577,18 @@ uintptr_t VMATracking::DeleteSHMRegion(FEXCore::Context::Context* CTX, uintptr_t
|
||||
|
||||
return ShmLength;
|
||||
}
|
||||
|
||||
FEXCore::MappedCodeCacheFile* VMATracking::FindMappedCodeCacheByHostAddress(uintptr_t HostAddr) const {
|
||||
for (auto& [_, Resource] : MappedResources) {
|
||||
if (Resource.MappedFile && Resource.MappedFile->MappedCache) {
|
||||
auto* Code = Resource.MappedFile->MappedCache.get();
|
||||
auto BufferStart = reinterpret_cast<uintptr_t>(Code->CodeBuffer.data());
|
||||
if (HostAddr >= BufferStart && HostAddr < BufferStart + Code->CodeBuffer.size_bytes()) {
|
||||
return Code;
|
||||
}
|
||||
}
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
} // namespace FEX::HLE::VMATracking
|
||||
@@ -4,12 +4,18 @@
|
||||
#include <cstdint>
|
||||
#include <tuple>
|
||||
|
||||
#include "FEXCore/Core/CodeCache.h"
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
|
||||
#include <elf.h>
|
||||
|
||||
namespace FEXCore {
|
||||
struct ExecutableFileInfo;
|
||||
struct MappedCodeCacheFile;
|
||||
} // namespace FEXCore
|
||||
|
||||
namespace FEX::HLE::VMATracking {
|
||||
///// VMA (Virtual Memory Area) tracking /////
|
||||
|
||||
@@ -32,6 +38,13 @@ struct MRID {
|
||||
|
||||
struct VMAEntry;
|
||||
|
||||
struct ExecutableFileState : FEXCore::ExecutableFileInfo {
|
||||
~ExecutableFileState();
|
||||
|
||||
bool AttemptedCacheLoad = false;
|
||||
fextl::unique_ptr<FEXCore::MappedCodeCacheFile> MappedCache;
|
||||
};
|
||||
|
||||
/**
|
||||
* Meta data associated to one system resource.
|
||||
*
|
||||
@@ -43,7 +56,7 @@ struct VMAEntry;
|
||||
struct MappedResource {
|
||||
using ContainerType = fextl::multimap<MRID, MappedResource>;
|
||||
|
||||
fextl::unique_ptr<FEXCore::ExecutableFileInfo> MappedFile;
|
||||
fextl::unique_ptr<ExecutableFileState> MappedFile;
|
||||
// Pointer to lowest memory range this file is mapped to
|
||||
VMAEntry* FirstVMA;
|
||||
uint64_t Length; // 0 if not fixed size
|
||||
@@ -136,8 +149,22 @@ struct VMATracking {
|
||||
return MappedResources.equal_range(mrid);
|
||||
}
|
||||
|
||||
// Find any MappedCodeCacheFile that contains the given host code address.
|
||||
// - Mutex must be shared_locked before calling
|
||||
FEXCore::MappedCodeCacheFile* FindMappedCodeCacheByHostAddress(uintptr_t HostAddr) const;
|
||||
|
||||
bool HasPendingResourceDeletions() const {
|
||||
return !PendingResourceDeletions.empty();
|
||||
}
|
||||
|
||||
// Flush pending MappedResource deletions. This must be called after code
|
||||
// invalidation related to unmapped/remapped memory to avoid memory leaks.
|
||||
// - Mutex must be unique_locked before calling
|
||||
void FlushPendingResourceDeletions();
|
||||
|
||||
private:
|
||||
MappedResource::ContainerType MappedResources;
|
||||
fextl::vector<MappedResource> PendingResourceDeletions;
|
||||
};
|
||||
|
||||
|
||||
|
||||
@@ -22,6 +22,16 @@ $end_info$
|
||||
namespace FEX::HLE::x64 {
|
||||
void RegisterTime(FEX::HLE::SyscallHandler* Handler) {
|
||||
using namespace FEXCore::IR;
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X64(gettimeofday, [](FEXCore::Core::CpuStateFrame* Frame, timeval* tv, struct timezone* tz) -> uint64_t {
|
||||
FaultSafeUserMemAccess::VerifyIsWritableOrNull(tv, sizeof(*tv));
|
||||
FaultSafeUserMemAccess::VerifyIsWritableOrNull(tz, sizeof(*tz));
|
||||
|
||||
// Passed through glibc to ensure vdso is used if possible.
|
||||
uint64_t Result = ::gettimeofday(tv, tz);
|
||||
SYSCALL_ERRNO();
|
||||
});
|
||||
|
||||
REGISTER_SYSCALL_IMPL_X64(time, [](FEXCore::Core::CpuStateFrame* Frame, time_t* tloc) -> uint64_t {
|
||||
FaultSafeUserMemAccess::VerifyIsWritableOrNull(tloc, sizeof(time_t));
|
||||
uint64_t Result = ::time(tloc);
|
||||
|
||||
@@ -18,9 +18,8 @@ check_target_ec:
|
||||
// Check if target is in fact x86 code
|
||||
ldr x16, [x18, #0x60] // TEB->PEB
|
||||
ldr x16, [x16, #0x368] // PEB->EcCodeBitMap
|
||||
lsr x17, x9, #15
|
||||
and x17, x17, #0x1fffffffffff8
|
||||
ldr x16, [x16, x17]
|
||||
lsr x17, x9, #18
|
||||
ldr x16, [x16, x17, lsl #3]
|
||||
lsr x17, x9, #12
|
||||
lsr x16, x16, x17
|
||||
tbnz x16, #0, ExitFunctionEC
|
||||
|
||||
@@ -29,6 +29,7 @@ $end_info$
|
||||
|
||||
#include "Windows/Common/Allocator.h"
|
||||
#include "Windows/Common/EnvironmentVariablesHandling.h"
|
||||
#include "Windows/Common/FEXUnixLib.h"
|
||||
#include "Common/CallRetStack.h"
|
||||
#include "Common/JITGuardPage.h"
|
||||
#include "Common/Config.h"
|
||||
@@ -496,7 +497,7 @@ static void RethrowGuestException(const EXCEPTION_RECORD& Rec, ARM64_NT_CONTEXT&
|
||||
EFlags &= ~(1 << FEXCore::X86State::RFLAG_TF_RAW_LOC);
|
||||
CTX->SetFlagsFromCompactedEFLAGS(Thread, EFlags);
|
||||
|
||||
Args->Rec = FEX::Windows::HandleGuestException(Fault, Rec, Args->Context.Pc, Args->Context.X8);
|
||||
Args->Rec = FEX::Windows::HandleGuestException(Fault, Rec, Args->Context.Pc, Args->Context.X8, Args->Context.X0);
|
||||
if (Args->Rec.ExceptionCode == EXCEPTION_SINGLE_STEP) {
|
||||
Args->Context.Cpsr &= ~(1 << 21); // PSTATE.SS
|
||||
} else if (Args->Rec.ExceptionCode == EXCEPTION_BREAKPOINT) {
|
||||
@@ -599,9 +600,10 @@ NTSTATUS ProcessInit() {
|
||||
FEX::Windows::SetupEnvironmentVariableValues(NtDll);
|
||||
|
||||
FEX::Windows::Allocator::SetupHooks(NtDll);
|
||||
FEX::Windows::UnixLib::Init(NtDll);
|
||||
|
||||
{
|
||||
auto HostFeatures = FEX::Windows::CPUFeatures::FetchHostFeatures(IsWine);
|
||||
auto HostFeatures = FEX::Windows::CPUFeatures::FetchHostFeatures(IsWine, FEXCore::HostFeatures::HostTypeEnum::Arm64ec);
|
||||
CTX = FEXCore::Context::Context::CreateNewContext(HostFeatures);
|
||||
}
|
||||
|
||||
@@ -619,7 +621,7 @@ NTSTATUS ProcessInit() {
|
||||
|
||||
CPUFeatures.emplace(*CTX);
|
||||
|
||||
X64ReturnInstr = ::VirtualAlloc(nullptr, FEXCore::Utils::FEX_PAGE_SIZE, MEM_COMMIT, PAGE_EXECUTE_READWRITE);
|
||||
X64ReturnInstr = ::VirtualAlloc(nullptr, FEXCore::Utils::FEX_PAGE_SIZE, MEM_COMMIT | MEM_TOP_DOWN, PAGE_EXECUTE_READWRITE);
|
||||
InvalidationTracker->HandleMemoryProtectionNotification(reinterpret_cast<uint64_t>(X64ReturnInstr), FEXCore::Utils::FEX_PAGE_SIZE,
|
||||
PAGE_EXECUTE_READ);
|
||||
*reinterpret_cast<uint8_t*>(X64ReturnInstr) = 0xc3;
|
||||
@@ -627,15 +629,6 @@ NTSTATUS ProcessInit() {
|
||||
const uintptr_t KiUserExceptionDispatcherFFS = reinterpret_cast<uintptr_t>(GetProcAddress(NtDll, "KiUserExceptionDispatcher"));
|
||||
Exception::KiUserExceptionDispatcher = NtDllRedirectionLUT[KiUserExceptionDispatcherFFS - NtDllBase] + NtDllBase;
|
||||
|
||||
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
|
||||
if (TSOEnabled()) {
|
||||
BOOL Enable = TRUE;
|
||||
NTSTATUS Status = NtSetInformationProcess(NtCurrentProcess(), ProcessFexHardwareTso, &Enable, sizeof(Enable));
|
||||
if (Status == STATUS_SUCCESS) {
|
||||
CTX->SetHardwareTSOSupport(true);
|
||||
}
|
||||
}
|
||||
|
||||
FEX_CONFIG_OPT(ProfileStats, PROFILESTATS);
|
||||
FEX_CONFIG_OPT(StartupSleep, STARTUPSLEEP);
|
||||
FEX_CONFIG_OPT(StartupSleepProcName, STARTUPSLEEPPROCNAME);
|
||||
@@ -919,7 +912,8 @@ NTSTATUS ThreadInit() {
|
||||
const auto CPUArea = GetCPUArea();
|
||||
|
||||
static constexpr size_t EmulatorStackSize = 0x40000;
|
||||
const uint64_t EmulatorStack = reinterpret_cast<uint64_t>(::VirtualAlloc(nullptr, EmulatorStackSize, MEM_COMMIT | MEM_RESERVE, PAGE_READWRITE));
|
||||
const uint64_t EmulatorStack =
|
||||
reinterpret_cast<uint64_t>(::VirtualAlloc(nullptr, EmulatorStackSize, MEM_COMMIT | MEM_RESERVE | MEM_TOP_DOWN, PAGE_READWRITE));
|
||||
CPUArea.EmulatorStackLimit() = EmulatorStack;
|
||||
CPUArea.EmulatorStackBase() = EmulatorStack + EmulatorStackSize;
|
||||
|
||||
|
||||
@@ -29,3 +29,5 @@ if (ARCHITECTURE_arm64ec)
|
||||
elseif (ARCHITECTURE_arm64)
|
||||
add_subdirectory(WOW64)
|
||||
endif()
|
||||
|
||||
add_subdirectory(../Tools/FEXOfflineCompiler ${CMAKE_CURRENT_BINARY_DIR}/FEXOfflineCompiler)
|
||||
@@ -1,6 +1,7 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <FEXCore/Utils/AllocatorHooks.h>
|
||||
#include <FEXCore/Utils/TypeDefines.h>
|
||||
#include "Windows/Common/FEXUnixLib.h"
|
||||
|
||||
#include <array>
|
||||
#include <chrono>
|
||||
@@ -13,228 +14,13 @@
|
||||
#include <wine/debug.h>
|
||||
|
||||
namespace FEX::Windows::Allocator {
|
||||
#define PR_SET_VMA 0x53564d41
|
||||
#define PR_SET_VMA_ANON_NAME 0
|
||||
|
||||
#define MADV_HUGEPAGE 14
|
||||
#define MADV_NOHUGEPAGE 15
|
||||
|
||||
namespace Trampoline {
|
||||
struct madvise_data {
|
||||
const void* addr;
|
||||
size_t size;
|
||||
int advise;
|
||||
};
|
||||
|
||||
struct prctl_data {
|
||||
int op;
|
||||
uint64_t attr;
|
||||
const void* addr;
|
||||
size_t size;
|
||||
const char* name;
|
||||
uint64_t ret;
|
||||
};
|
||||
|
||||
__attribute__((naked)) uint64_t wine_prctl(prctl_data* d) {
|
||||
asm volatile(
|
||||
R"(
|
||||
.globl wine_prctl_begin
|
||||
wine_prctl_begin:
|
||||
mov x19, x0;
|
||||
mov x8, 167; // prctl
|
||||
ldr x0, [x19]; // op
|
||||
ldp x1, x2, [x19, %[attr_offset]]; // {attr, addr}
|
||||
ldp x3, x4, [x19, %[size_offset]]; // {size, name}
|
||||
svc #0;
|
||||
str x0, [x19, %[ret_offset]];
|
||||
// Tell wine it was all groovy.
|
||||
mov x0, 0;
|
||||
ret;
|
||||
|
||||
.globl wine_prctl_end
|
||||
wine_prctl_end:
|
||||
)"
|
||||
:
|
||||
: [attr_offset] "i"(offsetof(prctl_data, attr)), [size_offset] "i"(offsetof(prctl_data, size)), [ret_offset] "i"(offsetof(prctl_data, ret))
|
||||
: "memory");
|
||||
};
|
||||
|
||||
__attribute__((naked)) uint64_t wine_madvise(madvise_data* d) {
|
||||
asm volatile(R"(
|
||||
.globl wine_madvise_begin
|
||||
wine_madvise_begin:
|
||||
mov x8, 233; // madvise
|
||||
ldr x2, [x0, %[advise_offset]]; // advise
|
||||
ldp x0, x1, [x0]; // {addr, size}
|
||||
svc #0;
|
||||
// Tell wine it was all groovy.
|
||||
mov x0, 0;
|
||||
ret;
|
||||
|
||||
.globl wine_madvise_end
|
||||
wine_madvise_end:
|
||||
)" ::[advise_offset] "i"(offsetof(madvise_data, advise))
|
||||
: "memory");
|
||||
}
|
||||
|
||||
extern "C" uint64_t wine_madvise_begin;
|
||||
extern "C" uint64_t wine_madvise_end;
|
||||
|
||||
void* const wine_madvise_begin_loc = &wine_madvise_begin;
|
||||
void* const wine_madvise_end_loc = &wine_madvise_end;
|
||||
|
||||
extern "C" uint64_t wine_prctl_begin;
|
||||
extern "C" uint64_t wine_prctl_end;
|
||||
|
||||
void* const wine_prctl_begin_loc = &wine_prctl_begin;
|
||||
void* const wine_prctl_end_loc = &wine_prctl_end;
|
||||
|
||||
extern NTSTATUS(WINAPI* __wine_unix_call_dispatcher)(uint64_t, unsigned int, void*);
|
||||
decltype(__wine_unix_call_dispatcher) WineUnixCall;
|
||||
|
||||
enum unix_function_indexes {
|
||||
INDEX_PRCTL = 0,
|
||||
INDEX_MADVISE = 1,
|
||||
INDEX_MAX,
|
||||
};
|
||||
static std::array<void*, INDEX_MAX> unix_functions {};
|
||||
|
||||
static uint64_t wine_prctl_trampoline(int op, uint64_t attr, const void* addr, size_t size, const char* name) {
|
||||
prctl_data d {
|
||||
.op = op,
|
||||
.attr = attr,
|
||||
.addr = addr,
|
||||
.size = size,
|
||||
.name = name,
|
||||
};
|
||||
WineUnixCall(reinterpret_cast<uint64_t>(unix_functions.data()), INDEX_PRCTL, &d);
|
||||
return d.ret;
|
||||
}
|
||||
|
||||
static uint64_t wine_madvise_trampoline(const void* addr, size_t size, int advice) {
|
||||
madvise_data d {
|
||||
.addr = addr,
|
||||
.size = size,
|
||||
.advise = advice,
|
||||
};
|
||||
WineUnixCall(reinterpret_cast<uint64_t>(unix_functions.data()), INDEX_MADVISE, &d);
|
||||
return 0;
|
||||
}
|
||||
|
||||
void VirtualName(const char* Name, const void* Ptr, size_t Size) {
|
||||
static bool Supports {true};
|
||||
if (Supports) {
|
||||
auto Result = wine_prctl_trampoline(PR_SET_VMA, PR_SET_VMA_ANON_NAME, Ptr, Size, Name);
|
||||
if (Result != 0) {
|
||||
// Disable any additional attempts.
|
||||
Supports = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void VirtualTHPControl(const void* Ptr, size_t Size, FEXCore::Allocator::THPControl Control) {
|
||||
wine_madvise_trampoline(Ptr, Size, Control == FEXCore::Allocator::THPControl::Disable ? MADV_NOHUGEPAGE : MADV_HUGEPAGE);
|
||||
}
|
||||
} // namespace Trampoline
|
||||
|
||||
// This code path will eventually crash once Wine and the kernel implements `userspace syscall dispatch`.
|
||||
// FEX will need to switch over to using WINE's unixlib syscall approach then.
|
||||
// See `SetupHooks` for why unixlib doesn't work today.
|
||||
namespace Illegal {
|
||||
__attribute__((naked)) uint64_t prctl(int op, uint64_t attr, const void* addr, size_t size, const char* Name) {
|
||||
asm volatile(R"(
|
||||
mov x8, 167; // prctl
|
||||
svc #0;
|
||||
ret;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
|
||||
__attribute__((naked)) uint64_t madvise(const void* addr, size_t size, int advice) {
|
||||
asm volatile(R"(
|
||||
mov x8, 233; // madvise
|
||||
svc #0;
|
||||
ret;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
|
||||
void VirtualName(const char* Name, const void* Ptr, size_t Size) {
|
||||
static bool Supports {true};
|
||||
if (Supports) {
|
||||
auto Result = prctl(PR_SET_VMA, PR_SET_VMA_ANON_NAME, Ptr, Size, Name);
|
||||
if (Result != 0) {
|
||||
// Disable any additional attempts.
|
||||
Supports = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void VirtualTHPControl(const void* Ptr, size_t Size, FEXCore::Allocator::THPControl Control) {
|
||||
madvise(Ptr, Size, Control == FEXCore::Allocator::THPControl::Disable ? MADV_NOHUGEPAGE : MADV_HUGEPAGE);
|
||||
}
|
||||
} // namespace Illegal
|
||||
|
||||
void SetupHooks(HMODULE ntdll) {
|
||||
// If this symbol doesn't exist, then we aren't running under WINE.
|
||||
const auto Sym = GetProcAddress(ntdll, "__wine_unix_call_dispatcher");
|
||||
|
||||
if (!Sym) {
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::Allocator::HookPtrs Ptrs {};
|
||||
|
||||
// Wine will soon require us to use unixlib for calling helper routines that use Linux syscalls.
|
||||
// It needs this for `userspace syscall dispatch` to capture rogue applications doing raw Windows syscalls.
|
||||
// If FEX doesn't use unixlib, then when WINE and a kernel implements this, then our "illegal" path will start crashing.
|
||||
//
|
||||
// This code currently conflicts with FEX's `Call Checker` since the latter captures uses of `unixlib`.
|
||||
// We hence have to stick to the `Illegal::` functions until a workaround is implemented.
|
||||
if constexpr (false) {
|
||||
// NTSTATUS __wine_unix_call_dispatcher( unixlib_handle_t, unsigned int, void * );
|
||||
// - unixlib_handle_t is just an array of functions
|
||||
// - uint32_t is just an index in to that
|
||||
// - void* is the user provided pointer, gets loaded in to x0 in for the unix_function called.
|
||||
// - Return value - SUCCESS or other error.
|
||||
Trampoline::WineUnixCall = *reinterpret_cast<decltype(Trampoline::WineUnixCall)*>(Sym);
|
||||
|
||||
// This code must be copied over to allocated memory from top down allocations apparently.
|
||||
auto Code = reinterpret_cast<uint8_t*>(
|
||||
::VirtualAlloc(nullptr, FEXCore::Utils::FEX_PAGE_SIZE, MEM_RESERVE | MEM_COMMIT | MEM_TOP_DOWN, PAGE_EXECUTE_READWRITE));
|
||||
if (!Code) {
|
||||
return;
|
||||
}
|
||||
|
||||
size_t CurrentOffset {};
|
||||
const size_t prctl_size =
|
||||
reinterpret_cast<uintptr_t>(Trampoline::wine_prctl_end_loc) - reinterpret_cast<uintptr_t>(Trampoline::wine_prctl_begin_loc);
|
||||
const size_t madvise_size =
|
||||
reinterpret_cast<uintptr_t>(Trampoline::wine_madvise_end_loc) - reinterpret_cast<uintptr_t>(Trampoline::wine_madvise_begin_loc);
|
||||
|
||||
// Copy prctl.
|
||||
memcpy(Code + CurrentOffset, Trampoline::wine_prctl_begin_loc, prctl_size);
|
||||
Trampoline::unix_functions[Trampoline::INDEX_PRCTL] = Code + CurrentOffset;
|
||||
CurrentOffset += prctl_size;
|
||||
|
||||
// Copy madvise.
|
||||
memcpy(Code + CurrentOffset, Trampoline::wine_madvise_begin_loc, madvise_size);
|
||||
Trampoline::unix_functions[Trampoline::INDEX_MADVISE] = Code + CurrentOffset;
|
||||
|
||||
// Protect the page now.
|
||||
FEXCore::Allocator::VirtualProtect(Code, FEXCore::Utils::FEX_PAGE_SIZE,
|
||||
FEXCore::Allocator::ProtectOptions::Read | FEXCore::Allocator::ProtectOptions::Exec);
|
||||
|
||||
Ptrs = {
|
||||
.VirtualName = Trampoline::VirtualName,
|
||||
.VirtualTHPControl = Trampoline::VirtualTHPControl,
|
||||
};
|
||||
} else {
|
||||
Ptrs = {
|
||||
.VirtualName = Illegal::VirtualName,
|
||||
.VirtualTHPControl = Illegal::VirtualTHPControl,
|
||||
};
|
||||
}
|
||||
Ptrs = {
|
||||
.VirtualName = UnixLib::VirtualName,
|
||||
.VirtualTHPControl = UnixLib::VirtualTHPControl,
|
||||
};
|
||||
|
||||
SYSTEM_INFO system_info {};
|
||||
GetSystemInfo(&system_info);
|
||||
|
||||
@@ -10,6 +10,7 @@ add_library(CommonWindows STATIC
|
||||
Allocator.cpp
|
||||
CPUFeatures.cpp
|
||||
EnvironmentVariablesHandling.cpp
|
||||
FEXUnixLib.cpp
|
||||
SHMStats.cpp
|
||||
InvalidationTracker.cpp
|
||||
ImageTracker.cpp
|
||||
|
||||
@@ -56,7 +56,7 @@ public:
|
||||
}
|
||||
};
|
||||
|
||||
FEXCore::HostFeatures CPUFeatures::FetchHostFeatures(bool IsWine) {
|
||||
FEXCore::HostFeatures CPUFeatures::FetchHostFeatures(bool IsWine, FEXCore::HostFeatures::HostTypeEnum HostType) {
|
||||
HKEY Key = OpenProcessorKey(0);
|
||||
if (!Key) {
|
||||
ERROR_AND_DIE_FMT("Couldn't detect CPU features");
|
||||
@@ -82,6 +82,13 @@ FEXCore::HostFeatures CPUFeatures::FetchHostFeatures(bool IsWine) {
|
||||
HostFeatures.SupportsSVE256 = false;
|
||||
|
||||
HostFeatures.SupportsCPUIndexInTPIDRRO = !IsWine;
|
||||
|
||||
HostFeatures.HostType = HostType;
|
||||
|
||||
if (HostType == FEXCore::HostFeatures::HostTypeEnum::Wow64) {
|
||||
// AVX is unsupported for WOW64
|
||||
HostFeatures.SupportsAVX = false;
|
||||
}
|
||||
return HostFeatures;
|
||||
}
|
||||
|
||||
|
||||
@@ -16,7 +16,7 @@ class Context;
|
||||
namespace FEX::Windows {
|
||||
class CPUFeatures {
|
||||
public:
|
||||
static FEXCore::HostFeatures FetchHostFeatures(bool IsWine);
|
||||
static FEXCore::HostFeatures FetchHostFeatures(bool IsWine, FEXCore::HostFeatures::HostTypeEnum HostType);
|
||||
|
||||
CPUFeatures(FEXCore::Context::Context& CTX);
|
||||
|
||||
|
||||
@@ -22,8 +22,8 @@ CallRetStackInfo GetInfoThread(FEXCore::Core::InternalThreadState* Thread) {
|
||||
|
||||
void InitializeThread(FEXCore::Core::InternalThreadState* Thread) {
|
||||
// Allocate the call-ret stack with guard pages on both sides
|
||||
const void* CallRetStackAlloc = ::VirtualAlloc(
|
||||
nullptr, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE + 2 * FEXCore::Utils::FEX_PAGE_SIZE, MEM_RESERVE, PAGE_NOACCESS);
|
||||
const void* CallRetStackAlloc = ::VirtualAlloc(nullptr, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE + 2 * FEXCore::Utils::FEX_PAGE_SIZE,
|
||||
MEM_RESERVE | MEM_TOP_DOWN, PAGE_NOACCESS);
|
||||
|
||||
FEXCore::Allocator::VirtualName("FEXMem_CallRetStacks", CallRetStackAlloc,
|
||||
FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE + 2 * FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
|
||||
@@ -10,8 +10,8 @@
|
||||
|
||||
namespace FEX::Windows {
|
||||
template<typename TReg>
|
||||
static inline EXCEPTION_RECORD
|
||||
HandleGuestException(FEXCore::Core::CpuStateFrame::SynchronousFaultDataStruct& Fault, const EXCEPTION_RECORD& Src, TReg& Rip, TReg Rax) {
|
||||
static inline EXCEPTION_RECORD HandleGuestException(FEXCore::Core::CpuStateFrame::SynchronousFaultDataStruct& Fault,
|
||||
const EXCEPTION_RECORD& Src, TReg& Rip, TReg Rax, TReg Cx) {
|
||||
EXCEPTION_RECORD Dst = Src;
|
||||
Dst.ExceptionAddress = reinterpret_cast<void*>(Rip);
|
||||
|
||||
@@ -42,12 +42,18 @@ HandleGuestException(FEXCore::Core::CpuStateFrame::SynchronousFaultDataStruct& F
|
||||
case FEXCore::X86State::X86_TRAPNO_GP:
|
||||
if ((Fault.err_code & 0b111) == 0b010) {
|
||||
switch (Fault.err_code >> 3) {
|
||||
case 0x29:
|
||||
Dst.ExceptionCode = STATUS_STACK_BUFFER_OVERRUN;
|
||||
Dst.ExceptionAddress = reinterpret_cast<void*>(Rip);
|
||||
Dst.NumberParameters = 1;
|
||||
Dst.ExceptionInformation[0] = Cx;
|
||||
return Dst;
|
||||
case 0x2d:
|
||||
Rip += 3;
|
||||
Dst.ExceptionCode = EXCEPTION_BREAKPOINT;
|
||||
Dst.ExceptionAddress = reinterpret_cast<void*>(Rip);
|
||||
Dst.NumberParameters = 1;
|
||||
Dst.ExceptionInformation[0] = Rax; // RAX
|
||||
Dst.ExceptionInformation[0] = Rax;
|
||||
// Note that ExceptionAddress doesn't equal the reported context RIP here, this discrepancy expected and not having it can trigger anti-debug logic.
|
||||
return Dst;
|
||||
default: LogMan::Msg::EFmt("Unknown interrupt: 0x{:X}", Fault.err_code >> 3); break;
|
||||
|
||||
@@ -0,0 +1,290 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <windef.h>
|
||||
#include <winbase.h>
|
||||
#define __WINESRC__
|
||||
#include <winternl.h>
|
||||
#include <libloaderapi.h>
|
||||
#include "FEXUnixLib.h"
|
||||
#include "Priv.h"
|
||||
|
||||
#define PR_SET_VMA 0x53564d41
|
||||
#define PR_SET_VMA_ANON_NAME 0
|
||||
|
||||
#define MADV_HUGEPAGE 14
|
||||
#define MADV_NOHUGEPAGE 15
|
||||
|
||||
extern "C" IMAGE_DOS_HEADER __ImageBase;
|
||||
|
||||
namespace FEX::Windows::UnixLib {
|
||||
static bool UsingNTQueryPath {};
|
||||
static bool SupportsVirtualName {true};
|
||||
unixlib_handle_t UnixLibHandle {};
|
||||
|
||||
decltype(__wine_unix_call_dispatcher) UnixCallDispatcher {};
|
||||
|
||||
#ifdef ARCHITECTURE_arm64ec
|
||||
// On ARM64EC, indirect calls go through __os_arm64x_dispatch_icall which invokes
|
||||
// FEX's custom call checker. Use a naked trampoline to bypass the dispatch mechanism
|
||||
// and call the unix dispatcher directly via a register branch.
|
||||
static decltype(__wine_unix_call_dispatcher) UnixCallDispatcherDirect {};
|
||||
|
||||
static NTSTATUS __attribute__((naked)) TrampolineCall(unixlib_handle_t, unsigned int, void*) {
|
||||
asm(R"(
|
||||
adrp x16, %[Displace];
|
||||
ldr x16, [x16, #:lo12:%[Displace]]
|
||||
br x16
|
||||
)" ::[Displace] "S"(&UnixCallDispatcherDirect)
|
||||
: "memory");
|
||||
}
|
||||
#endif
|
||||
// This code path will eventually crash once Wine and the kernel implements `userspace syscall dispatch`.
|
||||
// FEX will need to switch over to using WINE's unixlib syscall approach then.
|
||||
namespace Illegal {
|
||||
__attribute__((naked)) uint64_t prctl(int op, uint64_t attr, const void* addr, size_t size, const char* Name) {
|
||||
asm volatile(R"(
|
||||
mov x8, 167; // prctl
|
||||
svc #0;
|
||||
ret;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
|
||||
__attribute__((naked)) uint64_t madvise(const void* addr, size_t size, int advice) {
|
||||
asm volatile(R"(
|
||||
mov x8, 233; // madvise
|
||||
svc #0;
|
||||
ret;
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
|
||||
__attribute__((naked)) uint64_t linux_getpid() {
|
||||
asm volatile(R"(
|
||||
mov x8, 172;
|
||||
svc #0;
|
||||
ret;
|
||||
)" ::
|
||||
: "r0", "r8");
|
||||
}
|
||||
|
||||
void VirtualName(const char* Name, const void* Ptr, size_t Size) {
|
||||
if (SupportsVirtualName) {
|
||||
auto Result = prctl(PR_SET_VMA, PR_SET_VMA_ANON_NAME, Ptr, Size, Name);
|
||||
if (Result != 0) {
|
||||
// Disable any additional attempts.
|
||||
SupportsVirtualName = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void VirtualTHPControl(const void* Ptr, size_t Size, FEXCore::Allocator::THPControl Control) {
|
||||
madvise(Ptr, Size, Control == FEXCore::Allocator::THPControl::Disable ? MADV_NOHUGEPAGE : MADV_HUGEPAGE);
|
||||
}
|
||||
} // namespace Illegal
|
||||
|
||||
bool Init(HMODULE NtDll) {
|
||||
const auto Sym = GetProcAddress(NtDll, "__wine_unix_call_dispatcher");
|
||||
|
||||
if (!Sym) {
|
||||
return false;
|
||||
}
|
||||
|
||||
auto TryNewWineMethod = []() {
|
||||
#ifdef ARCHITECTURE_arm64ec
|
||||
auto Name = InitUnicodeString(L"libarm64ecfex");
|
||||
#else
|
||||
auto Name = InitUnicodeString(L"libwow64fex");
|
||||
#endif
|
||||
|
||||
// Not supported in Proton at all, but supported in upstream WINE.
|
||||
uint64_t Result[2];
|
||||
if (NtQueryVirtualMemory(NtCurrentProcess(), &Name, MemoryWineLoadUnixLibByName, Result, sizeof(Result), nullptr)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Result[0] = unixlib_module_t
|
||||
// Result[1] = unixlib_handle_t
|
||||
// Module is ignored as it's only used to unload.
|
||||
UnixLibHandle = Result[1];
|
||||
return true;
|
||||
};
|
||||
|
||||
auto TryOldWineMethod = []() {
|
||||
// Supported in Proton 11 and Experimental (2026-06-26).
|
||||
return NtQueryVirtualMemory(NtCurrentProcess(), &__ImageBase, MemoryWineUnixFuncs, &UnixLibHandle, sizeof(UnixLibHandle), nullptr) == 0;
|
||||
};
|
||||
|
||||
if (!TryNewWineMethod() && !TryOldWineMethod()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
#ifdef ARCHITECTURE_arm64ec
|
||||
UnixCallDispatcherDirect = *reinterpret_cast<decltype(__wine_unix_call_dispatcher)*>(Sym);
|
||||
UnixCallDispatcher = TrampolineCall;
|
||||
#else
|
||||
UnixCallDispatcher = *reinterpret_cast<decltype(__wine_unix_call_dispatcher)*>(Sym);
|
||||
#endif
|
||||
|
||||
// Give a log saying that the unix lib was loaded.
|
||||
LogMan::Msg::IFmt("FEX: Loaded FEXUnixLib");
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool UnixLibAvailable() {
|
||||
return UnixLibHandle != 0;
|
||||
}
|
||||
|
||||
bool TryEnableHardwareTSO() {
|
||||
if (UnixLibAvailable()) {
|
||||
// UnixLib path.
|
||||
FEXUnixLib_SetHardwareTSOControlArgs Args {
|
||||
.Enable = true,
|
||||
};
|
||||
|
||||
return Call(FEXUnixLibFunctions::SetHardwareTSOControl, &Args) == STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
// Legacy Proton path.
|
||||
bool Enable = TRUE;
|
||||
NTSTATUS Status = NtSetInformationProcess(NtCurrentProcess(), ProcessFexHardwareTso, &Enable, sizeof(Enable));
|
||||
return Status == STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
bool SetKernelUnalignedAtomicControl(uint64_t Flags) {
|
||||
if (UnixLibAvailable()) {
|
||||
// UnixLib path.
|
||||
FEXUnixLib_SetKernelUnalignedAtomicControl Args {
|
||||
.Flags = Flags,
|
||||
};
|
||||
|
||||
return Call(FEXUnixLibFunctions::SetKernelUnalignedAtomicControl, &Args) == STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
// Legacy Proton path.
|
||||
return NtSetInformationProcess(NtCurrentProcess(), ProcessFexUnalignAtomic, &Flags, sizeof(Flags)) == STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
void VirtualTHPControl(const void* Ptr, size_t Size, FEXCore::Allocator::THPControl Control) {
|
||||
if (UnixLibAvailable()) {
|
||||
// UnixLib path.
|
||||
FEXUnixLib_Madvise Args {
|
||||
.Addr = Ptr,
|
||||
.Size = Size,
|
||||
.Advise = Control == FEXCore::Allocator::THPControl::Disable ? MADV_NOHUGEPAGE : MADV_HUGEPAGE,
|
||||
};
|
||||
|
||||
Call(FEXUnixLibFunctions::Madvise, &Args);
|
||||
return;
|
||||
}
|
||||
|
||||
// Legacy Proton path.
|
||||
Illegal::VirtualTHPControl(Ptr, Size, Control);
|
||||
}
|
||||
|
||||
void VirtualName(const char* Name, const void* Ptr, size_t Size) {
|
||||
if (!SupportsVirtualName) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (UnixLibAvailable()) {
|
||||
// UnixLib path.
|
||||
FEXUnixLib_SetVMAName Args {
|
||||
.Addr = Ptr,
|
||||
.Size = Size,
|
||||
.Name = Name,
|
||||
};
|
||||
|
||||
if (Call(FEXUnixLibFunctions::SetVMAName, &Args) != STATUS_SUCCESS) {
|
||||
SupportsVirtualName = false;
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
// Legacy Proton path.
|
||||
Illegal::VirtualName(Name, Ptr, Size);
|
||||
}
|
||||
|
||||
SHMSlotResult AllocateSHMSlots(void* SHMBase, uint32_t MapSize, uint32_t MaxSize) {
|
||||
if (UnixLibAvailable()) {
|
||||
// UnixLib path.
|
||||
FEXUnixLib_GetSHMStatsVMA Args {
|
||||
.SHMBase = SHMBase,
|
||||
.MapSize = MapSize,
|
||||
.MaxSize = MaxSize,
|
||||
};
|
||||
|
||||
if (Call(FEXUnixLibFunctions::GetSHMStatsVMA, &Args) == STATUS_SUCCESS) {
|
||||
return {
|
||||
.SHMBase = Args.SHMBase,
|
||||
.MappedSize = Args.MapSize,
|
||||
};
|
||||
}
|
||||
|
||||
return {};
|
||||
}
|
||||
|
||||
// Legacy Proton path.
|
||||
// Magic WINE+FEX path.
|
||||
if (SHMBase == nullptr || UsingNTQueryPath) {
|
||||
MEMORY_FEX_STATS_SHM_INFORMATION Info {
|
||||
.shm_base = SHMBase,
|
||||
.map_size = MapSize,
|
||||
.max_size = MaxSize,
|
||||
};
|
||||
size_t Length {};
|
||||
auto Result = NtQueryVirtualMemory(NtCurrentProcess(), nullptr, MemoryFexStatsShm, &Info, sizeof(Info), &Length);
|
||||
if (!Result) {
|
||||
UsingNTQueryPath = true;
|
||||
return {
|
||||
.SHMBase = Info.shm_base,
|
||||
.MappedSize = MapSize,
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
// Opaque handle path, doesn't support resizing.
|
||||
auto handle = CreateFile(fextl::fmt::format("/dev/shm/fex-{}-stats", Illegal::linux_getpid()).c_str(), GENERIC_READ | GENERIC_WRITE,
|
||||
FILE_SHARE_READ, nullptr, CREATE_ALWAYS, FILE_ATTRIBUTE_NORMAL, nullptr);
|
||||
|
||||
// Create the section mapping for the file handle for the full size.
|
||||
HANDLE SectionMapping;
|
||||
LARGE_INTEGER SectionSize {{MaxSize}};
|
||||
auto Result = NtCreateSection(&SectionMapping, SECTION_EXTEND_SIZE | SECTION_MAP_READ | SECTION_MAP_WRITE, nullptr, &SectionSize,
|
||||
PAGE_READWRITE, SEC_COMMIT, handle);
|
||||
if (Result != STATUS_SUCCESS) {
|
||||
CloseHandle(handle);
|
||||
return {};
|
||||
}
|
||||
|
||||
// Section mapping is used from now on.
|
||||
CloseHandle(handle);
|
||||
|
||||
// Now actually map the view of the section.
|
||||
void* Base = nullptr;
|
||||
size_t FullSize = MaxSize;
|
||||
Result = NtMapViewOfSection(SectionMapping, NtCurrentProcess(), &Base, 0, 0, nullptr, &FullSize, ViewUnmap, MEM_RESERVE | MEM_TOP_DOWN,
|
||||
PAGE_READWRITE);
|
||||
if (Result != STATUS_SUCCESS) {
|
||||
CloseHandle(SectionMapping);
|
||||
return {};
|
||||
}
|
||||
|
||||
return {.SHMBase = Base, .MappedSize = MaxSize};
|
||||
}
|
||||
|
||||
void DeleteSHMStatsFile() {
|
||||
if (UnixLibAvailable()) {
|
||||
// UnixLib path.
|
||||
Call(FEXUnixLibFunctions::DeleteSHMStatsFile, nullptr);
|
||||
return;
|
||||
}
|
||||
|
||||
// Legacy Proton path.
|
||||
DeleteFile(fextl::fmt::format("/dev/shm/fex-{}-stats", Illegal::linux_getpid()).c_str());
|
||||
}
|
||||
|
||||
} // namespace FEX::Windows::UnixLib
|
||||
@@ -0,0 +1,48 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <cstdint>
|
||||
#include <FEXCore/Utils/AllocatorHooks.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include "../UnixLib/FEXUnixLib.h"
|
||||
#include "wine/unixlib.h"
|
||||
|
||||
namespace FEX::Windows::UnixLib {
|
||||
extern decltype(__wine_unix_call_dispatcher) UnixCallDispatcher;
|
||||
extern unixlib_handle_t UnixLibHandle;
|
||||
|
||||
inline NTSTATUS Call(FEXUnixLibFunctions code, void* args) {
|
||||
if (!UnixCallDispatcher) {
|
||||
return STATUS_NOT_SUPPORTED;
|
||||
}
|
||||
|
||||
return UnixCallDispatcher(UnixLibHandle, FEXCore::ToUnderlying(code), args);
|
||||
}
|
||||
|
||||
bool Init(HMODULE NtDll);
|
||||
|
||||
/**
|
||||
* @brief Tries to enable hardware TSO if supported.
|
||||
*
|
||||
* @return true if hardware TSO is supported and enabled.
|
||||
*/
|
||||
bool TryEnableHardwareTSO();
|
||||
|
||||
/**
|
||||
* @brief Tries to enable kernel unaligned atomic handling if supported.
|
||||
*
|
||||
* @param Flags Which unaligned types to enable
|
||||
*
|
||||
* @return true if the flags were enabled
|
||||
*/
|
||||
bool SetKernelUnalignedAtomicControl(uint64_t Flags);
|
||||
|
||||
void VirtualTHPControl(const void* Ptr, size_t Size, FEXCore::Allocator::THPControl Control);
|
||||
void VirtualName(const char* Name, const void* Ptr, size_t Size);
|
||||
|
||||
struct SHMSlotResult {
|
||||
void* SHMBase;
|
||||
uint32_t MappedSize;
|
||||
};
|
||||
SHMSlotResult AllocateSHMSlots(void* SHMBase, uint32_t MapSize, uint32_t MaxSize);
|
||||
void DeleteSHMStatsFile();
|
||||
} // namespace FEX::Windows::UnixLib
|
||||
@@ -20,9 +20,14 @@ public:
|
||||
}
|
||||
|
||||
~ScopedHandle() {
|
||||
reset(INVALID_HANDLE_VALUE);
|
||||
}
|
||||
|
||||
void reset(HANDLE NewHandle = INVALID_HANDLE_VALUE) {
|
||||
if (Handle != INVALID_HANDLE_VALUE) {
|
||||
NtClose(Handle);
|
||||
}
|
||||
Handle = NewHandle;
|
||||
}
|
||||
|
||||
const HANDLE& operator*() const {
|
||||
|
||||
@@ -172,7 +172,7 @@ FEXCore::ExecutableFileSectionInfo ImageTracker::HandleImageMap(std::string_view
|
||||
|
||||
auto AOTImage = AOTImages.find(ID);
|
||||
if (AOTImage != AOTImages.end()) {
|
||||
CTX.GetCodeCache().LoadData(nullptr, AOTImage->second.Data, ImageInfo->SectionInfo);
|
||||
// TODO: CodeCache::EnableLoadedSection
|
||||
}
|
||||
}
|
||||
|
||||
@@ -180,7 +180,10 @@ FEXCore::ExecutableFileSectionInfo ImageTracker::HandleImageMap(std::string_view
|
||||
fextl::set<uint64_t> VolatileInstructions {};
|
||||
FEXCore::IntervalList<uint64_t> VolatileValidRanges {};
|
||||
LoadImageVolatileMetadata(VolatileInstructions, VolatileValidRanges, Module, Nt, Address, EndAddress);
|
||||
if (auto It = ExtendedMetaData.find(ModuleName); It != ExtendedMetaData.end()) {
|
||||
if (auto It = ExtendedMetaData.find(ID); It != ExtendedMetaData.end()) {
|
||||
FEX::VolatileMetadata::ApplyFEXExtendedVolatileMetadata(It->second, VolatileInstructions, VolatileValidRanges, Address, EndAddress);
|
||||
}
|
||||
if (auto It = ExtendedMetaData.find(fextl::string {ModuleName}); It != ExtendedMetaData.end()) {
|
||||
FEX::VolatileMetadata::ApplyFEXExtendedVolatileMetadata(It->second, VolatileInstructions, VolatileValidRanges, Address, EndAddress);
|
||||
}
|
||||
|
||||
@@ -277,6 +280,7 @@ void ImageTracker::LoadAOTImages(MappedImageInfo& ImageInfo) {
|
||||
RtlUnicodeToMultiByteN(UniqueId.data(), AnsiLength, NULL, Info->FileName, Info->FileNameLength);
|
||||
|
||||
AOTImages[UniqueId] = {.Data = static_cast<std::byte*>(LoadAddress)};
|
||||
// TODO: CodeCache::LoadCache, CodeCache::RegisterMappedCodeBuffer
|
||||
LogMan::Msg::IFmt("Loaded cache: {}", UniqueId);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -30,6 +30,19 @@ InvalidationTracker::InvalidationTracker(FEXCore::Context::Context& CTX, const s
|
||||
}
|
||||
}
|
||||
|
||||
static bool ProtHasExec(ULONG Prot) {
|
||||
return (Prot & (PAGE_EXECUTE | PAGE_EXECUTE_READ | PAGE_EXECUTE_READWRITE | PAGE_EXECUTE_WRITECOPY)) != 0;
|
||||
}
|
||||
|
||||
static bool ProtIsReadable(ULONG Prot) {
|
||||
return (Prot & (PAGE_READONLY | PAGE_READWRITE | PAGE_WRITECOPY | PAGE_EXECUTE | PAGE_EXECUTE_READ | PAGE_EXECUTE_READWRITE |
|
||||
PAGE_EXECUTE_WRITECOPY)) != 0;
|
||||
}
|
||||
|
||||
static bool ProtIsWritable(ULONG Prot) {
|
||||
return (Prot & (PAGE_READWRITE | PAGE_WRITECOPY | PAGE_EXECUTE_READWRITE | PAGE_EXECUTE_WRITECOPY)) != 0;
|
||||
}
|
||||
|
||||
void InvalidationTracker::HandleMemoryProtectionNotification(uint64_t Address, uint64_t Size, ULONG Prot) {
|
||||
const auto AlignedBase = Address & FEXCore::Utils::FEX_PAGE_MASK;
|
||||
const auto AlignedSize = (Address - AlignedBase + Size + FEXCore::Utils::FEX_PAGE_SIZE - 1) & FEXCore::Utils::FEX_PAGE_MASK;
|
||||
@@ -38,16 +51,27 @@ void InvalidationTracker::HandleMemoryProtectionNotification(uint64_t Address, u
|
||||
std::unique_lock Lock(IntervalsLock);
|
||||
|
||||
FEXCore::IntervalList<uint64_t>::Interval ProtInterval {AlignedBase, AlignedBase + AlignedSize};
|
||||
if (Prot & (PAGE_EXECUTE | PAGE_EXECUTE_READ | PAGE_EXECUTE_READWRITE | PAGE_EXECUTE_WRITECOPY)) {
|
||||
|
||||
const bool HasExec = ProtHasExec(Prot);
|
||||
const bool EffectiveExec = HasExec || (DEPDisabled && ProtIsReadable(Prot));
|
||||
const bool EffectiveRWX = EffectiveExec && ProtIsWritable(Prot);
|
||||
|
||||
if (EffectiveExec) {
|
||||
XIntervals.Insert(ProtInterval);
|
||||
if (Prot & (PAGE_EXECUTE_WRITECOPY | PAGE_EXECUTE_READWRITE)) {
|
||||
if (EffectiveRWX) {
|
||||
LogMan::Msg::DFmt("Add SMC interval: {:X} - {:X}", AlignedBase, AlignedBase + AlignedSize);
|
||||
RWXIntervals.Insert(ProtInterval);
|
||||
}
|
||||
if (DEPDisabled && !HasExec) {
|
||||
DEPPromotedIntervals.Insert(ProtInterval);
|
||||
}
|
||||
return true;
|
||||
} else if (XIntervals.Intersect(ProtInterval)) {
|
||||
XIntervals.Remove(ProtInterval);
|
||||
RWXIntervals.Remove(ProtInterval);
|
||||
if (DEPDisabled) {
|
||||
DEPPromotedIntervals.Remove(ProtInterval);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -60,6 +84,53 @@ void InvalidationTracker::HandleMemoryProtectionNotification(uint64_t Address, u
|
||||
}
|
||||
}
|
||||
|
||||
void InvalidationTracker::HandleProcessExecuteFlagsChange(ULONG Flags) {
|
||||
const bool DisableDEP = (Flags & MEM_EXECUTE_OPTION_ENABLE) != 0;
|
||||
|
||||
std::scoped_lock CodeLock(CTX.GetCodeInvalidationMutex());
|
||||
std::unique_lock Lock(IntervalsLock);
|
||||
|
||||
if (DisableDEP == DEPDisabled) {
|
||||
return;
|
||||
}
|
||||
|
||||
DEPDisabled = DisableDEP;
|
||||
|
||||
if (DisableDEP) {
|
||||
DEPPromotedIntervals.Clear();
|
||||
|
||||
MEMORY_BASIC_INFORMATION Info;
|
||||
uint64_t Address = 0;
|
||||
|
||||
while (VirtualQuery(reinterpret_cast<LPCVOID>(Address), &Info, sizeof(Info))) {
|
||||
uint64_t BaseAddress = reinterpret_cast<uint64_t>(Info.BaseAddress);
|
||||
if (Info.State == MEM_COMMIT && ProtIsReadable(Info.Protect) && !ProtHasExec(Info.Protect)) {
|
||||
const auto AlignedBase = BaseAddress & FEXCore::Utils::FEX_PAGE_MASK;
|
||||
const auto AlignedSize = (BaseAddress - AlignedBase + Info.RegionSize + FEXCore::Utils::FEX_PAGE_SIZE - 1) & FEXCore::Utils::FEX_PAGE_MASK;
|
||||
FEXCore::IntervalList<uint64_t>::Interval ProtInterval {AlignedBase, AlignedBase + AlignedSize};
|
||||
|
||||
XIntervals.Insert(ProtInterval);
|
||||
if (ProtIsWritable(Info.Protect)) {
|
||||
RWXIntervals.Insert(ProtInterval);
|
||||
}
|
||||
DEPPromotedIntervals.Insert(ProtInterval);
|
||||
}
|
||||
|
||||
Address = BaseAddress + Info.RegionSize;
|
||||
}
|
||||
} else {
|
||||
for (const auto& Interval : DEPPromotedIntervals) {
|
||||
XIntervals.Remove(Interval);
|
||||
RWXIntervals.Remove(Interval);
|
||||
}
|
||||
DEPPromotedIntervals.Clear();
|
||||
}
|
||||
|
||||
// Invalidate all cached code: previously-compiled blocks may contain NoExec stubs for addresses
|
||||
// that are now executable (or reference regions whose executability just changed).
|
||||
InvalidateIntervalInternalLocked(0, std::numeric_limits<uint64_t>::max());
|
||||
}
|
||||
|
||||
void InvalidationTracker::HandleImageMap(std::string_view Name, uint64_t Address) {
|
||||
auto* Nt = RtlImageNtHeader(reinterpret_cast<HMODULE>(Address));
|
||||
auto* SectionsBegin = IMAGE_FIRST_SECTION(Nt);
|
||||
@@ -146,9 +217,12 @@ void InvalidationTracker::ReprotectRWXIntervals(uint64_t Address, uint64_t Size)
|
||||
}
|
||||
|
||||
bool InvalidationTracker::HandleRWXAccessViolation(FEXCore::Core::InternalThreadState* Thread, uint64_t HostPc, uint64_t FaultAddress) {
|
||||
const bool NeedsInvalidate = [&](uint64_t Address) {
|
||||
const auto [NeedsInvalidate, UntrapProt] = [&](uint64_t Address) -> std::pair<bool, ULONG> {
|
||||
std::shared_lock Lock(IntervalsLock);
|
||||
return RWXIntervals.Query(Address).Enclosed;
|
||||
if (!RWXIntervals.Query(Address).Enclosed) {
|
||||
return {false, 0};
|
||||
}
|
||||
return {true, GetUntrapProt(Address)};
|
||||
}(FaultAddress);
|
||||
|
||||
if (NeedsInvalidate) {
|
||||
@@ -162,7 +236,7 @@ bool InvalidationTracker::HandleRWXAccessViolation(FEXCore::Core::InternalThread
|
||||
ULONG TmpProt;
|
||||
void* TmpAddress = reinterpret_cast<void*>(FaultAddress);
|
||||
SIZE_T TmpSize = 1;
|
||||
NtProtectVirtualMemory(NtCurrentProcess(), &TmpAddress, &TmpSize, PAGE_EXECUTE_READWRITE, &TmpProt);
|
||||
NtProtectVirtualMemory(NtCurrentProcess(), &TmpAddress, &TmpSize, UntrapProt, &TmpProt);
|
||||
}
|
||||
DetectMonoBackpatcherBlock(Thread, HostPc);
|
||||
return true;
|
||||
@@ -223,7 +297,7 @@ void InvalidationTracker::DisableSMCDetection() {
|
||||
SMCDetectionDisabled = true;
|
||||
uint64_t Address = 0;
|
||||
|
||||
// Reprotect all RWX intervals as RWX
|
||||
// Reprotect all RWX intervals as writable
|
||||
FEXCore::IntervalList<uint64_t>::QueryResult Query;
|
||||
do {
|
||||
Query = RWXIntervals.Query(Address);
|
||||
@@ -231,12 +305,26 @@ void InvalidationTracker::DisableSMCDetection() {
|
||||
void* TmpAddress = reinterpret_cast<void*>(Address);
|
||||
SIZE_T TmpSize = static_cast<SIZE_T>(Query.Size);
|
||||
ULONG TmpProt;
|
||||
NtProtectVirtualMemory(NtCurrentProcess(), &TmpAddress, &TmpSize, PAGE_EXECUTE_READWRITE, &TmpProt);
|
||||
NtProtectVirtualMemory(NtCurrentProcess(), &TmpAddress, &TmpSize, GetUntrapProt(Address), &TmpProt);
|
||||
}
|
||||
Address += Query.Size;
|
||||
} while (Query.Size);
|
||||
}
|
||||
|
||||
ULONG InvalidationTracker::GetTrapProt(uint64_t Address) const {
|
||||
if (DEPDisabled && DEPPromotedIntervals.Query(Address).Enclosed) {
|
||||
return PAGE_READONLY;
|
||||
}
|
||||
return PAGE_EXECUTE_READ;
|
||||
}
|
||||
|
||||
ULONG InvalidationTracker::GetUntrapProt(uint64_t Address) const {
|
||||
if (DEPDisabled && DEPPromotedIntervals.Query(Address).Enclosed) {
|
||||
return PAGE_READWRITE;
|
||||
}
|
||||
return PAGE_EXECUTE_READWRITE;
|
||||
}
|
||||
|
||||
void InvalidationTracker::InvalidateIntervalInternal(uint64_t Address, uint64_t Size) {
|
||||
std::scoped_lock CodeLock(CTX.GetCodeInvalidationMutex());
|
||||
InvalidateIntervalInternalLocked(Address, Size);
|
||||
@@ -275,7 +363,7 @@ bool InvalidationTracker::ProtectRWXIntervalsInternal(uint64_t Address, uint64_t
|
||||
void* TmpAddress = reinterpret_cast<void*>(Address);
|
||||
SIZE_T TmpSize = static_cast<SIZE_T>(std::min(End, Address + Query.Size) - Address);
|
||||
ULONG TmpProt;
|
||||
NtProtectVirtualMemory(NtCurrentProcess(), &TmpAddress, &TmpSize, ForWriteLocked ? PAGE_EXECUTE_READWRITE : PAGE_EXECUTE_READ, &TmpProt);
|
||||
NtProtectVirtualMemory(NtCurrentProcess(), &TmpAddress, &TmpSize, ForWriteLocked ? GetUntrapProt(Address) : GetTrapProt(Address), &TmpProt);
|
||||
} else if (!Query.Size) {
|
||||
// No more regions past `Address` in the interval list
|
||||
break;
|
||||
|
||||
@@ -24,6 +24,7 @@ class InvalidationTracker {
|
||||
public:
|
||||
InvalidationTracker(FEXCore::Context::Context& CTX, const std::unordered_map<DWORD, FEXCore::Core::InternalThreadState*>& Threads);
|
||||
void HandleMemoryProtectionNotification(uint64_t Address, uint64_t Size, ULONG Prot);
|
||||
void HandleProcessExecuteFlagsChange(ULONG Flags);
|
||||
void HandleImageMap(std::string_view Name, uint64_t Address);
|
||||
struct InvalidateContainingSectionResult {
|
||||
uint64_t SectionStart;
|
||||
@@ -51,12 +52,20 @@ private:
|
||||
// and any code in the range will be invalidated before protection as RWX, otherwise protects as RX if false.
|
||||
bool ProtectRWXIntervalsInternal(uint64_t Address, uint64_t Size, bool ForWriteLocked);
|
||||
|
||||
// Returns the correct protection for trapping (removing write) or untrapping (restoring write) an RWX interval.
|
||||
// For DEP-promoted regions (originally non-exec), uses PAGE_READONLY/PAGE_READWRITE instead of PAGE_EXECUTE_READ/PAGE_EXECUTE_READWRITE.
|
||||
// NOTE: Must be called with IntervalsLock held.
|
||||
ULONG GetTrapProt(uint64_t Address) const;
|
||||
ULONG GetUntrapProt(uint64_t Address) const;
|
||||
|
||||
FEXCore::IntervalList<uint64_t> XIntervals;
|
||||
FEXCore::IntervalList<uint64_t> RWXIntervals;
|
||||
std::shared_mutex IntervalsLock;
|
||||
FEXCore::Context::Context& CTX;
|
||||
const std::unordered_map<DWORD, FEXCore::Core::InternalThreadState*>& Threads;
|
||||
bool SMCDetectionDisabled {false}; // Protected by IntervalsLock
|
||||
bool SMCDetectionDisabled {false}; // Protected by IntervalsLock
|
||||
bool DEPDisabled {false}; // Protected by IntervalsLock
|
||||
FEXCore::IntervalList<uint64_t> DEPPromotedIntervals; // Protected by IntervalsLock
|
||||
|
||||
bool MonoBackpatcherDetectionPending {false};
|
||||
uint64_t MonoBase {0};
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Windows/Common/FEXUnixLib.h"
|
||||
#include "Windows/Common/SHMStats.h"
|
||||
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
@@ -12,88 +13,31 @@
|
||||
#include <wine/debug.h>
|
||||
|
||||
namespace FEX::Windows {
|
||||
__attribute__((naked)) uint64_t linux_getpid() {
|
||||
asm volatile(R"(
|
||||
mov x8, 172;
|
||||
svc #0;
|
||||
ret;
|
||||
)" ::
|
||||
: "r0", "r8");
|
||||
}
|
||||
|
||||
uint32_t StatAlloc::FrontendAllocateSlots(uint32_t NewSize) {
|
||||
if (CurrentSize == MAX_STATS_SIZE || !UsingNTQueryPath) {
|
||||
if (CurrentSize == MAX_STATS_SIZE) {
|
||||
LogMan::Msg::DFmt("Ran out of slots. Can't allocate more");
|
||||
return CurrentSize;
|
||||
}
|
||||
|
||||
MEMORY_FEX_STATS_SHM_INFORMATION Info {
|
||||
.shm_base = nullptr,
|
||||
.map_size = std::min(CurrentSize * 2, MAX_STATS_SIZE),
|
||||
.max_size = MAX_STATS_SIZE,
|
||||
};
|
||||
size_t Length {};
|
||||
auto Result = NtQueryVirtualMemory(NtCurrentProcess(), nullptr, MemoryFexStatsShm, &Info, sizeof(Info), &Length);
|
||||
if (!Result) {
|
||||
CurrentSize = Info.map_size;
|
||||
}
|
||||
auto Result = UnixLib::AllocateSHMSlots(Base, std::min(NewSize, MAX_STATS_SIZE), MAX_STATS_SIZE);
|
||||
|
||||
return CurrentSize;
|
||||
// Return the new size allocated, if it happened to change.
|
||||
return Result.MappedSize;
|
||||
}
|
||||
|
||||
StatAlloc::StatAlloc(FEXCore::SHMStats::AppType AppType) {
|
||||
// Try wine+fex magic path.
|
||||
auto Result = UnixLib::AllocateSHMSlots(Base, FEXCore::Utils::FEX_PAGE_SIZE, MAX_STATS_SIZE);
|
||||
|
||||
{
|
||||
MEMORY_FEX_STATS_SHM_INFORMATION Info {
|
||||
.shm_base = nullptr,
|
||||
.map_size = FEXCore::Utils::FEX_PAGE_SIZE,
|
||||
.max_size = MAX_STATS_SIZE,
|
||||
};
|
||||
size_t Length {};
|
||||
auto Result = NtQueryVirtualMemory(NtCurrentProcess(), nullptr, MemoryFexStatsShm, &Info, sizeof(Info), &Length);
|
||||
if (!Result) {
|
||||
UsingNTQueryPath = true;
|
||||
CurrentSize = Info.map_size;
|
||||
Base = Info.shm_base;
|
||||
SaveHeader(AppType);
|
||||
return;
|
||||
}
|
||||
}
|
||||
CurrentSize = MAX_STATS_SIZE;
|
||||
|
||||
auto handle = CreateFile(fextl::fmt::format("/dev/shm/fex-{}-stats", linux_getpid()).c_str(), GENERIC_READ | GENERIC_WRITE,
|
||||
FILE_SHARE_READ, nullptr, CREATE_ALWAYS, FILE_ATTRIBUTE_NORMAL, nullptr);
|
||||
|
||||
// Create the section mapping for the file handle for the full size.
|
||||
HANDLE SectionMapping;
|
||||
LARGE_INTEGER SectionSize {{MAX_STATS_SIZE}};
|
||||
auto Result = NtCreateSection(&SectionMapping, SECTION_EXTEND_SIZE | SECTION_MAP_READ | SECTION_MAP_WRITE, nullptr, &SectionSize,
|
||||
PAGE_READWRITE, SEC_COMMIT, handle);
|
||||
if (Result != 0) {
|
||||
CloseHandle(handle);
|
||||
if (Result.SHMBase) {
|
||||
CurrentSize = Result.MappedSize;
|
||||
Base = Result.SHMBase;
|
||||
SaveHeader(AppType);
|
||||
return;
|
||||
}
|
||||
|
||||
// Section mapping is used from now on.
|
||||
CloseHandle(handle);
|
||||
|
||||
// Now actually map the view of the section.
|
||||
Base = 0;
|
||||
size_t FullSize = MAX_STATS_SIZE;
|
||||
Result = NtMapViewOfSection(SectionMapping, NtCurrentProcess(), &Base, 0, 0, nullptr, &FullSize, ViewUnmap, MEM_RESERVE | MEM_TOP_DOWN,
|
||||
PAGE_READWRITE);
|
||||
if (Result != 0) {
|
||||
CloseHandle(SectionMapping);
|
||||
return;
|
||||
}
|
||||
|
||||
// Once WINE supports NtExtendSection and SECTION_EXTEND_SIZE correctly then we can map/commit a single page, map the full MAX_STATS_SIZE
|
||||
// view as reserved, and extend the view using NtExtendSection.
|
||||
SaveHeader(AppType);
|
||||
}
|
||||
|
||||
StatAlloc::~StatAlloc() {
|
||||
DeleteFile(fextl::fmt::format("/dev/shm/fex-{}-stats", linux_getpid()).c_str());
|
||||
UnixLib::DeleteSHMStatsFile();
|
||||
}
|
||||
|
||||
} // namespace FEX::Windows
|
||||
@@ -23,7 +23,6 @@ public:
|
||||
|
||||
private:
|
||||
uint32_t FrontendAllocateSlots(uint32_t NewSize) override;
|
||||
bool UsingNTQueryPath {};
|
||||
};
|
||||
|
||||
} // namespace FEX::Windows
|
||||
@@ -5,6 +5,8 @@
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/ArchHelpers/Arm64.h>
|
||||
|
||||
#include "Windows/Common/FEXUnixLib.h"
|
||||
|
||||
namespace FEX::Windows {
|
||||
class TSOHandlerConfig final {
|
||||
public:
|
||||
@@ -15,18 +17,14 @@ public:
|
||||
UnalignedHandlerType = FEXCore::ArchHelpers::Arm64::UnalignedHandlerType::NonAtomic;
|
||||
}
|
||||
|
||||
if (TSOEnabled()) {
|
||||
BOOL Enable = TRUE;
|
||||
NTSTATUS Status = NtSetInformationProcess(NtCurrentProcess(), ProcessFexHardwareTso, &Enable, sizeof(Enable));
|
||||
if (Status == STATUS_SUCCESS) {
|
||||
CTX.SetHardwareTSOSupport(true);
|
||||
}
|
||||
if (TSOEnabled() && FEX::Windows::UnixLib::TryEnableHardwareTSO()) {
|
||||
CTX.SetHardwareTSOSupport(true);
|
||||
}
|
||||
|
||||
uint64_t Flags = (StrictInProcessSplitLocks() ? FEX_UNALIGN_ATOMIC_STRICT_SPLIT_LOCKS : 0) |
|
||||
(KernelUnalignedAtomicBackpatching() ? FEX_UNALIGN_ATOMIC_BACKPATCH : 0) | FEX_UNALIGN_ATOMIC_EMULATE;
|
||||
|
||||
if (NtSetInformationProcess(NtCurrentProcess(), ProcessFexUnalignAtomic, &Flags, sizeof(Flags)) == STATUS_SUCCESS) {
|
||||
if (UnixLib::SetKernelUnalignedAtomicControl(Flags)) {
|
||||
LogMan::Msg::IFmt("FEX: Kernel unaligned atomics enabled!");
|
||||
}
|
||||
}
|
||||
|
||||
Loaded 100 of 400 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user