mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-07 16:00:18 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
fd3e988a20 | ||
|
|
b1d98f4e58 | ||
|
|
9e7daf61d0 | ||
|
|
6fbe25753b | ||
|
|
03f0edc5b5 | ||
|
|
5536f1e835 | ||
|
|
0de36706da | ||
|
|
17722dad6d | ||
|
|
0371599996 | ||
|
|
199649b30f | ||
|
|
4ef35488db | ||
|
|
70a91ee6ce | ||
|
|
418a27e47e | ||
|
|
61c76d02cc | ||
|
|
d98641221d | ||
|
|
8a14f87a44 | ||
|
|
02ce71734c | ||
|
|
96c2743280 | ||
|
|
7bfc34b51c | ||
|
|
40d820fd05 | ||
|
|
d69287aaf7 | ||
|
|
d475b0ba9e | ||
|
|
1638b744b7 | ||
|
|
8b19894a06 | ||
|
|
d2e0dc99de | ||
|
|
d04e40b5fd | ||
|
|
75d797b5cd | ||
|
|
ecf4891087 | ||
|
|
0e1a418678 | ||
|
|
5bef13df94 | ||
|
|
d8386121a8 | ||
|
|
000677abb6 | ||
|
|
64eb87e9b5 | ||
|
|
aa5e92bee2 | ||
|
|
0bf79dc5d6 | ||
|
|
adb2171c0a | ||
|
|
d6f8923f86 | ||
|
|
cf91ab9d5f | ||
|
|
a0fb9531db | ||
|
|
eca9353b28 | ||
|
|
a259730639 | ||
|
|
2e93d10eba | ||
|
|
70a3ceb64e | ||
|
|
b726f60afd | ||
|
|
2fa1a64999 | ||
|
|
a42b659af9 | ||
|
|
004c3230a4 | ||
|
|
2332c41510 | ||
|
|
ec3039c5a2 | ||
|
|
639d6e6071 | ||
|
|
cd518d4726 | ||
|
|
b7d9c00dff | ||
|
|
4b17575f5a | ||
|
|
5ba4bba138 | ||
|
|
b3ee5dba0f | ||
|
|
d87ff5afa9 | ||
|
|
62a24bd38f | ||
|
|
8d373c15b8 | ||
|
|
671f3e74a4 | ||
|
|
7e810233d9 | ||
|
|
4700dbd676 | ||
|
|
74e18f4317 | ||
|
|
1eea95cf18 | ||
|
|
7291b10727 | ||
|
|
6804916697 | ||
|
|
819e61bf14 | ||
|
|
b8f7e4c8ec | ||
|
|
17bcc0eed4 | ||
|
|
13003da289 | ||
|
|
9273538955 | ||
|
|
9750189def | ||
|
|
cb17ee9871 | ||
|
|
4c3b78ba9a | ||
|
|
e00b6a401b | ||
|
|
ac0ab8a7b4 | ||
|
|
0aff3941f4 | ||
|
|
27b022d4d9 | ||
|
|
e188928742 | ||
|
|
780e3c7fb7 | ||
|
|
1b5146d3ac | ||
|
|
80cf3ca6b9 | ||
|
|
2272b30a91 | ||
|
|
b5fb1cb07c | ||
|
|
4ea34a9c22 | ||
|
|
48e7de9f9e | ||
|
|
76dd2369a7 | ||
|
|
78e0cd6e77 | ||
|
|
5514a04cb4 | ||
|
|
99ca78b235 | ||
|
|
ab45db1665 | ||
|
|
bb38bcb67d | ||
|
|
6cc2912542 | ||
|
|
340b2ca624 | ||
|
|
5baa15de03 | ||
|
|
6ddca804d1 | ||
|
|
f9831a85fb | ||
|
|
7261033b7f | ||
|
|
3ad6866198 | ||
|
|
1c7d4165ab | ||
|
|
3e48b1a8ac | ||
|
|
07be100daf | ||
|
|
de9351eefb | ||
|
|
2c44b5b3a1 | ||
|
|
136f1e2fc7 | ||
|
|
ad39add55f | ||
|
|
f3c301e359 | ||
|
|
1c37a1b4d6 | ||
|
|
7222529904 | ||
|
|
f26eccd00f | ||
|
|
73375a76ac | ||
|
|
d4416d200e | ||
|
|
d1b235dd83 | ||
|
|
3ac5e0423a | ||
|
|
fc6de5f3c0 | ||
|
|
b1e475d81d | ||
|
|
4a09a4324f | ||
|
|
47f94327c5 | ||
|
|
a009ed0b6b | ||
|
|
2476a686e7 | ||
|
|
f0db93773f | ||
|
|
fabe824c8b | ||
|
|
78a077397e | ||
|
|
0c4b456aaa | ||
|
|
09185167bc | ||
|
|
5d78c3203c | ||
|
|
fa5322d3f9 | ||
|
|
d21aa5cac2 | ||
|
|
102d5c57cb | ||
|
|
ffb4de9fd9 | ||
|
|
6b3d8886e5 | ||
|
|
ddc10272a0 | ||
|
|
0b5ef00165 | ||
|
|
2a50416fc3 | ||
|
|
ce514d9f83 | ||
|
|
9b77e7fd13 | ||
|
|
abb44d3327 | ||
|
|
11eaf3d48a | ||
|
|
76c2cc2c3e | ||
|
|
0e6c8bd12e | ||
|
|
a9fb008317 | ||
|
|
e9f3a5b3e4 | ||
|
|
ebc45dff45 | ||
|
|
fc4a5ebfd3 | ||
|
|
1d7b688c55 | ||
|
|
f14a5ffbbf | ||
|
|
cada0d593c | ||
|
|
0436540791 | ||
|
|
a87ac86e18 | ||
|
|
1278b23150 | ||
|
|
02f5ea4b9d | ||
|
|
2e14e613d0 | ||
|
|
85c2889652 | ||
|
|
23dd056b60 | ||
|
|
71043e372a | ||
|
|
c412d073b9 | ||
|
|
8fb03ff1b9 | ||
|
|
c2b6aef6f4 | ||
|
|
24547318c6 | ||
|
|
b693112c80 | ||
|
|
48d1184066 | ||
|
|
8da9ebc2e0 | ||
|
|
5ba510474b | ||
|
|
51214d1be1 | ||
|
|
4d6e15d7af | ||
|
|
4721894427 | ||
|
|
d429865b6e | ||
|
|
ca5881a72c | ||
|
|
7151b9daff | ||
|
|
d9b5e28b22 | ||
|
|
aa7954a7d6 | ||
|
|
802c70d1ab | ||
|
|
8f905988e9 | ||
|
|
478c5595ad | ||
|
|
3977e1f29e | ||
|
|
2b1ef97354 | ||
|
|
eaddf7f1a5 | ||
|
|
3237de3085 | ||
|
|
df3d398d31 | ||
|
|
b44b3401b7 | ||
|
|
c28ca0fac9 | ||
|
|
235e2b6c2c | ||
|
|
d68b84bc27 | ||
|
|
edca528608 | ||
|
|
b75e8f2abf | ||
|
|
ec3158e4cd | ||
|
|
9c8c8041e0 | ||
|
|
e3adaacb51 | ||
|
|
c49e11484f | ||
|
|
7b4b9a80fa | ||
|
|
c7ad066987 | ||
|
|
f4d229f1ba | ||
|
|
25a8a00771 | ||
|
|
a67f7422b2 | ||
|
|
ed8150cfb6 | ||
|
|
462a163ba7 | ||
|
|
6374175a64 | ||
|
|
280b15ba2a | ||
|
|
2a0b488e99 | ||
|
|
3ef7c4ab51 | ||
|
|
e72d746036 | ||
|
|
0bb4091e34 | ||
|
|
6822fc595c | ||
|
|
a506a589dd | ||
|
|
684a5977dd | ||
|
|
ac3682e058 | ||
|
|
d5faf01f5a | ||
|
|
ecba1b6838 | ||
|
|
8d8b029285 | ||
|
|
bf6f855868 | ||
|
|
aa6a499329 | ||
|
|
0971650ef9 | ||
|
|
64c4fdccf7 | ||
|
|
d715ffbc8e | ||
|
|
aef801b5b5 | ||
|
|
6c9e29796b | ||
|
|
364bb3ac1e | ||
|
|
428ea68507 | ||
|
|
5cf59408a7 | ||
|
|
1596843015 | ||
|
|
af6582ff5b | ||
|
|
1799d4c675 | ||
|
|
808e1c0330 | ||
|
|
dacd96cab5 | ||
|
|
ca4d3bf64d | ||
|
|
ea38b043c1 | ||
|
|
a39746df2e | ||
|
|
2367a8e50b | ||
|
|
cb121d7f17 | ||
|
|
412793c21d | ||
|
|
ce2286c48b | ||
|
|
d5694d6de0 | ||
|
|
121218aa8a | ||
|
|
6bb53fa758 | ||
|
|
89aa0c5471 | ||
|
|
53fcbf6afa | ||
|
|
2f5643ae6b | ||
|
|
d162ac8b3d | ||
|
|
c9a704fbde | ||
|
|
9d5a822a3a | ||
|
|
50eba4066a | ||
|
|
6116ae5330 | ||
|
|
447226576f | ||
|
|
3f8b872f17 | ||
|
|
e573ddc2db | ||
|
|
fe9aa681f0 | ||
|
|
1b2f2c1559 | ||
|
|
84c75a86c3 | ||
|
|
eedbde6f15 | ||
|
|
4e441e5a08 | ||
|
|
3e287a36c2 | ||
|
|
eadc477695 | ||
|
|
59aa324678 | ||
|
|
219bce1467 | ||
|
|
46bde401bd | ||
|
|
01beac4956 | ||
|
|
fcd981e6b7 | ||
|
|
825833cfcc | ||
|
|
4a4c49bf68 | ||
|
|
763cea423a | ||
|
|
0fee355ff5 | ||
|
|
6f6f3c9dc5 | ||
|
|
25e5d88ab2 | ||
|
|
87013340bb | ||
|
|
47c075ccc9 | ||
|
|
b1a32d4ccf | ||
|
|
7af6a8dbdf | ||
|
|
383e99e4ef | ||
|
|
8c7cfc4d11 | ||
|
|
496ee730c8 | ||
|
|
71f7ff5101 | ||
|
|
1ea00f68a2 | ||
|
|
8df7c2d84f | ||
|
|
46557a7a1f | ||
|
|
d8c2a8271f | ||
|
|
cc4c705fc0 | ||
|
|
22f249fcf6 | ||
|
|
c262362a03 | ||
|
|
212df9aa7b | ||
|
|
691e39ec76 | ||
|
|
107cae2975 | ||
|
|
6742e0c376 | ||
|
|
8f70137b1a | ||
|
|
707db51b1b | ||
|
|
5b5fa1aa29 | ||
|
|
2b9cc9666a | ||
|
|
82eba22292 | ||
|
|
5c84e8f23c | ||
|
|
081b61677a | ||
|
|
9c54814b98 | ||
|
|
35d7b855ed | ||
|
|
0b8799274c | ||
|
|
ace2b737d8 | ||
|
|
ad85268524 | ||
|
|
832a320e22 | ||
|
|
83763df6fd | ||
|
|
bee868e9ba | ||
|
|
c41de81694 | ||
|
|
41aaeb1ff0 | ||
|
|
16be2792ab | ||
|
|
6610bb355c | ||
|
|
4312fd7291 | ||
|
|
89a225a96d | ||
|
|
a590977639 | ||
|
|
0d0d116bde | ||
|
|
341bdb5a54 | ||
|
|
2f3dbfb289 | ||
|
|
169cfbbeed | ||
|
|
f97a4afd8f | ||
|
|
868e4a6d81 | ||
|
|
6f48f7d3ac | ||
|
|
1ed3ecb409 | ||
|
|
2cb455b9d4 | ||
|
|
d4b5bf0f78 | ||
|
|
69013772c1 | ||
|
|
3448c83431 | ||
|
|
e4c84542ea | ||
|
|
acddc0323b | ||
|
|
c4285f0d30 | ||
|
|
977d6dd247 | ||
|
|
2bb27fffb7 | ||
|
|
0261ed353d | ||
|
|
96fecfd7c5 | ||
|
|
9fac1b8105 | ||
|
|
95fbcd7b9a | ||
|
|
6adf227611 | ||
|
|
b5cb429243 | ||
|
|
26ba8079a3 | ||
|
|
f999d30bc5 | ||
|
|
ecb1cc4ed4 | ||
|
|
5622bcae16 | ||
|
|
f4539ee289 | ||
|
|
d2138694b4 | ||
|
|
1767e21273 | ||
|
|
8f9d799342 | ||
|
|
a3b0b246f4 | ||
|
|
dee85f14fe | ||
|
|
27309114be | ||
|
|
704afed97b | ||
|
|
121f0a2c6c | ||
|
|
1c580ec92c | ||
|
|
ab8fc721a0 | ||
|
|
790447115c | ||
|
|
80abeac28a | ||
|
|
54915f87ce | ||
|
|
f34f1309a7 | ||
|
|
0ad52b7d19 | ||
|
|
809f60df06 | ||
|
|
cbdcd8253c | ||
|
|
7ac2cd7cc8 | ||
|
|
cf1bb1348c | ||
|
|
b60a26ff9e | ||
|
|
b805c07342 | ||
|
|
8d69f539ac | ||
|
|
c8c0054f67 | ||
|
|
1085385bbe | ||
|
|
56460b220c | ||
|
|
a583ebe590 | ||
|
|
ce8175a800 | ||
|
|
c987e1ef44 | ||
|
|
b36ec152d2 | ||
|
|
44c62e703e | ||
|
|
0f59c1d5e3 | ||
|
|
5739f0b459 | ||
|
|
4145fabfb6 |
No files matched your search
@@ -64,7 +64,7 @@ jobs:
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DENABLE_INTERPRETER=True -DBUILD_FEX_LINUX_TESTS=True -DBUILD_THUNKS=True
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DENABLE_INTERPRETER=True -DBUILD_FEX_LINUX_TESTS=True -DBUILD_THUNKS=True -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
@@ -188,6 +188,40 @@ jobs:
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ThunkgenTests.log || true
|
||||
|
||||
- name: Install
|
||||
if: matrix.arch[1] == 'x64'
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target install
|
||||
|
||||
- name: Test GL No-Thunks
|
||||
if: matrix.arch[1] == 'x64'
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
env:
|
||||
DISPLAY: ":0"
|
||||
run: cmake --build . --config $BUILD_TYPE --target thunk_functional_tests_nothunks
|
||||
|
||||
- name: No thunks Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_NoThunkResults.log || true
|
||||
|
||||
- name: Test GL Thunks
|
||||
if: matrix.arch[1] == 'x64'
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
env:
|
||||
DISPLAY: ":0"
|
||||
run: cmake --build . --config $BUILD_TYPE --target thunk_functional_tests_thunks
|
||||
|
||||
- name: Thunks Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ThunkResults.log || true
|
||||
|
||||
- name: Truncate test results
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
@@ -207,4 +241,3 @@ jobs:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
retention-days: 3
|
||||
|
||||
@@ -0,0 +1,118 @@
|
||||
name: Vixl Simulator run
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
pull_request:
|
||||
branches:
|
||||
- main
|
||||
|
||||
env:
|
||||
# Customize the CMake build type here (Release, Debug, RelWithDebInfo, etc.)
|
||||
BUILD_TYPE: Release
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ${{ matrix.arch }}
|
||||
strategy:
|
||||
matrix:
|
||||
# Only the x86-64 runner is fast enough to run this
|
||||
arch: [[self-hosted, x64], [self-hosted, ARMv8.4]]
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
|
||||
- name: Set rootfs paths
|
||||
run: |
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
|
||||
|
||||
- name : submodule checkout
|
||||
# Need to update submodules
|
||||
run: |
|
||||
git submodule sync --recursive
|
||||
git submodule update --init --depth 1
|
||||
|
||||
- name: Clean Build Environment
|
||||
run: rm -Rf ${{runner.workspace}}/build
|
||||
|
||||
- name: Create Build Environment
|
||||
# Some projects don't allow in-source building, so create a separate build directory
|
||||
# We'll use this as our working directory for all subsequent commands
|
||||
run: cmake -E make_directory ${{runner.workspace}}/build
|
||||
|
||||
- name: Configure CMake
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_VIXL_SIMULATOR=True -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the build. You can specify a specific target with "--target <NAME>"
|
||||
run: cmake --build . --config $BUILD_TYPE
|
||||
|
||||
- name: ASM Tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the unit tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target asm_tests
|
||||
|
||||
- name: ASM Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_ASM.log || true
|
||||
|
||||
- name: IR Tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the unit tests
|
||||
run: cmake --build . --config $BUILD_TYPE --target ir_tests
|
||||
|
||||
- name: IR Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_IR.log || true
|
||||
|
||||
- name: Truncate test results
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Cap out the log files at 20M in case something crash spins and dumps fault text
|
||||
# ASM tests get quite close to 10MB
|
||||
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
|
||||
|
||||
- name: Set runner name
|
||||
if: ${{ always() }}
|
||||
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v2'
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
retention-days: 3
|
||||
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"ThunksDB": {
|
||||
"GL": 1
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"ThunksDB": {
|
||||
"Vulkan": 1
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,21 @@
|
||||
if(NOT EXISTS "@CMAKE_BINARY_DIR@/install_manifest.txt")
|
||||
message(FATAL_ERROR "Cannot find install manifest: @CMAKE_BINARY_DIR@/install_manifest.txt")
|
||||
endif()
|
||||
|
||||
file(READ "@CMAKE_BINARY_DIR@/install_manifest.txt" files)
|
||||
string(REGEX REPLACE "\n" ";" files "${files}")
|
||||
foreach(file ${files})
|
||||
message(STATUS "Uninstalling $ENV{DESTDIR}${file}")
|
||||
if(IS_SYMLINK "$ENV{DESTDIR}${file}" OR EXISTS "$ENV{DESTDIR}${file}")
|
||||
exec_program(
|
||||
"@CMAKE_COMMAND@" ARGS "-E remove \"$ENV{DESTDIR}${file}\""
|
||||
OUTPUT_VARIABLE rm_out
|
||||
RETURN_VALUE rm_retval
|
||||
)
|
||||
if(NOT "${rm_retval}" STREQUAL 0)
|
||||
message(FATAL_ERROR "Problem when removing $ENV{DESTDIR}${file}")
|
||||
endif()
|
||||
else(IS_SYMLINK "$ENV{DESTDIR}${file}" OR EXISTS "$ENV{DESTDIR}${file}")
|
||||
message(STATUS "File $ENV{DESTDIR}${file} does not exist.")
|
||||
endif()
|
||||
endforeach()
|
||||
+72
-6
@@ -7,6 +7,7 @@ CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
|
||||
option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
|
||||
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
|
||||
option(BUILD_THUNKS "Build thunks" FALSE)
|
||||
option(BUILD_CLANG_THUNKS "Build thunks with clang" FALSE)
|
||||
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
|
||||
option(ENABLE_IWYU "Enables include what you use program" FALSE)
|
||||
option(ENABLE_LTO "Enable LTO with compilation" TRUE)
|
||||
@@ -27,10 +28,36 @@ option(ENABLE_LIBCXX "Enables LLVM libc++" FALSE)
|
||||
option(ENABLE_INTERPRETER "Enables FEX's Interpreter" FALSE)
|
||||
option(ENABLE_CCACHE "Enables ccache for compile caching" TRUE)
|
||||
option(ENABLE_TERMUX_BUILD "Forces building for Termux on a non-Termux build machine" FALSE)
|
||||
option(ENABLE_VIXL_SIMULATOR "Forces the FEX JIT to use the VIXL simulator" FALSE)
|
||||
option(ENABLE_FEXCORE_PROFILER "Enables use of the FEXCore timeline profiling capabilities" FALSE)
|
||||
set (FEXCORE_PROFILER_BACKEND "gpuvis" CACHE STRING "Set which backend you want to use for the FEXCore profiler")
|
||||
|
||||
set (X86_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86.cmake" CACHE FILEPATH "Toolchain file for the x86 (cross-)compiler")
|
||||
set (X86_32_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_32.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting i686")
|
||||
set (X86_64_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/toolchain_x86_64.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting x86_64")
|
||||
set (DATA_DIRECTORY "${CMAKE_INSTALL_PREFIX}/share/fex-emu" CACHE PATH "global data directory")
|
||||
|
||||
if (ENABLE_FEXCORE_PROFILER)
|
||||
add_definitions(-DENABLE_FEXCORE_PROFILER=1)
|
||||
string(TOUPPER "${FEXCORE_PROFILER_BACKEND}" FEXCORE_PROFILER_BACKEND)
|
||||
|
||||
if (FEXCORE_PROFILER_BACKEND STREQUAL "GPUVIS")
|
||||
add_definitions(-DFEXCORE_PROFILER_BACKEND=1)
|
||||
else()
|
||||
message(FATAL_ERROR "Unknown FEXCore profiler backend ${FEXCORE_PROFILER_BACKEND}")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# uninstall target
|
||||
if(NOT TARGET uninstall)
|
||||
configure_file(
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/CMakeFiles/cmake_uninstall.cmake.in"
|
||||
"${CMAKE_CURRENT_BINARY_DIR}/CMakeFiles/cmake_uninstall.cmake"
|
||||
IMMEDIATE @ONLY)
|
||||
|
||||
add_custom_target(uninstall
|
||||
COMMAND ${CMAKE_COMMAND} -P ${CMAKE_CURRENT_BINARY_DIR}/CMakeFiles/cmake_uninstall.cmake)
|
||||
endif()
|
||||
|
||||
# These options are meant for package management
|
||||
set (TUNE_CPU "native" CACHE STRING "Override the CPU the build is tuned for")
|
||||
set (TUNE_ARCH "generic" CACHE STRING "Override the Arch the build is tuned for")
|
||||
@@ -84,10 +111,9 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
|
||||
set(_M_X86_64 1)
|
||||
add_definitions(-D_M_X86_64=1)
|
||||
set (CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcx16")
|
||||
set (X86_TOOLCHAIN_FILE "")
|
||||
endif()
|
||||
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "^aarch64|^arm64|^armv8\.*")
|
||||
set(_M_ARM_64 1)
|
||||
add_definitions(-D_M_ARM_64=1)
|
||||
endif()
|
||||
@@ -363,8 +389,6 @@ if (BUILD_TESTS)
|
||||
endif()
|
||||
|
||||
add_subdirectory(FEXHeaderUtils/)
|
||||
include_directories(FEXHeaderUtils/)
|
||||
|
||||
add_subdirectory(External/FEXCore)
|
||||
|
||||
# Binfmt_misc files must be installed prior to Source/ installs
|
||||
@@ -403,8 +427,28 @@ if (BUILD_THUNKS)
|
||||
SOURCE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/ThunkLibs/GuestLibs"
|
||||
BINARY_DIR "Guest"
|
||||
CMAKE_ARGS
|
||||
"-DBITNESS=64"
|
||||
"-DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}"
|
||||
"-DCMAKE_TOOLCHAIN_FILE:FILEPATH=${X86_TOOLCHAIN_FILE}"
|
||||
"-DENABLE_CLANG_THUNKS=${ENABLE_CLANG_THUNKS}"
|
||||
"-DCMAKE_TOOLCHAIN_FILE:FILEPATH=${X86_64_TOOLCHAIN_FILE}"
|
||||
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
|
||||
"-DSTRUCT_VERIFIER=${CMAKE_SOURCE_DIR}/Scripts/StructPackVerifier.py"
|
||||
"-DFEX_PROJECT_SOURCE_DIR=${FEX_PROJECT_SOURCE_DIR}"
|
||||
"-DGENERATOR_EXE=$<TARGET_FILE:thunkgen>"
|
||||
INSTALL_COMMAND ""
|
||||
BUILD_ALWAYS ON
|
||||
DEPENDS thunkgen
|
||||
)
|
||||
|
||||
ExternalProject_Add(guest-libs-32
|
||||
PREFIX guest-libs-32
|
||||
SOURCE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/ThunkLibs/GuestLibs"
|
||||
BINARY_DIR "Guest_32"
|
||||
CMAKE_ARGS
|
||||
"-DBITNESS=32"
|
||||
"-DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}"
|
||||
"-DENABLE_CLANG_THUNKS=${ENABLE_CLANG_THUNKS}"
|
||||
"-DCMAKE_TOOLCHAIN_FILE:FILEPATH=${X86_32_TOOLCHAIN_FILE}"
|
||||
"-DCMAKE_INSTALL_PREFIX=${CMAKE_INSTALL_PREFIX}"
|
||||
"-DSTRUCT_VERIFIER=${CMAKE_SOURCE_DIR}/Scripts/StructPackVerifier.py"
|
||||
"-DFEX_PROJECT_SOURCE_DIR=${FEX_PROJECT_SOURCE_DIR}"
|
||||
@@ -422,6 +466,28 @@ if (BUILD_THUNKS)
|
||||
)"
|
||||
DEPENDS guest-libs
|
||||
)
|
||||
|
||||
install(
|
||||
CODE "MESSAGE(\"-- Installing: guest-libs-32\")"
|
||||
CODE "
|
||||
EXECUTE_PROCESS(COMMAND ${CMAKE_COMMAND} --build . --target install
|
||||
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest_32
|
||||
)"
|
||||
DEPENDS guest-libs-32
|
||||
)
|
||||
|
||||
add_custom_target(uninstall_guest-libs
|
||||
COMMAND ${CMAKE_COMMAND} "--build" "." "--target" "uninstall"
|
||||
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest
|
||||
)
|
||||
|
||||
add_custom_target(uninstall_guest-libs-32
|
||||
COMMAND ${CMAKE_COMMAND} "--build" "." "--target" "uninstall"
|
||||
WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/Guest_32
|
||||
)
|
||||
|
||||
add_dependencies(uninstall uninstall_guest-libs)
|
||||
add_dependencies(uninstall uninstall_guest-libs-32)
|
||||
endif()
|
||||
|
||||
set(FEX_VERSION_MAJOR "0")
|
||||
|
||||
@@ -165,6 +165,14 @@
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libXfixes.so.3.1.0"
|
||||
]
|
||||
},
|
||||
"OpenCL": {
|
||||
"Library" : "libOpenCL-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libOpenCL.so",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libOpenCL.so.1",
|
||||
"@PREFIX_LIB@/x86_64-linux-gnu/libOpenCL.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"":{}
|
||||
}
|
||||
}
|
||||
Vendored
+10
-4
@@ -9,12 +9,19 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64")
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -mcx16")
|
||||
endif()
|
||||
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "^aarch64|^arm64|^armv8\.*")
|
||||
set(_M_ARM_64 1)
|
||||
endif()
|
||||
|
||||
set(ENABLE_JIT_X86_64 ${_M_X86_64} CACHE BOOL "Enable the x86_64 JIT")
|
||||
set(ENABLE_JIT_ARM64 ${_M_ARM_64} CACHE BOOL "Enable the ARM64 JIT")
|
||||
if (ENABLE_VIXL_SIMULATOR)
|
||||
# If the vixl simulator is enabled then we are using the ARM64 JIT
|
||||
option(ENABLE_JIT_X86_64 "Enable the x86_64 JIT" FALSE)
|
||||
option(ENABLE_JIT_ARM64 "Enable the ARM64 JIT" TRUE)
|
||||
else()
|
||||
option(ENABLE_JIT_X86_64 "Enable the x86_64 JIT" ${_M_X86_64})
|
||||
option(ENABLE_JIT_ARM64 "Enable the ARM64 JIT" ${_M_ARM_64})
|
||||
endif()
|
||||
|
||||
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
|
||||
|
||||
set(CMAKE_POSITION_INDEPENDENT_CODE ON)
|
||||
@@ -27,7 +34,6 @@ set(CMAKE_INCLUDE_CURRENT_DIR ON)
|
||||
include(CheckCXXCompilerFlag)
|
||||
include(CheckIncludeFileCXX)
|
||||
|
||||
|
||||
if (EXISTS ${CMAKE_CURRENT_DIR}/External/vixl/)
|
||||
# Useful to have for freestanding libFEXCore
|
||||
add_subdirectory(External/vixl/)
|
||||
|
||||
+10
-4
@@ -143,6 +143,7 @@ set (SRCS
|
||||
Utils/NetStream.cpp
|
||||
Utils/Telemetry.cpp
|
||||
Utils/Threads.cpp
|
||||
Utils/Profiler.cpp
|
||||
)
|
||||
|
||||
if (ENABLE_INTERPRETER)
|
||||
@@ -177,6 +178,11 @@ if (_M_ARM_64)
|
||||
list(APPEND DEFINES -D_M_ARM_64=1)
|
||||
endif()
|
||||
|
||||
if (ENABLE_VIXL_SIMULATOR)
|
||||
# We can run the simulator on both x86-64 or AArch64 hosts
|
||||
list(APPEND DEFINES -DVIXL_SIMULATOR=1 -DVIXL_INCLUDE_SIMULATOR_AARCH64=1)
|
||||
endif()
|
||||
|
||||
if (ENABLE_JIT_X86_64)
|
||||
list(APPEND SRCS
|
||||
Interface/Core/JIT/x86_64/JIT.cpp
|
||||
@@ -213,7 +219,7 @@ if (ENABLE_JIT_ARM64)
|
||||
)
|
||||
endif()
|
||||
|
||||
set (LIBS vixl dl xxhash tiny-json)
|
||||
set (LIBS fmt::fmt vixl dl xxhash tiny-json FEXHeaderUtils)
|
||||
if (ENABLE_JEMALLOC)
|
||||
list (APPEND LIBS FEX_jemalloc)
|
||||
endif()
|
||||
@@ -359,14 +365,14 @@ endfunction()
|
||||
|
||||
# Build FEXCore_Config static library
|
||||
add_library(FEXCore_Base STATIC ${FEXCORE_BASE_SRCS})
|
||||
target_link_libraries(FEXCore_Base fmt::fmt tiny-json)
|
||||
target_link_libraries(FEXCore_Base ${LIBS})
|
||||
AddDefaultOptionsToTarget(FEXCore_Base)
|
||||
|
||||
function(AddObject Name Type)
|
||||
add_library(${Name} ${Type} ${SRCS})
|
||||
add_dependencies(${Name} IR_INC)
|
||||
|
||||
target_link_libraries(${Name} FEXCore_Base ${LIBS})
|
||||
target_link_libraries(${Name} FEXCore_Base)
|
||||
AddDefaultOptionsToTarget(${Name})
|
||||
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
@@ -374,7 +380,7 @@ endfunction()
|
||||
|
||||
function(AddLibrary Name Type)
|
||||
add_library(${Name} ${Type} $<TARGET_OBJECTS:${PROJECT_NAME}_object>)
|
||||
target_link_libraries(${Name} FEXCore_Base ${LIBS})
|
||||
target_link_libraries(${Name} FEXCore_Base)
|
||||
set_target_properties(${Name} PROPERTIES OUTPUT_NAME FEXCore)
|
||||
|
||||
AddDefaultOptionsToTarget(${Name})
|
||||
|
||||
@@ -90,6 +90,13 @@
|
||||
"Folder to find the guest-side thunking libraries."
|
||||
]
|
||||
},
|
||||
"ThunkGuestLibs32": {
|
||||
"Type": "str",
|
||||
"Default": "@CMAKE_INSTALL_PREFIX@/share/fex-emu/GuestThunks_32/",
|
||||
"Desc": [
|
||||
"Folder to find the 32-bit guest-side thunking libraries."
|
||||
]
|
||||
},
|
||||
"ThunkConfig": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
|
||||
+2
-1
@@ -15,6 +15,7 @@
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
#include <stdint.h>
|
||||
|
||||
|
||||
@@ -189,7 +190,7 @@ namespace FEXCore::Context {
|
||||
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), gettid());
|
||||
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == FHU::Syscalls::gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), FHU::Syscalls::gettid());
|
||||
|
||||
FHU::ScopedSignalMaskWithUniqueLock lk(Thread->CTX->CodeInvalidationMutex);
|
||||
|
||||
|
||||
@@ -17,23 +17,35 @@
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define STATE x28
|
||||
|
||||
// We want vixl to not allocate a default buffer. Jit and dispatcher will manually create one.
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::Context *ctx, size_t size)
|
||||
: vixl::aarch64::Assembler(size, vixl::aarch64::PositionDependentCode)
|
||||
: vixl::aarch64::Assembler(size ? (byte*)FEXCore::Allocator::mmap(nullptr, size, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0) : reinterpret_cast<byte*>(~0ULL),
|
||||
size,
|
||||
vixl::aarch64::PositionDependentCode)
|
||||
, EmitterCTX {ctx} {
|
||||
CPU.SetUp();
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
auto Features = vixl::CPUFeatures::All();
|
||||
#else
|
||||
auto Features = vixl::CPUFeatures::InferFromOS();
|
||||
if (ctx->HostFeatures.SupportsAtomics) {
|
||||
// Hypervisor can hide this on the c630?
|
||||
Features.Combine(vixl::CPUFeatures::Feature::kLORegions);
|
||||
}
|
||||
#endif
|
||||
|
||||
SetCPUFeatures(Features);
|
||||
}
|
||||
|
||||
Arm64Emitter::~Arm64Emitter() {
|
||||
auto CodeBuffer = GetBuffer();
|
||||
if (CodeBuffer->GetCapacity()) {
|
||||
FEXCore::Allocator::munmap(CodeBuffer->GetStartAddress<void*>(), CodeBuffer->GetCapacity());
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant, bool NOPPad) {
|
||||
bool Is64Bit = Reg.IsX();
|
||||
int Segments = Is64Bit ? 4 : 2;
|
||||
@@ -214,7 +226,8 @@ void Arm64Emitter::SpillStaticRegs(bool FPRs, uint32_t GPRSpillMask, uint32_t FP
|
||||
const auto Reg = SRAFPR[i];
|
||||
|
||||
if (((1U << Reg.GetCode()) & FPRSpillMask) != 0) {
|
||||
str(Reg.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.avx.data[i][0])));
|
||||
mov(TMP4, offsetof(Core::CpuStateFrame, State.xmm.avx.data[i][0]));
|
||||
st1b(Reg.Z().VnB(), PRED_TMP_32B, SVEMemOperand(STATE, TMP4));
|
||||
}
|
||||
}
|
||||
} else {
|
||||
@@ -257,11 +270,19 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
|
||||
if (FPRs) {
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
// Set up predicate registers.
|
||||
// We don't bother spilling these in SpillStaticRegs,
|
||||
// since all that matters is we restore them on a fill.
|
||||
// It's not a concern if they get trounced by something else.
|
||||
ptrue(PRED_TMP_16B.VnB(), SVE_VL16);
|
||||
ptrue(PRED_TMP_32B.VnB(), SVE_VL32);
|
||||
|
||||
for (size_t i = 0; i < SRAFPR.size(); i++) {
|
||||
const auto Reg = SRAFPR[i];
|
||||
|
||||
if (((1U << Reg.GetCode()) & FPRFillMask) != 0) {
|
||||
ldr(Reg.Q(), MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.xmm.avx.data[i][0])));
|
||||
mov(TMP4, offsetof(Core::CpuStateFrame, State.xmm.avx.data[i][0]));
|
||||
ld1b(Reg.Z().VnB(), PRED_TMP_32B.Zeroing(), SVEMemOperand(STATE, TMP4));
|
||||
}
|
||||
}
|
||||
} else {
|
||||
@@ -286,20 +307,31 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
}
|
||||
|
||||
void Arm64Emitter::PushDynamicRegsAndLR() {
|
||||
uint64_t SPOffset = AlignUp((RA64.size() + 1) * 8 + RAFPR.size() * 16, 16);
|
||||
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
|
||||
const auto GPRSize = (RA64.size() + 1) * Core::CPUState::GPR_REG_SIZE;
|
||||
const auto FPRRegSize = CanUseSVE ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto FPRSize = RAFPR.size() * FPRRegSize;
|
||||
const uint64_t SPOffset = AlignUp(GPRSize + FPRSize, 16);
|
||||
|
||||
sub(sp, sp, SPOffset);
|
||||
int i = 0;
|
||||
|
||||
for (auto RA : RAFPR)
|
||||
{
|
||||
str(RA.Q(), MemOperand(sp, i * 8));
|
||||
i+=2;
|
||||
if (CanUseSVE) {
|
||||
for (const auto& RA : RAFPR) {
|
||||
mov(TMP4, i * 8);
|
||||
st1b(RA.Z().VnB(), PRED_TMP_32B, SVEMemOperand(sp, TMP4));
|
||||
i += 4;
|
||||
}
|
||||
} else {
|
||||
for (const auto& RA : RAFPR) {
|
||||
str(RA.Q(), MemOperand(sp, i * 8));
|
||||
i += 2;
|
||||
}
|
||||
}
|
||||
|
||||
#if 0 // All GPRs should be caller saved
|
||||
for (auto RA : RA64)
|
||||
{
|
||||
for (const auto& RA : RA64) {
|
||||
str(RA, MemOperand(sp, i * 8));
|
||||
i++;
|
||||
}
|
||||
@@ -309,18 +341,29 @@ void Arm64Emitter::PushDynamicRegsAndLR() {
|
||||
}
|
||||
|
||||
void Arm64Emitter::PopDynamicRegsAndLR() {
|
||||
uint64_t SPOffset = AlignUp((RA64.size() + 1) * 8 + RAFPR.size() * 16, 16);
|
||||
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
|
||||
const auto GPRSize = (RA64.size() + 1) * Core::CPUState::GPR_REG_SIZE;
|
||||
const auto FPRRegSize = CanUseSVE ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto FPRSize = RAFPR.size() * FPRRegSize;
|
||||
const uint64_t SPOffset = AlignUp(GPRSize + FPRSize, 16);
|
||||
int i = 0;
|
||||
|
||||
for (auto RA : RAFPR)
|
||||
{
|
||||
ldr(RA.Q(), MemOperand(sp, i * 8));
|
||||
i+=2;
|
||||
if (CanUseSVE) {
|
||||
for (const auto& RA : RAFPR) {
|
||||
mov(TMP4, i * 8);
|
||||
ld1b(RA.Z().VnB(), PRED_TMP_32B.Zeroing(), SVEMemOperand(sp, TMP4));
|
||||
i += 4;
|
||||
}
|
||||
} else {
|
||||
for (const auto& RA : RAFPR) {
|
||||
ldr(RA.Q(), MemOperand(sp, i * 8));
|
||||
i += 2;
|
||||
}
|
||||
}
|
||||
|
||||
#if 0 // All GPRs should be caller saved
|
||||
for (auto RA : RA64)
|
||||
{
|
||||
for (const auto& RA : RA64) {
|
||||
ldr(RA, MemOperand(sp, i * 8));
|
||||
i++;
|
||||
}
|
||||
|
||||
@@ -8,6 +8,10 @@
|
||||
#include <aarch64/cpu-aarch64.h>
|
||||
#include <aarch64/operands-aarch64.h>
|
||||
#include <platform-vixl.h>
|
||||
#ifdef VIXL_SIMULATOR
|
||||
#include <aarch64/simulator-aarch64.h>
|
||||
#include <aarch64/simulator-constants-aarch64.h>
|
||||
#endif
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
|
||||
@@ -58,15 +62,41 @@ const std::array<aarch64::VRegister, 12> RAFPR = {
|
||||
v8, v9, v10, v11, v12, v13, v14, v15
|
||||
};
|
||||
|
||||
// Contains the address to the currently available CPU state
|
||||
#define STATE x28
|
||||
|
||||
// GPR temporaries. Only x3 can be used across spill boundaries
|
||||
// so if these ever need to change, be very careful about that.
|
||||
#define TMP1 x0
|
||||
#define TMP2 x1
|
||||
#define TMP3 x2
|
||||
#define TMP4 x3
|
||||
|
||||
// Vector temporaries
|
||||
#define VTMP1 v1
|
||||
#define VTMP2 v2
|
||||
#define VTMP3 v3
|
||||
|
||||
// Predicate register temporaries (used when AVX support is enabled)
|
||||
// PRED_TMP_16B indicates a predicate register that indicates the first 16 bytes set to 1.
|
||||
// PRED_TMP_32B indicates a predicate register that indicates the first 32 bytes set to 1.
|
||||
#define PRED_TMP_16B p6
|
||||
#define PRED_TMP_32B p7
|
||||
|
||||
// This class contains common emitter utility functions that can
|
||||
// be used by both Arm64 JIT and ARM64 Dispatcher
|
||||
class Arm64Emitter : public vixl::aarch64::Assembler {
|
||||
protected:
|
||||
Arm64Emitter(FEXCore::Context::Context *ctx, size_t size);
|
||||
~Arm64Emitter();
|
||||
|
||||
FEXCore::Context::Context *EmitterCTX;
|
||||
vixl::aarch64::CPU CPU;
|
||||
void LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant, bool NOPPad = false);
|
||||
|
||||
// NOTE: These functions WILL clobber the register TMP4 if AVX support is enabled
|
||||
// and FPRs are being spilled or filled. If only GPRs are spilled/filled, then
|
||||
// TMP4 is left alone.
|
||||
void SpillStaticRegs(bool FPRs = true, uint32_t GPRSpillMask = ~0U, uint32_t FPRSpillMask = ~0U);
|
||||
void FillStaticRegs(bool FPRs = true, uint32_t GPRFillMask = ~0U, uint32_t FPRFillMask = ~0U);
|
||||
|
||||
@@ -83,6 +113,71 @@ protected:
|
||||
void PopCalleeSavedRegisters();
|
||||
|
||||
void Align16B();
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// Generates a vixl simulator runtime call.
|
||||
//
|
||||
// This matches behaviour of vixl's macro assembler, but we need to reimplement it since we aren't using the macro assembler.
|
||||
// This isn't too complex with how vixl emits this.
|
||||
//
|
||||
// Emit:
|
||||
// 1) hlt(kRuntimeCallOpcode)
|
||||
// 2) Simulator wrapper handler
|
||||
// 3) Function to call
|
||||
// 4) Style of the function call (Call versus tail-call)
|
||||
template<typename R, typename... P>
|
||||
void GenerateRuntimeCall(R (*Function)(P...)) {
|
||||
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(
|
||||
&(Simulator::RuntimeCallStructHelper<R, P...>::Wrapper));
|
||||
|
||||
uintptr_t FunctionAddress = reinterpret_cast<uintptr_t>(Function);
|
||||
|
||||
hlt(kRuntimeCallOpcode);
|
||||
|
||||
// Simulator wrapper address pointer.
|
||||
dc(SimulatorWrapperAddress);
|
||||
|
||||
// Runtime function address to call
|
||||
dc(FunctionAddress);
|
||||
|
||||
// Call type
|
||||
dc32(kCallRuntime);
|
||||
}
|
||||
|
||||
template<typename R, typename... P>
|
||||
void GenerateIndirectRuntimeCall(vixl::aarch64::Register Reg) {
|
||||
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(
|
||||
&(Simulator::RuntimeCallStructHelper<R, P...>::Wrapper));
|
||||
|
||||
hlt(kIndirectRuntimeCallOpcode);
|
||||
|
||||
// Simulator wrapper address pointer.
|
||||
dc(SimulatorWrapperAddress);
|
||||
|
||||
// Register that contains the function to call
|
||||
dc(Reg.GetCode());
|
||||
|
||||
// Call type
|
||||
dc32(kCallRuntime);
|
||||
}
|
||||
|
||||
template<>
|
||||
void GenerateIndirectRuntimeCall<float, __uint128_t>(vixl::aarch64::Register Reg) {
|
||||
uintptr_t SimulatorWrapperAddress = reinterpret_cast<uintptr_t>(
|
||||
&(Simulator::RuntimeCallStructHelper<float, __uint128_t>::Wrapper));
|
||||
|
||||
hlt(kIndirectRuntimeCallOpcode);
|
||||
|
||||
// Simulator wrapper address pointer.
|
||||
dc(SimulatorWrapperAddress);
|
||||
|
||||
// Register that contains the function to call
|
||||
dc(Reg.GetCode());
|
||||
|
||||
// Call type
|
||||
dc32(kCallRuntime);
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
|
||||
};
|
||||
|
||||
+19
-13
@@ -44,6 +44,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
#include <FEXHeaderUtils/TodoDefines.h>
|
||||
|
||||
@@ -221,7 +222,7 @@ namespace FEXCore::Context {
|
||||
#if (_M_X86_64 && JIT_X86_64)
|
||||
FEXCore::CPU::InitializeX86JITSignalHandlers(this);
|
||||
BackendFeatures = FEXCore::CPU::GetX86JITBackendFeatures();
|
||||
#elif (_M_ARM_64 && JIT_ARM64)
|
||||
#elif (_M_ARM_64 && JIT_ARM64) || defined(VIXL_SIMULATOR)
|
||||
FEXCore::CPU::InitializeArm64JITSignalHandlers(this);
|
||||
BackendFeatures = FEXCore::CPU::GetArm64JITBackendFeatures();
|
||||
#else
|
||||
@@ -238,16 +239,16 @@ namespace FEXCore::Context {
|
||||
|
||||
DispatcherConfig.StaticRegisterAllocation = Config.StaticRegisterAllocation && BackendFeatures.SupportsStaticRegisterAllocation;
|
||||
|
||||
#if (_M_X86_64)
|
||||
Dispatcher = FEXCore::CPU::Dispatcher::CreateX86(this, DispatcherConfig);
|
||||
#elif (_M_ARM_64)
|
||||
#if JIT_ARM64
|
||||
Dispatcher = FEXCore::CPU::Dispatcher::CreateArm64(this, DispatcherConfig);
|
||||
#elif JIT_X86_64
|
||||
Dispatcher = FEXCore::CPU::Dispatcher::CreateX86(this, DispatcherConfig);
|
||||
#else
|
||||
ERROR_AND_DIE_FMT("FEXCore has been compiled with an unknown target");
|
||||
#endif
|
||||
|
||||
// Initialize common signal handlers
|
||||
|
||||
|
||||
auto PauseHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
return Thread->CTX->Dispatcher->HandleSignalPause(Thread, Signal, info, ucontext);
|
||||
};
|
||||
@@ -573,7 +574,7 @@ namespace FEXCore::Context {
|
||||
|
||||
#if (_M_X86_64 && JIT_X86_64)
|
||||
Thread->CPUBackend = FEXCore::CPU::CreateX86JITCore(this, Thread);
|
||||
#elif (_M_ARM_64 && JIT_ARM64)
|
||||
#elif (_M_ARM_64 && JIT_ARM64) || defined(VIXL_SIMULATOR)
|
||||
Thread->CPUBackend = FEXCore::CPU::CreateArm64JITCore(this, Thread);
|
||||
#else
|
||||
ERROR_AND_DIE_FMT("FEXCore has been compiled without a viable JIT core");
|
||||
@@ -674,6 +675,8 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
void Context::ClearCodeCache(FEXCore::Core::InternalThreadState *Thread) {
|
||||
FEXCORE_PROFILE_INSTANT("ClearCodeCache");
|
||||
|
||||
{
|
||||
// Ensure the Code Object Serialization service has fully serialized this thread's data before clearing the cache
|
||||
// Use the thread's object cache ref counter for this
|
||||
@@ -740,7 +743,9 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
Context::GenerateIRResult Context::GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo) {
|
||||
Context::GenerateIRResult Context::GenerateIR(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, bool ExtendedDebugInfo) {
|
||||
FEXCORE_PROFILE_SCOPED("GenerateIR");
|
||||
|
||||
Thread->OpDispatcher->ReownOrClaimBuffer();
|
||||
Thread->OpDispatcher->ResetWorkingList();
|
||||
|
||||
@@ -749,7 +754,7 @@ namespace FEXCore::Context {
|
||||
|
||||
|
||||
std::shared_lock lk(CustomIRMutex);
|
||||
|
||||
|
||||
auto Handler = CustomIRHandlers.find(GuestRIP);
|
||||
if (Handler != CustomIRHandlers.end()) {
|
||||
TotalInstructions = 1;
|
||||
@@ -872,7 +877,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Thread->OpDispatcher->Finalize();
|
||||
|
||||
Thread->FrontendDecoder->DelayedDisownBuffer();
|
||||
@@ -1011,6 +1016,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
uintptr_t Context::CompileBlock(FEXCore::Core::CpuStateFrame *Frame, uint64_t GuestRIP) {
|
||||
FEXCORE_PROFILE_SCOPED("CompileBlock");
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
|
||||
@@ -1182,7 +1188,7 @@ namespace FEXCore::Context {
|
||||
|
||||
static void InvalidateGuestCodeRangeInternal(FEXCore::Context::Context *CTX, uint64_t Start, uint64_t Length) {
|
||||
std::lock_guard lk(CTX->ThreadCreationMutex);
|
||||
|
||||
|
||||
for (auto &Thread : CTX->Threads) {
|
||||
InvalidateGuestThreadCodeRange(Thread, Start, Length);
|
||||
}
|
||||
@@ -1190,7 +1196,7 @@ namespace FEXCore::Context {
|
||||
|
||||
void InvalidateGuestCodeRange(FEXCore::Context::Context *CTX, uint64_t Start, uint64_t Length) {
|
||||
FHU::ScopedSignalMaskWithUniqueLock CodeInvalidationLock(CTX->CodeInvalidationMutex);
|
||||
|
||||
|
||||
InvalidateGuestCodeRangeInternal(CTX, Start, Length);
|
||||
}
|
||||
|
||||
@@ -1233,9 +1239,9 @@ namespace FEXCore::Context {
|
||||
Thread->LookupCache->AddBlockLink(GuestDestination, HostLink, delinker);
|
||||
}
|
||||
|
||||
void Context::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
void Context::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) {
|
||||
LogMan::Throw::AFmt(Thread->CTX->CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to be unique_locked here");
|
||||
|
||||
|
||||
std::lock_guard<std::recursive_mutex> lk(Thread->LookupCache->WriteLock);
|
||||
|
||||
Thread->DebugStore.erase(GuestRIP);
|
||||
|
||||
@@ -38,11 +38,20 @@ namespace FEXCore::CPU {
|
||||
using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
|
||||
static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
|
||||
#define STATE x28
|
||||
constexpr size_t MAX_DISPATCHER_CODE_SIZE = 8192;
|
||||
|
||||
Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config)
|
||||
: FEXCore::CPU::Dispatcher(ctx, config), Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE) {
|
||||
: FEXCore::CPU::Dispatcher(ctx, config), Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE)
|
||||
#ifdef VIXL_SIMULATOR
|
||||
, Simulator {&Decoder}
|
||||
#endif
|
||||
{
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// Hardcode a 256-bit vector width if we are running in the simulator.
|
||||
Simulator.SetVectorLengthInBits(256);
|
||||
#endif
|
||||
|
||||
SetAllowAssembler(true);
|
||||
|
||||
DispatchPtr = GetCursorAddress<AsmDispatch>();
|
||||
@@ -178,7 +187,12 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
ret();
|
||||
}
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// VIXL simulator can't run syscalls.
|
||||
constexpr bool SignalSafeCompile = false;
|
||||
#else
|
||||
constexpr bool SignalSafeCompile = true;
|
||||
#endif
|
||||
{
|
||||
ExitFunctionLinkerAddress = GetCursorAddress<uint64_t>();
|
||||
if (config.StaticRegisterAllocation)
|
||||
@@ -206,8 +220,12 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
mov(x0, STATE);
|
||||
mov(x1, lr);
|
||||
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
|
||||
blr(x3);
|
||||
ldr(x2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uintptr_t, void *, void *>(x2);
|
||||
#else
|
||||
blr(x2);
|
||||
#endif
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
@@ -266,8 +284,11 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
ldr(x3, &l_CompileBlock);
|
||||
|
||||
// X2 contains our guest RIP
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<void, void *, uint64_t, void *>(x3);
|
||||
#else
|
||||
blr(x3); // { CTX, Frame, RIP}
|
||||
|
||||
#endif
|
||||
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
@@ -349,7 +370,11 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
ldr(x0, &l_CTX);
|
||||
mov(x1, STATE);
|
||||
ldr(x2, &l_Sleep);
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<void, void *, void *>(x2);
|
||||
#else
|
||||
blr(x2);
|
||||
#endif
|
||||
|
||||
PauseReturnInstruction = GetCursorAddress<uint64_t>();
|
||||
// Fault to start running again
|
||||
@@ -412,11 +437,14 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
LUDIVHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
SpillStaticRegs();
|
||||
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUDIV));
|
||||
|
||||
SpillStaticRegs();
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(x3);
|
||||
#else
|
||||
blr(x3);
|
||||
#endif
|
||||
FillStaticRegs();
|
||||
|
||||
// Result is now in x0
|
||||
@@ -431,11 +459,14 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
LDIVHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
SpillStaticRegs();
|
||||
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LDIV));
|
||||
|
||||
SpillStaticRegs();
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(x3);
|
||||
#else
|
||||
blr(x3);
|
||||
#endif
|
||||
FillStaticRegs();
|
||||
|
||||
// Result is now in x0
|
||||
@@ -450,11 +481,14 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
LUREMHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
SpillStaticRegs();
|
||||
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUREM));
|
||||
|
||||
SpillStaticRegs();
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(x3);
|
||||
#else
|
||||
blr(x3);
|
||||
#endif
|
||||
FillStaticRegs();
|
||||
|
||||
// Result is now in x0
|
||||
@@ -469,11 +503,14 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
LREMHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
SpillStaticRegs();
|
||||
|
||||
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LREM));
|
||||
|
||||
SpillStaticRegs();
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t>(x3);
|
||||
#else
|
||||
blr(x3);
|
||||
#endif
|
||||
FillStaticRegs();
|
||||
|
||||
// Result is now in x0
|
||||
@@ -504,13 +541,27 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const Dispatche
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
void Arm64Dispatcher::ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) {
|
||||
Simulator.WriteXRegister(0, reinterpret_cast<int64_t>(Frame));
|
||||
Simulator.RunFrom(reinterpret_cast<Instruction const*>(DispatchPtr));
|
||||
}
|
||||
|
||||
void Arm64Dispatcher::ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) {
|
||||
Simulator.WriteXRegister(0, reinterpret_cast<int64_t>(Frame));
|
||||
Simulator.WriteXRegister(1, RIP);
|
||||
Simulator.RunFrom(reinterpret_cast<Instruction const*>(CallbackPtr));
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
// Used by GenerateGDBPauseCheck, GenerateInterpreterTrampoline, destination buffer is set before use
|
||||
static thread_local vixl::aarch64::Assembler emit((uint8_t*)&emit, 1);
|
||||
|
||||
size_t Arm64Dispatcher::GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) {
|
||||
|
||||
*emit.GetBuffer() = vixl::CodeBuffer(CodeBuffer, MaxGDBPauseCheckSize);
|
||||
|
||||
|
||||
vixl::CodeBufferCheckScope scope(&emit, MaxGDBPauseCheckSize, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
|
||||
aarch64::Label RunBlock;
|
||||
@@ -546,7 +597,7 @@ size_t Arm64Dispatcher::GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t Gues
|
||||
|
||||
size_t Arm64Dispatcher::GenerateInterpreterTrampoline(uint8_t *CodeBuffer) {
|
||||
LOGMAN_THROW_AA_FMT(!config.StaticRegisterAllocation, "GenerateInterpreterTrampoline dispatcher does not support SRA");
|
||||
|
||||
|
||||
*emit.GetBuffer() = vixl::CodeBuffer(CodeBuffer, MaxInterpreterTrampolineSize);
|
||||
|
||||
vixl::CodeBufferCheckScope scope(&emit, MaxInterpreterTrampolineSize, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
|
||||
@@ -594,7 +645,7 @@ void Arm64Dispatcher::SpillSRA(FEXCore::Core::InternalThreadState *Thread, void
|
||||
}
|
||||
|
||||
void Arm64Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) {
|
||||
// Setup dispatcher specific pointers that need to be accessed from JIT code
|
||||
// Setup dispatcher specific pointers that need to be accessed from JIT code
|
||||
{
|
||||
auto &Common = Thread->CurrentFrame->Pointers.Common;
|
||||
|
||||
|
||||
@@ -3,6 +3,10 @@
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
#include <aarch64/simulator-aarch64.h>
|
||||
#endif
|
||||
|
||||
namespace FEXCore::Context {
|
||||
struct Context;
|
||||
}
|
||||
@@ -20,6 +24,11 @@ class Arm64Dispatcher final : public Dispatcher, public Arm64Emitter {
|
||||
size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) override;
|
||||
size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) override;
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) override;
|
||||
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) override;
|
||||
#endif
|
||||
|
||||
protected:
|
||||
void SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) override;
|
||||
|
||||
@@ -29,6 +38,11 @@ class Arm64Dispatcher final : public Dispatcher, public Arm64Emitter {
|
||||
uint64_t LDIVHandlerAddress{};
|
||||
uint64_t LUREMHandlerAddress{};
|
||||
uint64_t LREMHandlerAddress{};
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
vixl::aarch64::Decoder Decoder;
|
||||
vixl::aarch64::Simulator Simulator;
|
||||
#endif
|
||||
};
|
||||
|
||||
}
|
||||
@@ -212,12 +212,20 @@ void Dispatcher::RestoreThreadState(FEXCore::Core::InternalThreadState *Thread,
|
||||
Frame->State.flags[9] = 1;
|
||||
|
||||
Frame->State.rip = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP];
|
||||
Frame->State.cs = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS];
|
||||
Frame->State.ds = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS];
|
||||
Frame->State.es = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES];
|
||||
Frame->State.fs = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS];
|
||||
Frame->State.gs = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS];
|
||||
Frame->State.ss = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS];
|
||||
Frame->State.cs_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS];
|
||||
Frame->State.ds_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS];
|
||||
Frame->State.es_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES];
|
||||
Frame->State.fs_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS];
|
||||
Frame->State.gs_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS];
|
||||
Frame->State.ss_idx = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS];
|
||||
|
||||
Frame->State.cs_cached = Frame->State.gdt[Frame->State.cs_idx >> 3].base;
|
||||
Frame->State.ds_cached = Frame->State.gdt[Frame->State.ds_idx >> 3].base;
|
||||
Frame->State.es_cached = Frame->State.gdt[Frame->State.es_idx >> 3].base;
|
||||
Frame->State.fs_cached = Frame->State.gdt[Frame->State.fs_idx >> 3].base;
|
||||
Frame->State.gs_cached = Frame->State.gdt[Frame->State.gs_idx >> 3].base;
|
||||
Frame->State.ss_cached = Frame->State.gdt[Frame->State.ss_idx >> 3].base;
|
||||
|
||||
#define COPY_REG(x) \
|
||||
Frame->State.gregs[X86State::REG_##x] = guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_##x];
|
||||
COPY_REG(RDI);
|
||||
@@ -565,10 +573,13 @@ bool Dispatcher::HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, i
|
||||
auto *xstate = reinterpret_cast<x86::xstate*>(FPStateLocation);
|
||||
SetXStateInfo(xstate, IsAVXEnabled);
|
||||
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS] = Frame->State.gs;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS] = Frame->State.fs;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES] = Frame->State.es;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS] = Frame->State.ds;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS] = Frame->State.cs_idx;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS] = Frame->State.ds_idx;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES] = Frame->State.es_idx;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_FS] = Frame->State.fs_idx;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_GS] = Frame->State.gs_idx;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS] = Frame->State.ss_idx;
|
||||
|
||||
if (ContextBackup->FaultToTopAndGeneratedException) {
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = Frame->SynchronousFaultData.TrapNo;
|
||||
guest_siginfo->si_code = Frame->SynchronousFaultData.si_code;
|
||||
@@ -581,10 +592,8 @@ bool Dispatcher::HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, i
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = ConvertSignalToError(Signal, HostSigInfo);
|
||||
}
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EIP] = Frame->State.rip;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_CS] = Frame->State.cs;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_EFL] = 0;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_UESP] = 0;
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_SS] = Frame->State.ss;
|
||||
|
||||
#define COPY_REG(x) \
|
||||
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_##x] = Frame->State.gregs[X86State::REG_##x];
|
||||
|
||||
@@ -32,7 +32,7 @@ struct DispatcherConfig {
|
||||
class Dispatcher {
|
||||
public:
|
||||
virtual ~Dispatcher() = default;
|
||||
|
||||
|
||||
/**
|
||||
* @name Dispatch Helper functions
|
||||
* @{ */
|
||||
@@ -75,12 +75,12 @@ public:
|
||||
|
||||
static std::unique_ptr<Dispatcher> CreateX86(FEXCore::Context::Context *CTX, const DispatcherConfig &Config);
|
||||
static std::unique_ptr<Dispatcher> CreateArm64(FEXCore::Context::Context *CTX, const DispatcherConfig &Config);
|
||||
|
||||
void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) {
|
||||
|
||||
virtual void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) {
|
||||
DispatchPtr(Frame);
|
||||
}
|
||||
|
||||
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) {
|
||||
virtual void ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) {
|
||||
CallbackPtr(Frame, RIP);
|
||||
}
|
||||
|
||||
|
||||
+2
-1
@@ -18,6 +18,7 @@ $end_info$
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
#include <FEXHeaderUtils/TypeDefines.h>
|
||||
#include <set>
|
||||
@@ -1132,6 +1133,7 @@ const uint8_t *Decoder::AdjustAddrForSpecialRegion(uint8_t const* _InstStream, u
|
||||
}
|
||||
|
||||
void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC, std::function<void(uint64_t BlockEntry, uint64_t Start, uint64_t Length)> AddContainedCodePage) {
|
||||
FEXCORE_PROFILE_SCOPED("DecodeInstructions");
|
||||
Blocks.clear();
|
||||
BlocksToDecode.clear();
|
||||
HasBlocks.clear();
|
||||
@@ -1166,7 +1168,6 @@ void Decoder::DecodeInstructionsAtEntry(uint8_t const* _InstStream, uint64_t PC,
|
||||
std::set<uint64_t> CodePages = { CurrentCodePage };
|
||||
|
||||
AddContainedCodePage(PC, CurrentCodePage, FHU::FEX_PAGE_SIZE);
|
||||
|
||||
|
||||
while (!BlocksToDecode.empty()) {
|
||||
auto BlockDecodeIt = BlocksToDecode.begin();
|
||||
|
||||
+42
-25
@@ -1,7 +1,7 @@
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
#if defined(_M_ARM_64) || defined(VIXL_SIMULATOR)
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
#include "aarch64/disasm-aarch64.h"
|
||||
@@ -50,8 +50,12 @@ static uint32_t GetDCZID() {
|
||||
|
||||
|
||||
HostFeatures::HostFeatures() {
|
||||
#ifdef _M_ARM_64
|
||||
#if defined(_M_ARM_64) || defined(VIXL_SIMULATOR)
|
||||
#ifdef VIXL_SIMULATOR
|
||||
auto Features = vixl::CPUFeatures::All();
|
||||
#else
|
||||
auto Features = vixl::CPUFeatures::InferFromOS();
|
||||
#endif
|
||||
SupportsAES = Features.Has(vixl::CPUFeatures::Feature::kAES);
|
||||
SupportsCRC = Features.Has(vixl::CPUFeatures::Feature::kCRC32);
|
||||
SupportsAtomics = Features.Has(vixl::CPUFeatures::Feature::kAtomics);
|
||||
@@ -64,12 +68,22 @@ HostFeatures::HostFeatures() {
|
||||
|
||||
Supports3DNow = true;
|
||||
SupportsSSE4A = true;
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// Hardcode enable SVE with 256-bit wide registers.
|
||||
SupportsAVX = true;
|
||||
#else
|
||||
SupportsAVX = Features.Has(vixl::CPUFeatures::Feature::kSVE2) &&
|
||||
vixl::aarch64::CPU::ReadSVEVectorLengthInBits() >= 256;
|
||||
#endif
|
||||
SupportsSHA = true;
|
||||
SupportsBMI1 = true;
|
||||
SupportsBMI2 = true;
|
||||
|
||||
if (!SupportsAtomics) {
|
||||
WARN_ONCE_FMT("Host CPU doesn't support atomics. Expect bad performance");
|
||||
}
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
// We need to get the CPU's cache line size
|
||||
// We expect sane targets that have correct cacheline sizes across clusters
|
||||
uint64_t CTR;
|
||||
@@ -79,11 +93,28 @@ HostFeatures::HostFeatures() {
|
||||
DCacheLineSize = 4 << ((CTR >> 16) & 0xF);
|
||||
ICacheLineSize = 4 << (CTR & 0xF);
|
||||
|
||||
if (!SupportsAtomics) {
|
||||
WARN_ONCE_FMT("Host CPU doesn't support atomics. Expect bad performance");
|
||||
}
|
||||
// Test if this CPU supports float exception trapping by attempting to enable
|
||||
// On unsupported these bits are architecturally defined as RAZ/WI
|
||||
constexpr uint32_t ExceptionEnableTraps =
|
||||
(1U << 8) | // Invalid Operation float exception trap enable
|
||||
(1U << 9) | // Divide by zero float exception trap enable
|
||||
(1U << 10) | // Overflow float exception trap enable
|
||||
(1U << 11) | // Underflow float exception trap enable
|
||||
(1U << 12) | // Inexact float exception trap enable
|
||||
(1U << 15); // Input Denormal float exception trap enable
|
||||
|
||||
uint32_t OriginalFPCR = GetFPCR();
|
||||
uint32_t FPCR = OriginalFPCR | ExceptionEnableTraps;
|
||||
SetFPCR(FPCR);
|
||||
FPCR = GetFPCR();
|
||||
SupportsFloatExceptions = (FPCR & ExceptionEnableTraps) == ExceptionEnableTraps;
|
||||
|
||||
// Set FPCR back to original just in case anything changed
|
||||
SetFPCR(OriginalFPCR);
|
||||
#endif
|
||||
#ifdef _M_X86_64
|
||||
|
||||
#endif
|
||||
#if defined(_M_X86_64) && !defined(VIXL_SIMULATOR)
|
||||
Xbyak::util::Cpu Features{};
|
||||
SupportsAES = Features.has(Xbyak::util::Cpu::tAESNI);
|
||||
SupportsCRC = Features.has(Xbyak::util::Cpu::tSSE42);
|
||||
@@ -109,27 +140,12 @@ HostFeatures::HostFeatures() {
|
||||
|
||||
SupportsFlushInputsToZero = true;
|
||||
SupportsFloatExceptions = true;
|
||||
#else
|
||||
// Test if this CPU supports float exception trapping by attempting to enable
|
||||
// On unsupported these bits are architecturally defined as RAZ/WI
|
||||
constexpr uint32_t ExceptionEnableTraps =
|
||||
(1U << 8) | // Invalid Operation float exception trap enable
|
||||
(1U << 9) | // Divide by zero float exception trap enable
|
||||
(1U << 10) | // Overflow float exception trap enable
|
||||
(1U << 11) | // Underflow float exception trap enable
|
||||
(1U << 12) | // Inexact float exception trap enable
|
||||
(1U << 15); // Input Denormal float exception trap enable
|
||||
|
||||
uint32_t OriginalFPCR = GetFPCR();
|
||||
uint32_t FPCR = OriginalFPCR | ExceptionEnableTraps;
|
||||
SetFPCR(FPCR);
|
||||
FPCR = GetFPCR();
|
||||
SupportsFloatExceptions = (FPCR & ExceptionEnableTraps) == ExceptionEnableTraps;
|
||||
|
||||
// Set FPCR back to original just in case anything changed
|
||||
SetFPCR(OriginalFPCR);
|
||||
#endif
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// simulator doesn't support dc(ZVA)
|
||||
SupportsCLZERO = false;
|
||||
#else
|
||||
// Check if we can support cacheline clears
|
||||
uint32_t DCZID = GetDCZID();
|
||||
if ((DCZID & DCZID_DZP_MASK) == 0) {
|
||||
@@ -139,5 +155,6 @@ HostFeatures::HostFeatures() {
|
||||
// This means we can use the instruction
|
||||
SupportsCLZERO = DCZID_Bytes == CPUIDEmu::CACHELINE_SIZE;
|
||||
}
|
||||
#endif
|
||||
}
|
||||
}
|
||||
+35
-17
@@ -894,33 +894,51 @@ DEF_OP(Select) {
|
||||
}
|
||||
|
||||
DEF_OP(VExtractToGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
const auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
constexpr auto AVXRegSize = Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
constexpr auto SSERegSize = Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
constexpr auto SSEBitSize = SSERegSize * 8;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto ElementSizeBits = ElementSize * 8;
|
||||
const auto Shift = ElementSizeBits * Op->Index;
|
||||
|
||||
const uint32_t SourceSize = GetOpSize(Data->CurrentIR, Op->Vector);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(IROp->Size <= 16, "OpSize is too large for VExtractToGPR: {}", IROp->Size);
|
||||
LOGMAN_THROW_AA_FMT(OpSize <= AVXRegSize,
|
||||
"OpSize is too large for VExtractToGPR: {}", OpSize);
|
||||
|
||||
if (SourceSize == 16) {
|
||||
__uint128_t SourceMask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
|
||||
uint64_t Shift = Op->Header.ElementSize * Op->Index * 8;
|
||||
if (Op->Header.ElementSize == 8)
|
||||
if (SourceSize >= SSERegSize) {
|
||||
__uint128_t SourceMask = (1ULL << ElementSizeBits) - 1;
|
||||
if (ElementSize == 8) {
|
||||
SourceMask = ~0ULL;
|
||||
}
|
||||
|
||||
__uint128_t Src = *GetSrc<__uint128_t*>(Data->SSAData, Op->Vector);
|
||||
Src >>= Shift;
|
||||
Src &= SourceMask;
|
||||
memcpy(GDP, &Src, Op->Header.ElementSize);
|
||||
const auto Src = *GetSrc<InterpVector256*>(Data->SSAData, Op->Vector);
|
||||
|
||||
const auto GetResult = [&] {
|
||||
if (Shift >= SSEBitSize) {
|
||||
const auto NormalizedShift = Shift - SSEBitSize;
|
||||
return (Src.Upper >> NormalizedShift) & SourceMask;
|
||||
} else {
|
||||
return (Src.Lower >> Shift) & SourceMask;
|
||||
}
|
||||
};
|
||||
|
||||
const auto Result = GetResult();
|
||||
memcpy(GDP, &Result, ElementSize);
|
||||
}
|
||||
else {
|
||||
uint64_t SourceMask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
|
||||
uint64_t Shift = Op->Header.ElementSize * Op->Index * 8;
|
||||
if (Op->Header.ElementSize == 8)
|
||||
uint64_t SourceMask = (1ULL << ElementSizeBits) - 1;
|
||||
if (ElementSize == 8) {
|
||||
SourceMask = ~0ULL;
|
||||
}
|
||||
|
||||
uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Vector);
|
||||
Src >>= Shift;
|
||||
Src &= SourceMask;
|
||||
GD = Src;
|
||||
const uint64_t Src = *GetSrc<uint64_t*>(Data->SSAData, Op->Vector);
|
||||
const uint64_t Result = (Src >> Shift) & SourceMask;
|
||||
GD = Result;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -13,22 +13,46 @@ $end_info$
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(VInsGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->DestVector);
|
||||
auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Src);
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto ElementSizeBits = ElementSize * 8;
|
||||
constexpr auto SSEBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
|
||||
|
||||
uint64_t Offset = Op->DestIdx * Op->Header.ElementSize * 8;
|
||||
__uint128_t Mask = (1ULL << (Op->Header.ElementSize * 8)) - 1;
|
||||
if (Op->Header.ElementSize == 8) {
|
||||
const uint64_t Offset = Op->DestIdx * ElementSizeBits;
|
||||
const auto InUpperLane = Offset >= SSEBitSize;
|
||||
|
||||
__uint128_t Mask = (1ULL << ElementSizeBits) - 1;
|
||||
if (ElementSize == 8) {
|
||||
Mask = ~0ULL;
|
||||
}
|
||||
Src2 = Src2 & Mask;
|
||||
Mask <<= Offset;
|
||||
|
||||
const auto Src1 = *GetSrc<InterpVector256*>(Data->SSAData, Op->DestVector);
|
||||
const auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Src);
|
||||
|
||||
const auto Scalar = Src2 & Mask;
|
||||
const auto ScaledOffset = InUpperLane ? Offset - SSEBitSize
|
||||
: Offset;
|
||||
|
||||
// Now shift into place and set all bits but
|
||||
// the ones where we're going to insert our value.
|
||||
Mask <<= ScaledOffset;
|
||||
Mask = ~Mask;
|
||||
__uint128_t Dst = Src1 & Mask;
|
||||
Dst |= Src2 << Offset;
|
||||
|
||||
const auto Dst = [&] {
|
||||
if (InUpperLane) {
|
||||
return InterpVector256{
|
||||
.Lower = Src1.Lower,
|
||||
.Upper = (Src1.Upper & Mask) | (Scalar << ScaledOffset),
|
||||
};
|
||||
} else {
|
||||
return InterpVector256{
|
||||
.Lower = (Src1.Lower & Mask) | (Scalar << ScaledOffset),
|
||||
.Upper = Src1.Upper,
|
||||
};
|
||||
}
|
||||
}();
|
||||
|
||||
memcpy(GDP, &Dst, OpSize);
|
||||
}
|
||||
@@ -89,63 +113,73 @@ DEF_OP(Vector_SToF) {
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
|
||||
const auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, float, int32_t, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, double, int64_t, Func, 0, 0)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToZS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
|
||||
const auto Func = [](auto a, auto min, auto max) { return std::trunc(a); };
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, float, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, double, Func, 0, 0)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
|
||||
const auto Func = [](auto a, auto min, auto max) { return std::nearbyint(a); };
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP(4, int32_t, float, Func, 0, 0)
|
||||
DO_VECTOR_1SRC_2TYPE_OP(8, int64_t, double, Func, 0, 0)
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const uint16_t ElementSize = Op->Header.ElementSize;
|
||||
const uint16_t Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
const auto Func = [](auto a, auto min, auto max) { return a; };
|
||||
switch (Conv) {
|
||||
@@ -165,19 +199,22 @@ DEF_OP(Vector_FToF) {
|
||||
DO_VECTOR_1SRC_2TYPE_OP_NOSIZE(float, double, Func, 0, 0)
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Conversion Type : 0x{:04x}", Conv); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Conversion Type : 0x{:04x}", Conv);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToI) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Vector);
|
||||
uint8_t Tmp[16]{};
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE]{};
|
||||
|
||||
const uint8_t Elements = OpSize / Op->Header.ElementSize;
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
const auto Func_Nearest = [](auto a) { return std::rint(a); };
|
||||
const auto Func_Neg = [](auto a) { return std::floor(a); };
|
||||
const auto Func_Pos = [](auto a) { return std::ceil(a); };
|
||||
@@ -186,31 +223,31 @@ DEF_OP(Vector_FToI) {
|
||||
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Nearest)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Nearest)
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Neg)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Neg)
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Pos)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Pos)
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Trunc)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Trunc)
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_1SRC_OP(4, float, Func_Host)
|
||||
DO_VECTOR_1SRC_OP(8, double, Func_Host)
|
||||
}
|
||||
|
||||
@@ -181,13 +181,10 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
// Move ops
|
||||
REGISTER_OP(EXTRACTELEMENTPAIR, ExtractElementPair);
|
||||
REGISTER_OP(CREATEELEMENTPAIR, CreateElementPair);
|
||||
REGISTER_OP(MOV, Mov);
|
||||
|
||||
// Vector ops
|
||||
REGISTER_OP(VECTORZERO, VectorZero);
|
||||
REGISTER_OP(VECTORIMM, VectorImm);
|
||||
REGISTER_OP(SPLATVECTOR2, SplatVector);
|
||||
REGISTER_OP(SPLATVECTOR4, SplatVector);
|
||||
REGISTER_OP(VMOV, VMov);
|
||||
REGISTER_OP(VAND, VAnd);
|
||||
REGISTER_OP(VBIC, VBic);
|
||||
@@ -246,8 +243,6 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(VUSHRS, VUShrS);
|
||||
REGISTER_OP(VSSHRS, VSShrS);
|
||||
REGISTER_OP(VINSELEMENT, VInsElement);
|
||||
REGISTER_OP(VINSSCALARELEMENT, VInsScalarElement);
|
||||
REGISTER_OP(VEXTRACTELEMENT, VExtractElement);
|
||||
REGISTER_OP(VDUPELEMENT, VDupElement);
|
||||
REGISTER_OP(VEXTR, VExtr);
|
||||
REGISTER_OP(VSLI, VSLI);
|
||||
@@ -257,7 +252,6 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(VSHLI, VShlI);
|
||||
REGISTER_OP(VUSHRNI, VUShrNI);
|
||||
REGISTER_OP(VUSHRNI2, VUShrNI2);
|
||||
REGISTER_OP(VBITCAST, VBitcast);
|
||||
REGISTER_OP(VSXTL, VSXTL);
|
||||
REGISTER_OP(VSXTL2, VSXTL2);
|
||||
REGISTER_OP(VUXTL, VUXTL);
|
||||
|
||||
@@ -207,7 +207,6 @@ namespace FEXCore::CPU {
|
||||
///< Vector ops
|
||||
DEF_OP(VectorZero);
|
||||
DEF_OP(VectorImm);
|
||||
DEF_OP(SplatVector);
|
||||
DEF_OP(VMov);
|
||||
DEF_OP(VAnd);
|
||||
DEF_OP(VBic);
|
||||
@@ -264,8 +263,6 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(VUShrS);
|
||||
DEF_OP(VSShrS);
|
||||
DEF_OP(VInsElement);
|
||||
DEF_OP(VInsScalarElement);
|
||||
DEF_OP(VExtractElement);
|
||||
DEF_OP(VDupElement);
|
||||
DEF_OP(VExtr);
|
||||
DEF_OP(VSLI);
|
||||
@@ -275,7 +272,6 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(VShlI);
|
||||
DEF_OP(VUShrNI);
|
||||
DEF_OP(VUShrNI2);
|
||||
DEF_OP(VBitcast);
|
||||
DEF_OP(VSXTL);
|
||||
DEF_OP(VSXTL2);
|
||||
DEF_OP(VUXTL);
|
||||
|
||||
@@ -25,40 +25,45 @@ static inline void CacheLineFlush(char *Addr) {
|
||||
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(LoadContext) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
const auto Src = ContextPtr + Op->Offset;
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += Op->Offset;
|
||||
#define LOAD_CTX(x, y) \
|
||||
case x: { \
|
||||
y const *MemData = reinterpret_cast<y const*>(ContextPtr); \
|
||||
y const *MemData = reinterpret_cast<y const*>(Src); \
|
||||
GD = *MemData; \
|
||||
break; \
|
||||
}
|
||||
|
||||
switch (OpSize) {
|
||||
LOAD_CTX(1, uint8_t)
|
||||
LOAD_CTX(2, uint16_t)
|
||||
LOAD_CTX(4, uint32_t)
|
||||
LOAD_CTX(8, uint64_t)
|
||||
case 16: {
|
||||
void const *MemData = reinterpret_cast<void const*>(ContextPtr);
|
||||
case 16:
|
||||
case 32: {
|
||||
void const *MemData = reinterpret_cast<void const*>(Src);
|
||||
memcpy(GDP, MemData, OpSize);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
#undef LOAD_CTX
|
||||
}
|
||||
|
||||
DEF_OP(StoreContext) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += Op->Offset;
|
||||
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
const auto Dst = ContextPtr + Op->Offset;
|
||||
|
||||
void *MemData = reinterpret_cast<void*>(ContextPtr);
|
||||
void *MemData = reinterpret_cast<void*>(Dst);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Value);
|
||||
memcpy(MemData, Src, OpSize);
|
||||
}
|
||||
@@ -72,46 +77,51 @@ DEF_OP(StoreRegister) {
|
||||
}
|
||||
|
||||
DEF_OP(LoadContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
uint64_t Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
|
||||
const auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
const auto Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
|
||||
|
||||
ContextPtr += Op->BaseOffset;
|
||||
ContextPtr += Index * Op->Stride;
|
||||
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
const auto Src = ContextPtr + Op->BaseOffset + (Index * Op->Stride);
|
||||
|
||||
#define LOAD_CTX(x, y) \
|
||||
case x: { \
|
||||
y const *MemData = reinterpret_cast<y const*>(ContextPtr); \
|
||||
y const *MemData = reinterpret_cast<y const*>(Src); \
|
||||
GD = *MemData; \
|
||||
break; \
|
||||
}
|
||||
switch (IROp->Size) {
|
||||
|
||||
switch (OpSize) {
|
||||
LOAD_CTX(1, uint8_t)
|
||||
LOAD_CTX(2, uint16_t)
|
||||
LOAD_CTX(4, uint32_t)
|
||||
LOAD_CTX(8, uint64_t)
|
||||
case 16: {
|
||||
void const *MemData = reinterpret_cast<void const*>(ContextPtr);
|
||||
memcpy(GDP, MemData, IROp->Size);
|
||||
case 16:
|
||||
case 32: {
|
||||
void const *MemData = reinterpret_cast<void const*>(Src);
|
||||
memcpy(GDP, MemData, OpSize);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
#undef LOAD_CTX
|
||||
}
|
||||
|
||||
DEF_OP(StoreContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
uint64_t Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
|
||||
const auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += Op->BaseOffset;
|
||||
ContextPtr += Index * Op->Stride;
|
||||
const auto Index = *GetSrc<uint64_t*>(Data->SSAData, Op->Index);
|
||||
|
||||
void *MemData = reinterpret_cast<void*>(ContextPtr);
|
||||
const auto ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
const auto Dst = ContextPtr + Op->BaseOffset + (Index * Op->Stride);
|
||||
|
||||
void *MemData = reinterpret_cast<void*>(Dst);
|
||||
void *Src = GetSrc<void*>(Data->SSAData, Op->Value);
|
||||
memcpy(MemData, Src, IROp->Size);
|
||||
memcpy(MemData, Src, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(SpillRegister) {
|
||||
@@ -144,8 +154,8 @@ DEF_OP(StoreFlag) {
|
||||
}
|
||||
|
||||
DEF_OP(LoadMem) {
|
||||
auto Op = IROp->C<IR::IROp_LoadMem>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_LoadMem>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
uint8_t const *MemData = *GetSrc<uint8_t const**>(Data->SSAData, Op->Addr);
|
||||
|
||||
@@ -158,7 +168,8 @@ DEF_OP(LoadMem) {
|
||||
case IR::MEM_OFFSET_SXTW.Val: MemData += (int32_t)Offset; break;
|
||||
}
|
||||
}
|
||||
memset(GDP, 0, 16);
|
||||
|
||||
memset(GDP, 0, Core::CPUState::XMM_AVX_REG_SIZE);
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
auto D = reinterpret_cast<const std::atomic<uint8_t>*>(MemData);
|
||||
@@ -180,16 +191,15 @@ DEF_OP(LoadMem) {
|
||||
GD = D->load();
|
||||
break;
|
||||
}
|
||||
|
||||
default:
|
||||
memcpy(GDP, MemData, IROp->Size);
|
||||
memcpy(GDP, MemData, OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreMem) {
|
||||
auto Op = IROp->C<IR::IROp_StoreMem>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_StoreMem>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
uint8_t *MemData = *GetSrc<uint8_t **>(Data->SSAData, Op->Addr);
|
||||
|
||||
@@ -221,7 +231,7 @@ DEF_OP(StoreMem) {
|
||||
}
|
||||
|
||||
default:
|
||||
memcpy(MemData, GetSrc<void*>(Data->SSAData, Op->Value), IROp->Size);
|
||||
memcpy(MemData, GetSrc<void*>(Data->SSAData, Op->Value), OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -19,13 +19,6 @@ $end_info$
|
||||
#include <sys/random.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
[[noreturn]]
|
||||
static void StopThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
Thread->CTX->StopThread(Thread);
|
||||
|
||||
LOGMAN_MSG_A_FMT("unreachable");
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(Fence) {
|
||||
|
||||
@@ -30,13 +30,6 @@ DEF_OP(CreateElementPair) {
|
||||
memcpy(Dst + IROp->ElementSize, Src_Upper, IROp->ElementSize);
|
||||
}
|
||||
|
||||
DEF_OP(Mov) {
|
||||
auto Op = IROp->C<IR::IROp_Mov>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
memcpy(GDP, GetSrc<void*>(Data->SSAData, Op->Value), OpSize);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
+797
-559
File diff suppressed because it is too large.
Load diff
+72
-18
@@ -1168,25 +1168,79 @@ DEF_OP(Select) {
|
||||
}
|
||||
|
||||
DEF_OP(VExtractToGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
umov(GetReg<RA_32>(Node), GetSrc(Op->Vector.ID()).V16B(), Op->Index);
|
||||
break;
|
||||
case 2:
|
||||
umov(GetReg<RA_32>(Node), GetSrc(Op->Vector.ID()).V8H(), Op->Index);
|
||||
break;
|
||||
case 4:
|
||||
umov(GetReg<RA_32>(Node), GetSrc(Op->Vector.ID()).V4S(), Op->Index);
|
||||
break;
|
||||
case 8:
|
||||
umov(GetReg<RA_64>(Node), GetSrc(Op->Vector.ID()).V2D(), Op->Index);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled ExtractElementSize: {}", OpSize);
|
||||
break;
|
||||
constexpr auto AVXRegBitSize = Core::CPUState::XMM_AVX_REG_SIZE * 8;
|
||||
constexpr auto SSERegBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
|
||||
const auto ElementSizeBits = Op->Header.ElementSize * 8;
|
||||
|
||||
const auto Offset = ElementSizeBits * Op->Index;
|
||||
const auto Is256Bit = Offset >= SSERegBitSize;
|
||||
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
const auto PerformMove = [&](const aarch64::VRegister& reg, int index) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
umov(GetReg<RA_32>(Node), reg.V16B(), index);
|
||||
break;
|
||||
case 2:
|
||||
umov(GetReg<RA_32>(Node), reg.V8H(), index);
|
||||
break;
|
||||
case 4:
|
||||
umov(GetReg<RA_32>(Node), reg.V4S(), index);
|
||||
break;
|
||||
case 8:
|
||||
umov(GetReg<RA_64>(Node), reg.V2D(), index);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled ExtractElementSize: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
};
|
||||
|
||||
if (Offset < SSERegBitSize) {
|
||||
// Desired data lies within the lower 128-bit lane, so we
|
||||
// can treat the operation as a 128-bit operation, even
|
||||
// when acting on larger register sizes.
|
||||
PerformMove(Vector, Op->Index);
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(HostSupportsSVE,
|
||||
"Host doesn't support SVE. Cannot perform 256-bit operation.");
|
||||
LOGMAN_THROW_AA_FMT(Is256Bit,
|
||||
"Can't perform 256-bit extraction with op side: {}", OpSize);
|
||||
LOGMAN_THROW_AA_FMT(Offset < AVXRegBitSize,
|
||||
"Trying to extract element outside bounds of register. Offset={}, Index={}",
|
||||
Offset, Op->Index);
|
||||
|
||||
// We need to use the upper 128-bit lane, so lets move it down.
|
||||
// Inverting our dedicated predicate for 128-bit operations selects
|
||||
// all of the top lanes. We can then compact those into a temporary.
|
||||
const auto CompactPred = p0;
|
||||
not_(CompactPred.VnB(), PRED_TMP_32B.Zeroing(), PRED_TMP_16B.VnB());
|
||||
compact(VTMP1.Z().VnD(), CompactPred, Vector.Z().VnD());
|
||||
|
||||
// Sanitize the zero-based index to work on the now-moved
|
||||
// upper half of the vector.
|
||||
const auto SanitizedIndex = [OpSize, Op] {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
return Op->Index - 16;
|
||||
case 2:
|
||||
return Op->Index - 8;
|
||||
case 4:
|
||||
return Op->Index - 4;
|
||||
case 8:
|
||||
return Op->Index - 2;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled OpSize: {}", OpSize);
|
||||
return 0;
|
||||
}
|
||||
}();
|
||||
|
||||
// Move the value from the now-low-lane data.
|
||||
PerformMove(VTMP1, SanitizedIndex);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -40,7 +40,7 @@ DEF_OP(CallbackReturn) {
|
||||
ResetStack();
|
||||
|
||||
// We can now lower the ref counter again
|
||||
|
||||
|
||||
ldr(w2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter)));
|
||||
sub(w2, w2, 1);
|
||||
str(w2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter)));
|
||||
@@ -197,7 +197,11 @@ DEF_OP(Syscall) {
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SyscallHandlerFunc)));
|
||||
mov(x1, STATE);
|
||||
mov(x2, sp);
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint64_t, void*, void*, void*>(x3);
|
||||
#else
|
||||
blr(x3);
|
||||
#endif
|
||||
|
||||
add(sp, sp, SPOffset);
|
||||
|
||||
@@ -239,7 +243,6 @@ DEF_OP(InlineSyscall) {
|
||||
bool Intersects{};
|
||||
// We always need to spill x8 since we can't know if it is live at this SSA location
|
||||
uint32_t SpillMask = 1U << 8;
|
||||
std::vector<vixl::aarch64::Register> IntersectRegs(FEXCore::HLE::SyscallArguments::MAX_ARGS);
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS-1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) break;
|
||||
|
||||
@@ -381,7 +384,11 @@ DEF_OP(Thunk) {
|
||||
|
||||
auto thunkFn = ThreadState->CTX->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
LoadConstant(x2, (uintptr_t)thunkFn);
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<void, void*, void*>(x2);
|
||||
#else
|
||||
blr(x2);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -448,7 +455,11 @@ DEF_OP(ThreadRemoveCodeEntry) {
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.ThreadRemoveCodeEntryFromJIT)));
|
||||
SpillStaticRegs();
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<void, void*, void*>(x2);
|
||||
#else
|
||||
blr(x2);
|
||||
#endif
|
||||
FillStaticRegs();
|
||||
|
||||
// Fix the stack and any values that were stepped on
|
||||
@@ -459,6 +470,7 @@ DEF_OP(CPUID) {
|
||||
auto Op = IROp->C<IR::IROp_CPUID>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
SpillStaticRegs();
|
||||
|
||||
// x0 = CPUID Handler
|
||||
// x1 = CPUID Function
|
||||
@@ -467,10 +479,13 @@ DEF_OP(CPUID) {
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDFunction)));
|
||||
mov(x1, GetReg<RA_64>(Op->Function.ID()));
|
||||
mov(x2, GetReg<RA_64>(Op->Leaf.ID()));
|
||||
SpillStaticRegs();
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<__uint128_t, void*, uint64_t, uint64_t>(x3);
|
||||
#else
|
||||
blr(x3);
|
||||
FillStaticRegs();
|
||||
#endif
|
||||
|
||||
FillStaticRegs();
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Results are in x0, x1
|
||||
|
||||
+459
-100
@@ -12,26 +12,115 @@ using namespace vixl;
|
||||
using namespace vixl::aarch64;
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(VInsGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
mov(GetDst(Node), GetSrc(Op->DestVector.ID()));
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1: {
|
||||
ins(GetDst(Node).V16B(), Op->DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
const auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto DestIdx = Op->DestIdx;
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto DestVector = GetSrc(Op->DestVector.ID());
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
const auto ElementSizeBits = ElementSize * 8;
|
||||
const auto Offset = ElementSizeBits * DestIdx;
|
||||
|
||||
const auto SSEBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
|
||||
const auto InUpperLane = Offset >= SSEBitSize;
|
||||
|
||||
// This is going to be a little gross. Pls forgive me.
|
||||
// Since SVE has the whole vector length agnostic programming
|
||||
// thing going on, we can't exactly freely insert entries into
|
||||
// arbitrary locations in the vector.
|
||||
//
|
||||
// SVE *does* have INSR, however this only shifts the entire
|
||||
// vector to the left by an element size and inserts a value
|
||||
// at the beginning of the vector. Not *quite* what we need.
|
||||
// (though INSR *is* very useful for other things).
|
||||
//
|
||||
// The idea is (in the case of the upper lane), move the upper
|
||||
// lane down, insert into it and recombine with the lower lane.
|
||||
//
|
||||
// In the case of the lower lane, insert and then recombine with
|
||||
// the upper lane.
|
||||
|
||||
if (InUpperLane) {
|
||||
// Move the upper lane down for the insertion.
|
||||
const auto CompactPred = p0;
|
||||
not_(CompactPred.VnB(), PRED_TMP_32B.Zeroing(), PRED_TMP_16B.VnB());
|
||||
compact(VTMP1.Z().VnD(), CompactPred, DestVector.Z().VnD());
|
||||
}
|
||||
case 2: {
|
||||
ins(GetDst(Node).V8H(), Op->DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
|
||||
// Put data in place for destructive SPLICE below.
|
||||
mov(Dst.Z().VnD(), DestVector.Z().VnD());
|
||||
|
||||
// Inserts the GPR value into the given V register.
|
||||
// Also automatically adjusts the index in the case of using the
|
||||
// moved upper lane.
|
||||
const auto Insert = [&](const aarch64::VRegister& reg, int index) {
|
||||
switch (ElementSize) {
|
||||
case 1:
|
||||
if (InUpperLane) {
|
||||
index -= 16;
|
||||
}
|
||||
ins(reg.V16B(), index, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
case 2:
|
||||
if (InUpperLane) {
|
||||
index -= 8;
|
||||
}
|
||||
ins(reg.V8H(), index, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
case 4:
|
||||
if (InUpperLane) {
|
||||
index -= 4;
|
||||
}
|
||||
ins(reg.V4S(), index, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
case 8:
|
||||
if (InUpperLane) {
|
||||
index -= 2;
|
||||
}
|
||||
ins(reg.V2D(), index, GetReg<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
};
|
||||
|
||||
if (InUpperLane) {
|
||||
Insert(VTMP1, DestIdx);
|
||||
splice(Dst.Z().VnD(), PRED_TMP_16B, Dst.Z().VnD(), VTMP1.Z().VnD());
|
||||
} else {
|
||||
Insert(Dst, DestIdx);
|
||||
splice(Dst.Z().VnD(), PRED_TMP_16B, Dst.Z().VnD(), DestVector.Z().VnD());
|
||||
}
|
||||
case 4: {
|
||||
ins(GetDst(Node).V4S(), Op->DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
} else {
|
||||
mov(Dst, DestVector);
|
||||
|
||||
switch (ElementSize) {
|
||||
case 1: {
|
||||
ins(Dst.V16B(), DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
ins(Dst.V8H(), DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
ins(Dst.V4S(), DestIdx, GetReg<RA_32>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
ins(Dst.V2D(), DestIdx, GetReg<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
ins(GetDst(Node).V2D(), Op->DestIdx, GetReg<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -57,8 +146,11 @@ DEF_OP(VCastFromGPR) {
|
||||
}
|
||||
|
||||
DEF_OP(Float_FromGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
|
||||
const uint16_t ElementSize = Op->Header.ElementSize;
|
||||
const uint16_t Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0404: { // Float <- int32_t
|
||||
scvtf(GetDst(Node).S(), GetReg<RA_32>(Op->Src.ID()));
|
||||
@@ -76,6 +168,10 @@ DEF_OP(Float_FromGPR_S) {
|
||||
scvtf(GetDst(Node).D(), GetReg<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled conversion mask: Mask=0x{:04x}, ElementSize={}, SrcElementSize={}",
|
||||
Conv, ElementSize, Op->SrcElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -96,116 +192,379 @@ DEF_OP(Float_FToF) {
|
||||
}
|
||||
|
||||
DEF_OP(Vector_SToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
scvtf(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
scvtf(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", Op->Header.ElementSize);
|
||||
const auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
scvtf(Dst.Z().VnH(), Mask, Vector.Z().VnH());
|
||||
break;
|
||||
case 4:
|
||||
scvtf(Dst.Z().VnS(), Mask, Vector.Z().VnS());
|
||||
break;
|
||||
case 8:
|
||||
scvtf(Dst.Z().VnD(), Mask, Vector.Z().VnD());
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
scvtf(Dst.V8H(), Vector.V8H());
|
||||
break;
|
||||
case 4:
|
||||
scvtf(Dst.V4S(), Vector.V4S());
|
||||
break;
|
||||
case 8:
|
||||
scvtf(Dst.V2D(), Vector.V2D());
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToZS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
fcvtzs(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
break;
|
||||
case 8:
|
||||
fcvtzs(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", Op->Header.ElementSize);
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
fcvtzs(Dst.Z().VnH(), Mask, Vector.Z().VnH());
|
||||
break;
|
||||
case 4:
|
||||
fcvtzs(Dst.Z().VnS(), Mask, Vector.Z().VnS());
|
||||
break;
|
||||
case 8:
|
||||
fcvtzs(Dst.Z().VnD(), Mask, Vector.Z().VnD());
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
fcvtzs(Dst.V8H(), Vector.V8H());
|
||||
break;
|
||||
case 4:
|
||||
fcvtzs(Dst.V4S(), Vector.V4S());
|
||||
break;
|
||||
case 8:
|
||||
fcvtzs(Dst.V2D(), Vector.V2D());
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frinti(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
fcvtzs(GetDst(Node).V4S(), GetDst(Node).V4S());
|
||||
break;
|
||||
case 8:
|
||||
frinti(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
fcvtzs(GetDst(Node).V2D(), GetDst(Node).V2D());
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", Op->Header.ElementSize);
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frinti(Dst.Z().VnH(), Mask, Vector.Z().VnH());
|
||||
fcvtzs(Dst.Z().VnH(), Mask, Dst.Z().VnH());
|
||||
break;
|
||||
case 4:
|
||||
frinti(Dst.Z().VnS(), Mask, Vector.Z().VnS());
|
||||
fcvtzs(Dst.Z().VnS(), Mask, Dst.Z().VnS());
|
||||
break;
|
||||
case 8:
|
||||
frinti(Dst.Z().VnD(), Mask, Vector.Z().VnD());
|
||||
fcvtzs(Dst.Z().VnD(), Mask, Dst.Z().VnD());
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frinti(Dst.V8H(), Vector.V8H());
|
||||
fcvtzs(Dst.V8H(), Dst.V8H());
|
||||
break;
|
||||
case 4:
|
||||
frinti(Dst.V4S(), Vector.V4S());
|
||||
fcvtzs(Dst.V4S(), Dst.V4S());
|
||||
break;
|
||||
case 8:
|
||||
frinti(Dst.V2D(), Vector.V2D());
|
||||
fcvtzs(Dst.V2D(), Dst.V2D());
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- Float
|
||||
fcvtl(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2S());
|
||||
break;
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
// Curiously, FCVTLT and FCVTNT have no bottom variants,
|
||||
// and also interesting is that FCVTLT will iterate the
|
||||
// source vector by accessing each odd element and storing
|
||||
// them consecutively in the destination.
|
||||
//
|
||||
// FCVTNT is somewhat like the opposite. It will read each
|
||||
// consecutive element, but store each result into every odd
|
||||
// element in the destination vector.
|
||||
//
|
||||
// We need to undo the behavior of FCVTNT with UZP2. In the case
|
||||
// of FCVTLT, we instead need to set the vector up with ZIP1, so
|
||||
// that the elements will be processed correctly.
|
||||
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0402: { // Float <- Half
|
||||
zip1(Dst.Z().VnH(), Vector.Z().VnH(), Vector.Z().VnH());
|
||||
fcvtlt(Dst.Z().VnS(), Mask, Dst.Z().VnH());
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- Float
|
||||
zip1(Dst.Z().VnS(), Vector.Z().VnS(), Vector.Z().VnS());
|
||||
fcvtlt(Dst.Z().VnD(), Mask, Dst.Z().VnS());
|
||||
break;
|
||||
}
|
||||
case 0x0204: { // Half <- Float
|
||||
fcvtnt(Dst.Z().VnH(), Mask, Vector.Z().VnS());
|
||||
uzp2(Dst.Z().VnH(), Dst.Z().VnH(), Dst.Z().VnH());
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvtnt(Dst.Z().VnS(), Mask, Vector.Z().VnD());
|
||||
uzp2(Dst.Z().VnS(), Dst.Z().VnS(), Dst.Z().VnS());
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv);
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvtn(GetDst(Node).V2S(), GetSrc(Op->Vector.ID()).V2D());
|
||||
break;
|
||||
} else {
|
||||
switch (Conv) {
|
||||
case 0x0402: { // Float <- Half
|
||||
fcvtl(Dst.V4S(), Vector.V4H());
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- Float
|
||||
fcvtl(Dst.V2D(), Vector.V2S());
|
||||
break;
|
||||
}
|
||||
case 0x0204: { // Half <- Float
|
||||
fcvtn(Dst.V4H(), Vector.V4S());
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvtn(Dst.V2S(), Vector.V2D());
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToF Type : 0x{:04x}", Conv); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToI) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintn(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frintn(Dst.Z().VnH(), Mask, Vector.Z().VnH());
|
||||
break;
|
||||
case 4:
|
||||
frintn(Dst.Z().VnS(), Mask, Vector.Z().VnS());
|
||||
break;
|
||||
case 8:
|
||||
frintn(Dst.Z().VnD(), Mask, Vector.Z().VnD());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
frintn(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frintm(Dst.Z().VnH(), Mask, Vector.Z().VnH());
|
||||
break;
|
||||
case 4:
|
||||
frintm(Dst.Z().VnS(), Mask, Vector.Z().VnS());
|
||||
break;
|
||||
case 8:
|
||||
frintm(Dst.Z().VnD(), Mask, Vector.Z().VnD());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintm(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frintp(Dst.Z().VnH(), Mask, Vector.Z().VnH());
|
||||
break;
|
||||
case 4:
|
||||
frintp(Dst.Z().VnS(), Mask, Vector.Z().VnS());
|
||||
break;
|
||||
case 8:
|
||||
frintp(Dst.Z().VnD(), Mask, Vector.Z().VnD());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
frintm(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frintz(Dst.Z().VnH(), Mask, Vector.Z().VnH());
|
||||
break;
|
||||
case 4:
|
||||
frintz(Dst.Z().VnS(), Mask, Vector.Z().VnS());
|
||||
break;
|
||||
case 8:
|
||||
frintz(Dst.Z().VnD(), Mask, Vector.Z().VnD());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintp(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frinti(Dst.Z().VnH(), Mask, Vector.Z().VnH());
|
||||
break;
|
||||
case 4:
|
||||
frinti(Dst.Z().VnS(), Mask, Vector.Z().VnS());
|
||||
break;
|
||||
case 8:
|
||||
frinti(Dst.Z().VnD(), Mask, Vector.Z().VnD());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
frintp(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
}
|
||||
} else {
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frintn(Dst.V8H(), Vector.V8H());
|
||||
break;
|
||||
case 4:
|
||||
frintn(Dst.V4S(), Vector.V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintn(Dst.V2D(), Vector.V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frintz(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frintm(Dst.V8H(), Vector.V8H());
|
||||
break;
|
||||
case 4:
|
||||
frintm(Dst.V4S(), Vector.V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintm(Dst.V2D(), Vector.V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
frintz(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frintp(Dst.V8H(), Vector.V8H());
|
||||
break;
|
||||
case 4:
|
||||
frintp(Dst.V4S(), Vector.V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintp(Dst.V2D(), Vector.V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 4:
|
||||
frinti(GetDst(Node).V4S(), GetSrc(Op->Vector.ID()).V4S());
|
||||
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frintz(Dst.V8H(), Vector.V8H());
|
||||
break;
|
||||
case 4:
|
||||
frintz(Dst.V4S(), Vector.V4S());
|
||||
break;
|
||||
case 8:
|
||||
frintz(Dst.V2D(), Vector.V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
frinti(GetDst(Node).V2D(), GetSrc(Op->Vector.ID()).V2D());
|
||||
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
switch (ElementSize) {
|
||||
case 2:
|
||||
frinti(Dst.V8H(), Vector.V8H());
|
||||
break;
|
||||
case 4:
|
||||
frinti(Dst.V4S(), Vector.V4S());
|
||||
break;
|
||||
case 8:
|
||||
frinti(Dst.V2D(), Vector.V2D());
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+77
-11
@@ -28,6 +28,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
|
||||
@@ -94,7 +95,12 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
|
||||
uxth(w0, GetReg<RA_32>(IROp->Args[0].ID()));
|
||||
ldr(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<void, uint16_t>(x1);
|
||||
#else
|
||||
blr(x1);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -109,7 +115,11 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
|
||||
fmov(v0.S(), GetSrc(IROp->Args[0].ID()).S()) ;
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<__uint128_t, float>(x0);
|
||||
#else
|
||||
blr(x0);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -128,7 +138,11 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
|
||||
mov(v0.D(), GetSrc(IROp->Args[0].ID()).D());
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<__uint128_t, double>(x0);
|
||||
#else
|
||||
blr(x0);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -153,7 +167,11 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
mov(w0, GetReg<RA_32>(IROp->Args[0].ID()));
|
||||
}
|
||||
ldr(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint32_t>(x1);
|
||||
#else
|
||||
blr(x1);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -174,7 +192,11 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<float, uint64_t, uint64_t>(x2);
|
||||
#else
|
||||
blr(x2);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -193,7 +215,11 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<double, uint64_t, uint64_t>(x2);
|
||||
#else
|
||||
blr(x2);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -210,7 +236,11 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
|
||||
mov(v0.D(), GetSrc(IROp->Args[0].ID()).D());
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<double, double>(x0);
|
||||
#else
|
||||
blr(x0);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -229,7 +259,11 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
mov(v0.D(), GetSrc(IROp->Args[0].ID()).D());
|
||||
mov(v1.D(), GetSrc(IROp->Args[1].ID()).D());
|
||||
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<double, double, double>(x0);
|
||||
#else
|
||||
blr(x0);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -249,7 +283,11 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint64_t, uint64_t>(x2);
|
||||
#else
|
||||
blr(x2);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -267,7 +305,11 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint64_t, uint64_t>(x2);
|
||||
#else
|
||||
blr(x2);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -285,7 +327,11 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t>(x2);
|
||||
#else
|
||||
blr(x2);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -306,8 +352,11 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(w3, GetSrc(IROp->Args[1].ID()).V8H(), 4);
|
||||
|
||||
ldr(x4, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t, uint64_t>(x4);
|
||||
#else
|
||||
blr(x4);
|
||||
|
||||
#endif
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
@@ -324,7 +373,11 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(w1, GetSrc(IROp->Args[0].ID()).V8H(), 4);
|
||||
|
||||
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint64_t, uint64_t>(x2);
|
||||
#else
|
||||
blr(x2);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -347,7 +400,11 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
umov(w3, GetSrc(IROp->Args[1].ID()).V8H(), 4);
|
||||
|
||||
ldr(x4, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint64_t, uint64_t, uint64_t, uint64_t>(x4);
|
||||
#else
|
||||
blr(x4);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -473,7 +530,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::Intern
|
||||
|
||||
// Common
|
||||
auto &Common = ThreadState->CurrentFrame->Pointers.Common;
|
||||
|
||||
|
||||
Common.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
|
||||
Common.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
|
||||
Common.ThreadRemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::Context::ThreadRemoveCodeEntryFromJit);
|
||||
@@ -494,7 +551,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::Intern
|
||||
|
||||
// Platform Specific
|
||||
auto &AArch64 = ThreadState->CurrentFrame->Pointers.AArch64;
|
||||
|
||||
|
||||
AArch64.LUDIV = reinterpret_cast<uint64_t>(LUDIV);
|
||||
AArch64.LDIV = reinterpret_cast<uint64_t>(LDIV);
|
||||
AArch64.LUREM = reinterpret_cast<uint64_t>(LUREM);
|
||||
@@ -511,6 +568,7 @@ void Arm64JITCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
return Thread->CTX->Dispatcher->HandleSIGILL(Thread, Signal, info, ucontext);
|
||||
}, true);
|
||||
|
||||
#ifdef _M_ARM_64
|
||||
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
|
||||
if (!Thread->CPUBackend->IsAddressInCodeBuffer(ArchHelpers::Context::GetPc(ucontext))) {
|
||||
// Wasn't a sigbus in JIT code
|
||||
@@ -519,6 +577,7 @@ void Arm64JITCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
|
||||
|
||||
return FEXCore::ArchHelpers::Arm64::HandleSIGBUS(Thread->CTX->Config.ParanoidTSO(), Signal, info, ucontext);
|
||||
}, true);
|
||||
#endif
|
||||
}
|
||||
|
||||
void Arm64JITCore::EmitDetectionString() {
|
||||
@@ -530,7 +589,7 @@ void Arm64JITCore::EmitDetectionString() {
|
||||
|
||||
void Arm64JITCore::ClearCache() {
|
||||
// Get the backing code buffer
|
||||
|
||||
|
||||
auto CodeBuffer = GetEmptyCodeBuffer();
|
||||
*GetBuffer() = vixl::CodeBuffer(CodeBuffer->Ptr, CodeBuffer->Size);
|
||||
EmitDetectionString();
|
||||
@@ -673,6 +732,8 @@ void *Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
FEXCore::IR::RegisterAllocationData *RAData,
|
||||
bool GDBEnabled) {
|
||||
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
|
||||
|
||||
using namespace aarch64;
|
||||
JumpTargets.clear();
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
@@ -725,10 +786,12 @@ void *Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
SpillSlots = RAData->SpillSlots();
|
||||
|
||||
if (SpillSlots) {
|
||||
if (IsImmAddSub(SpillSlots * 16)) {
|
||||
sub(sp, sp, SpillSlots * 16);
|
||||
const auto TotalSpillSlotsSize = SpillSlots * MaxSpillSlotSize;
|
||||
|
||||
if (IsImmAddSub(TotalSpillSlotsSize)) {
|
||||
sub(sp, sp, TotalSpillSlotsSize);
|
||||
} else {
|
||||
LoadConstant(x0, SpillSlots * 16);
|
||||
LoadConstant(x0, TotalSpillSlotsSize);
|
||||
sub(sp, sp, x0);
|
||||
}
|
||||
}
|
||||
@@ -796,14 +859,17 @@ void *Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
}
|
||||
|
||||
void Arm64JITCore::ResetStack() {
|
||||
if (SpillSlots == 0)
|
||||
if (SpillSlots == 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (IsImmAddSub(SpillSlots * 16)) {
|
||||
add(sp, sp, SpillSlots * 16);
|
||||
const auto TotalSpillSlotsSize = SpillSlots * MaxSpillSlotSize;
|
||||
|
||||
if (IsImmAddSub(TotalSpillSlotsSize)) {
|
||||
add(sp, sp, TotalSpillSlotsSize);
|
||||
} else {
|
||||
// Too big to fit in a 12bit immediate
|
||||
LoadConstant(x0, SpillSlots * 16);
|
||||
LoadConstant(x0, TotalSpillSlotsSize);
|
||||
add(sp, sp, x0);
|
||||
}
|
||||
}
|
||||
|
||||
+11
-16
@@ -23,16 +23,6 @@ $end_info$
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#define STATE x28
|
||||
#define TMP1 x0
|
||||
#define TMP2 x1
|
||||
#define TMP3 x2
|
||||
#define TMP4 x3
|
||||
|
||||
#define VTMP1 v1
|
||||
#define VTMP2 v2
|
||||
#define VTMP3 v3
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
@@ -128,6 +118,17 @@ private:
|
||||
IR::MemOffsetType OffsetType,
|
||||
uint8_t OffsetScale);
|
||||
|
||||
// NOTE: Will use TMP1 as a way to encode immediates that happen to fall outside
|
||||
// the limits of the scalar plus immediate variant of SVE load/stores.
|
||||
//
|
||||
// TMP1 is safe to use again once this memory operand is used with its
|
||||
// equivalent loads or stores that this was called for.
|
||||
[[nodiscard]] SVEMemOperand GenerateSVEMemOperand(uint8_t AccessSize,
|
||||
aarch64::Register Base,
|
||||
IR::OrderedNodeWrapper Offset,
|
||||
IR::MemOffsetType OffsetType,
|
||||
uint8_t OffsetScale);
|
||||
|
||||
[[nodiscard]] bool IsInlineConstant(const IR::OrderedNodeWrapper& Node, uint64_t* Value = nullptr) const;
|
||||
[[nodiscard]] bool IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const;
|
||||
|
||||
@@ -365,13 +366,10 @@ private:
|
||||
///< Move ops
|
||||
DEF_OP(ExtractElementPair);
|
||||
DEF_OP(CreateElementPair);
|
||||
DEF_OP(Mov);
|
||||
|
||||
///< Vector ops
|
||||
DEF_OP(VectorZero);
|
||||
DEF_OP(VectorImm);
|
||||
DEF_OP(SplatVector2);
|
||||
DEF_OP(SplatVector4);
|
||||
DEF_OP(VMov);
|
||||
DEF_OP(VAnd);
|
||||
DEF_OP(VBic);
|
||||
@@ -430,8 +428,6 @@ private:
|
||||
DEF_OP(VUShrS);
|
||||
DEF_OP(VSShrS);
|
||||
DEF_OP(VInsElement);
|
||||
DEF_OP(VInsScalarElement);
|
||||
DEF_OP(VExtractElement);
|
||||
DEF_OP(VDupElement);
|
||||
DEF_OP(VExtr);
|
||||
DEF_OP(VSLI);
|
||||
@@ -441,7 +437,6 @@ private:
|
||||
DEF_OP(VShlI);
|
||||
DEF_OP(VUShrNI);
|
||||
DEF_OP(VUShrNI2);
|
||||
DEF_OP(VBitcast);
|
||||
DEF_OP(VSXTL);
|
||||
DEF_OP(VSXTL2);
|
||||
DEF_OP(VUXTL);
|
||||
|
||||
+461
-281
File diff suppressed because it is too large.
Load diff
@@ -119,8 +119,11 @@ DEF_OP(SetRoundingMode) {
|
||||
|
||||
mrs(TMP1, FPCR);
|
||||
|
||||
// vixl simulator doesn't support anything beyond ties-to-even rounding
|
||||
#ifndef VIXL_SIMULATOR
|
||||
// Insert the rounding flags
|
||||
bfi(TMP1, TMP2, 22, 2);
|
||||
#endif
|
||||
|
||||
// Insert the FTZ flag
|
||||
lsr(TMP2, Src, 2);
|
||||
@@ -134,6 +137,7 @@ DEF_OP(Print) {
|
||||
auto Op = IROp->C<IR::IROp_Print>();
|
||||
|
||||
PushDynamicRegsAndLR();
|
||||
SpillStaticRegs();
|
||||
|
||||
if (IsGPR(Op->Value.ID())) {
|
||||
mov(x0, GetReg<RA_64>(Op->Value.ID()));
|
||||
@@ -145,10 +149,10 @@ DEF_OP(Print) {
|
||||
fmov(x1, GetSrc(Op->Value.ID()).V1D(), 1);
|
||||
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.PrintVectorValue)));
|
||||
}
|
||||
SpillStaticRegs();
|
||||
blr(x3);
|
||||
FillStaticRegs();
|
||||
|
||||
blr(x3);
|
||||
|
||||
FillStaticRegs();
|
||||
PopDynamicRegsAndLR();
|
||||
}
|
||||
|
||||
|
||||
@@ -68,17 +68,11 @@ DEF_OP(CreateElementPair) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Mov) {
|
||||
auto Op = IROp->C<IR::IROp_Mov>();
|
||||
mov(GetReg<RA_64>(Node), GetReg<RA_64>(Op->Value.ID()));
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void Arm64JITCore::RegisterMoveHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &Arm64JITCore::Op_##x
|
||||
REGISTER_OP(EXTRACTELEMENTPAIR, ExtractElementPair);
|
||||
REGISTER_OP(CREATEELEMENTPAIR, CreateElementPair);
|
||||
REGISTER_OP(MOV, Mov);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
+4284
-1497
File diff suppressed because it is too large.
Load diff
+44
-11
@@ -1143,26 +1143,59 @@ DEF_OP(Select) {
|
||||
}
|
||||
|
||||
DEF_OP(VExtractToGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
const auto Op = IROp->C<IR::IROp_VExtractToGPR>();
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
constexpr auto SSERegSize = Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
constexpr auto SSEBitSize = SSERegSize * 8;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto ElementSizeBits = ElementSize * 8;
|
||||
const auto Offset = ElementSizeBits * Op->Index;
|
||||
|
||||
const auto Is256Bit = Offset >= SSEBitSize;
|
||||
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (ElementSize) {
|
||||
case 1: {
|
||||
pextrb(GetDst<RA_32>(Node), GetSrc(Op->Vector.ID()), Op->Index);
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vextracti128(xmm15, ToYMM(Vector), 1);
|
||||
pextrb(GetDst<RA_32>(Node), xmm15, Op->Index - 16);
|
||||
} else {
|
||||
pextrb(GetDst<RA_32>(Node), Vector, Op->Index);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
pextrw(GetDst<RA_32>(Node), GetSrc(Op->Vector.ID()), Op->Index);
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vextracti128(xmm15, ToYMM(Vector), 1);
|
||||
pextrw(GetDst<RA_32>(Node), xmm15, Op->Index - 8);
|
||||
} else {
|
||||
pextrw(GetDst<RA_32>(Node), Vector, Op->Index);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
pextrd(GetDst<RA_32>(Node), GetSrc(Op->Vector.ID()), Op->Index);
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vextracti128(xmm15, ToYMM(Vector), 1);
|
||||
pextrd(GetDst<RA_32>(Node), xmm15, Op->Index - 4);
|
||||
} else {
|
||||
pextrd(GetDst<RA_32>(Node), Vector, Op->Index);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
pextrq(GetDst<RA_64>(Node), GetSrc(Op->Vector.ID()), Op->Index);
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vextracti128(xmm15, ToYMM(Vector), 1);
|
||||
pextrq(GetDst<RA_64>(Node), xmm15, Op->Index - 2);
|
||||
} else {
|
||||
pextrq(GetDst<RA_64>(Node), Vector, Op->Index);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -33,7 +33,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(SignalReturn) {
|
||||
// Adjust the stack first for a regular return
|
||||
if (SpillSlots) {
|
||||
add(rsp, SpillSlots * 16); // + 8 to consume return address
|
||||
add(rsp, SpillSlots * MaxSpillSlotSize); // + 8 to consume return address
|
||||
}
|
||||
|
||||
jmp(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.SignalReturnHandler)]);
|
||||
@@ -42,7 +42,7 @@ DEF_OP(SignalReturn) {
|
||||
DEF_OP(CallbackReturn) {
|
||||
// Adjust the stack first for a regular return
|
||||
if (SpillSlots) {
|
||||
add(rsp, SpillSlots * 16); // + 8 to consume return address
|
||||
add(rsp, SpillSlots * MaxSpillSlotSize); // + 8 to consume return address
|
||||
}
|
||||
|
||||
// Make sure to adjust the refcounter so we don't clear the cache now
|
||||
@@ -71,7 +71,7 @@ DEF_OP(ExitFunction) {
|
||||
|
||||
|
||||
if (SpillSlots) {
|
||||
add(rsp, SpillSlots * 16);
|
||||
add(rsp, SpillSlots * MaxSpillSlotSize);
|
||||
}
|
||||
|
||||
uint64_t NewRIP;
|
||||
|
||||
+224
-74
@@ -17,27 +17,76 @@ namespace FEXCore::CPU {
|
||||
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
DEF_OP(VInsGPR) {
|
||||
auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
movapd(GetDst(Node), GetSrc(Op->DestVector.ID()));
|
||||
const auto Op = IROp->C<IR::IROp_VInsGPR>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
case 1: {
|
||||
pinsrb(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()), Op->DestIdx);
|
||||
break;
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto DestVector = GetSrc(Op->DestVector.ID());
|
||||
|
||||
const auto DestIdx = Op->DestIdx;
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto ElementSizeBits = ElementSize * 8;
|
||||
const auto Offset = ElementSizeBits * DestIdx;
|
||||
|
||||
constexpr auto SSEBitSize = Core::CPUState::XMM_SSE_REG_SIZE * 8;
|
||||
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto InUpperLane = Offset >= SSEBitSize;
|
||||
|
||||
if (InUpperLane && !Is256Bit) {
|
||||
LOGMAN_MSG_A_FMT("Attempt to access upper 128-bit lane in 128-bit operation! Offset={}",
|
||||
Offset);
|
||||
return;
|
||||
}
|
||||
|
||||
if (Is256Bit) {
|
||||
vmovapd(ToYMM(Dst), ToYMM(DestVector));
|
||||
} else {
|
||||
vmovapd(Dst, DestVector);
|
||||
}
|
||||
|
||||
const auto Insert = [&](const Xbyak::Xmm& reg, int index) {
|
||||
switch (ElementSize) {
|
||||
case 1: {
|
||||
if (InUpperLane) {
|
||||
index -= 16;
|
||||
}
|
||||
pinsrb(reg, GetSrc<RA_32>(Op->Src.ID()), index);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
if (InUpperLane) {
|
||||
index -= 8;
|
||||
}
|
||||
pinsrw(reg, GetSrc<RA_32>(Op->Src.ID()), index);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
if (InUpperLane) {
|
||||
index -= 4;
|
||||
}
|
||||
pinsrd(reg, GetSrc<RA_32>(Op->Src.ID()), index);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
if (InUpperLane) {
|
||||
index -= 2;
|
||||
}
|
||||
pinsrq(reg, GetSrc<RA_64>(Op->Src.ID()), index);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
pinsrw(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()), Op->DestIdx);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
pinsrd(GetDst(Node), GetSrc<RA_32>(Op->Src.ID()), Op->DestIdx);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
pinsrq(GetDst(Node), GetSrc<RA_64>(Op->Src.ID()), Op->DestIdx);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Element Size: {}", Op->Header.ElementSize); break;
|
||||
};
|
||||
|
||||
if (InUpperLane) {
|
||||
vextracti128(xmm15, ToYMM(Dst), 1);
|
||||
Insert(xmm15, DestIdx);
|
||||
vinserti128(ToYMM(Dst), ToYMM(Dst), xmm15, 1);
|
||||
} else {
|
||||
Insert(Dst, DestIdx);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -63,8 +112,10 @@ DEF_OP(VCastFromGPR) {
|
||||
}
|
||||
|
||||
DEF_OP(Float_FromGPR_S) {
|
||||
auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const auto Op = IROp->C<IR::IROp_Float_FromGPR_S>();
|
||||
|
||||
const uint16_t ElementSize = Op->Header.ElementSize;
|
||||
const uint16_t Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0404: { // Float <- int32_t
|
||||
@@ -83,6 +134,10 @@ DEF_OP(Float_FromGPR_S) {
|
||||
cvtsi2sd(GetDst(Node), GetSrc<RA_64>(Op->Src.ID()));
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled conversion mask: Mask=0x{:04x}, ElementSize={}, SrcElementSize={}",
|
||||
Conv, ElementSize, Op->SrcElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -104,99 +159,194 @@ DEF_OP(Float_FToF) {
|
||||
}
|
||||
|
||||
DEF_OP(Vector_SToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
const auto Op = IROp->C<IR::IROp_Vector_SToF>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (ElementSize) {
|
||||
case 4:
|
||||
cvtdq2ps(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vcvtdq2ps(ToYMM(Dst), ToYMM(Vector));
|
||||
} else {
|
||||
vcvtdq2ps(Dst, Vector);
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
// This operation is a bit disgusting in x86
|
||||
// There is no vector form of this instruction until AVX512VL + AVX512DQ (vcvtqq2pd)
|
||||
// 1) First extract the top 64bits
|
||||
// 2) Do a scalar conversion on each
|
||||
// 3) Make sure to merge them together at the end
|
||||
pextrq(rax, GetSrc(Op->Vector.ID()), 1);
|
||||
pextrq(rcx, GetSrc(Op->Vector.ID()), 0);
|
||||
cvtsi2sd(GetDst(Node), rcx);
|
||||
pextrq(rax, Vector, 1);
|
||||
pextrq(rcx, Vector, 0);
|
||||
cvtsi2sd(Dst, rcx);
|
||||
cvtsi2sd(xmm15, rax);
|
||||
movlhps(GetDst(Node), xmm15);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", Op->Header.ElementSize);
|
||||
vmovlhps(Dst, Dst, xmm15);
|
||||
if (Is256Bit) {
|
||||
vextracti128(xmm15, ToYMM(Vector), 1);
|
||||
|
||||
pextrq(rax, xmm15, 1);
|
||||
pextrq(rcx, xmm15, 0);
|
||||
cvtsi2sd(xmm15, rcx);
|
||||
cvtsi2sd(xmm14, rax);
|
||||
movlhps(xmm15, xmm14);
|
||||
|
||||
vinserti128(ToYMM(Dst), ToYMM(Dst), xmm15, 1);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_SToF element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToZS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToZS>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (ElementSize) {
|
||||
case 4:
|
||||
cvttps2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vcvttps2dq(ToYMM(Dst), ToYMM(Vector));
|
||||
} else {
|
||||
vcvttps2dq(Dst, Vector);
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
cvttpd2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", Op->Header.ElementSize);
|
||||
if (Is256Bit) {
|
||||
vcvttpd2dq(ToYMM(Dst), ToYMM(Vector));
|
||||
} else {
|
||||
vcvttpd2dq(Dst, Vector);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToZS element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToS) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
switch (Op->Header.ElementSize) {
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToS>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (ElementSize) {
|
||||
case 4:
|
||||
cvtps2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vcvtps2dq(ToYMM(Dst), ToYMM(Vector));
|
||||
} else {
|
||||
vcvtps2dq(Dst, Vector);
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
cvtpd2dq(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", Op->Header.ElementSize);
|
||||
if (Is256Bit) {
|
||||
vcvtpd2dq(ToYMM(Dst), ToYMM(Vector));
|
||||
} else {
|
||||
vcvtpd2dq(Dst, Vector);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToS element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToF) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToF>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0804: { // Double <- Float
|
||||
cvtps2pd(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
if (Is256Bit) {
|
||||
vcvtps2pd(ToYMM(Dst), Vector);
|
||||
} else {
|
||||
vcvtps2pd(Dst, Vector);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
cvtpd2ps(GetDst(Node), GetSrc(Op->Vector.ID()));
|
||||
if (Is256Bit) {
|
||||
vcvtpd2ps(Dst, ToYMM(Vector));
|
||||
} else {
|
||||
vcvtpd2ps(Dst, Vector);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Vector_FToF conversion type : 0x{:04x}", Conv); break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Vector_FToF conversion type : 0x{:04x}", Conv);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Vector_FToI) {
|
||||
auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
uint8_t RoundMode{};
|
||||
const auto Op = IROp->C<IR::IROp_Vector_FToI>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
RoundMode = 0b0000'0'0'00;
|
||||
break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
RoundMode = 0b0000'0'0'01;
|
||||
break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
RoundMode = 0b0000'0'0'10;
|
||||
break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
RoundMode = 0b0000'0'0'11;
|
||||
break;
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
RoundMode = 0b0000'0'1'00;
|
||||
break;
|
||||
}
|
||||
const uint8_t RoundMode = [Op] {
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val:
|
||||
return 0b0000'0'0'00;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val:
|
||||
return 0b0000'0'0'01;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val:
|
||||
return 0b0000'0'0'10;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val:
|
||||
return 0b0000'0'0'11;
|
||||
case FEXCore::IR::Round_Host.Val:
|
||||
return 0b0000'0'1'00;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled rounding mode");
|
||||
return 0;
|
||||
}
|
||||
}();
|
||||
|
||||
switch (Op->Header.ElementSize) {
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector = GetSrc(Op->Vector.ID());
|
||||
|
||||
switch (ElementSize) {
|
||||
case 4:
|
||||
roundps(GetDst(Node), GetSrc(Op->Vector.ID()), RoundMode);
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vroundps(ToYMM(Dst), ToYMM(Vector), RoundMode);
|
||||
} else {
|
||||
vroundps(Dst, Vector, RoundMode);
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
roundpd(GetDst(Node), GetSrc(Op->Vector.ID()), RoundMode);
|
||||
break;
|
||||
if (Is256Bit) {
|
||||
vroundpd(ToYMM(Dst), ToYMM(Vector), RoundMode);
|
||||
} else {
|
||||
vroundpd(Dst, Vector, RoundMode);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled element size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+30
-17
@@ -27,6 +27,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
@@ -60,32 +61,42 @@ static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) {
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
void X86JITCore::PushRegs() {
|
||||
sub(rsp, 16 * RAXMM_x.size());
|
||||
const auto AVXRegSize = Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
sub(rsp, AVXRegSize * RAXMM_x.size());
|
||||
for (size_t i = 0; i < RAXMM_x.size(); ++i) {
|
||||
movaps(ptr[rsp + i * 16], RAXMM_x[i]);
|
||||
vmovups(ptr[rsp + i * AVXRegSize], ToYMM(RAXMM_x[i]));
|
||||
}
|
||||
|
||||
for (auto &Reg : RA64)
|
||||
for (const auto &Reg : RA64) {
|
||||
push(Reg);
|
||||
}
|
||||
|
||||
auto NumPush = RA64.size();
|
||||
if (NumPush & 1)
|
||||
sub(rsp, 8); // Align
|
||||
const auto NumPush = RA64.size();
|
||||
if ((NumPush & 1) != 0) {
|
||||
// Align
|
||||
sub(rsp, 8);
|
||||
}
|
||||
}
|
||||
|
||||
void X86JITCore::PopRegs() {
|
||||
auto NumPush = RA64.size();
|
||||
const auto AVXRegSize = Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto NumPush = RA64.size();
|
||||
|
||||
if (NumPush & 1)
|
||||
add(rsp, 8); // Align
|
||||
for (uint32_t i = RA64.size(); i > 0; --i)
|
||||
pop(RA64[i - 1]);
|
||||
|
||||
for (size_t i = 0; i < RAXMM_x.size(); ++i) {
|
||||
movaps(RAXMM_x[i], ptr[rsp + i * 16]);
|
||||
if ((NumPush & 1) != 0) {
|
||||
// Align
|
||||
add(rsp, 8);
|
||||
}
|
||||
|
||||
add(rsp, 16 * RAXMM_x.size());
|
||||
for (uint32_t i = RA64.size(); i > 0; --i) {
|
||||
pop(RA64[i - 1]);
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < RAXMM_x.size(); ++i) {
|
||||
vmovups(ToYMM(RAXMM_x[i]), ptr[rsp + i * AVXRegSize]);
|
||||
}
|
||||
|
||||
add(rsp, AVXRegSize * RAXMM_x.size());
|
||||
}
|
||||
|
||||
void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
@@ -360,7 +371,7 @@ X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalTh
|
||||
|
||||
{
|
||||
auto &Common = ThreadState->CurrentFrame->Pointers.Common;
|
||||
|
||||
|
||||
Common.PrintValue = reinterpret_cast<uint64_t>(PrintValue);
|
||||
Common.PrintVectorValue = reinterpret_cast<uint64_t>(PrintVectorValue);
|
||||
Common.ThreadRemoveCodeEntryFromJIT = reinterpret_cast<uintptr_t>(&Context::Context::ThreadRemoveCodeEntryFromJit);
|
||||
@@ -572,6 +583,8 @@ std::tuple<X86JITCore::SetCC, X86JITCore::CMovCC, X86JITCore::JCC> X86JITCore::G
|
||||
}
|
||||
|
||||
void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) {
|
||||
|
||||
FEXCORE_PROFILE_SCOPED("x86::CompileCode");
|
||||
JumpTargets.clear();
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
|
||||
@@ -599,7 +612,7 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
|
||||
SpillSlots = RAData->SpillSlots();
|
||||
|
||||
if (SpillSlots) {
|
||||
sub(rsp, SpillSlots * 16);
|
||||
sub(rsp, SpillSlots * MaxSpillSlotSize);
|
||||
}
|
||||
|
||||
#ifdef BLOCKSTATS
|
||||
|
||||
@@ -209,7 +209,7 @@ private:
|
||||
* @brief Current guest RIP entrypoint
|
||||
*/
|
||||
uint8_t *GuestEntry{};
|
||||
|
||||
|
||||
using SetCC = void (X86JITCore::*)(const Operand& op);
|
||||
using CMovCC = void (X86JITCore::*)(const Reg& reg, const Operand& op);
|
||||
using JCC = void (X86JITCore::*)(const Label& label, LabelType type);
|
||||
@@ -366,12 +366,10 @@ private:
|
||||
///< Move ops
|
||||
DEF_OP(ExtractElementPair);
|
||||
DEF_OP(CreateElementPair);
|
||||
DEF_OP(Mov);
|
||||
|
||||
///< Vector ops
|
||||
DEF_OP(VectorZero);
|
||||
DEF_OP(VectorImm);
|
||||
DEF_OP(SplatVector);
|
||||
DEF_OP(VMov);
|
||||
DEF_OP(VAnd);
|
||||
DEF_OP(VBic);
|
||||
@@ -430,8 +428,6 @@ private:
|
||||
DEF_OP(VUShrS);
|
||||
DEF_OP(VSShrS);
|
||||
DEF_OP(VInsElement);
|
||||
DEF_OP(VInsScalarElement);
|
||||
DEF_OP(VExtractElement);
|
||||
DEF_OP(VDupElement);
|
||||
DEF_OP(VExtr);
|
||||
DEF_OP(VSLI);
|
||||
@@ -441,7 +437,6 @@ private:
|
||||
DEF_OP(VShlI);
|
||||
DEF_OP(VUShrNI);
|
||||
DEF_OP(VUShrNI2);
|
||||
DEF_OP(VBitcast);
|
||||
DEF_OP(VSXTL);
|
||||
DEF_OP(VSXTL2);
|
||||
DEF_OP(VUXTL);
|
||||
|
||||
+265
-177
@@ -21,130 +21,160 @@ namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void X86JITCore::Op_##x(IR::IROp_Header *IROp, IR::NodeID Node)
|
||||
|
||||
DEF_OP(LoadContext) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
movzx(GetDst<RA_32>(Node), byte [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 2: {
|
||||
movzx(GetDst<RA_32>(Node), word [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 4: {
|
||||
mov(GetDst<RA_32>(Node), dword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8: {
|
||||
mov(GetDst<RA_64>(Node), qword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 16: {
|
||||
LOGMAN_MSG_A_FMT("Invalid GPR load of size 16");
|
||||
break;
|
||||
}
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else {
|
||||
const auto Dst = GetDst(Node);
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
movzx(rax, byte [STATE + Op->Offset]);
|
||||
vmovq(GetDst(Node), rax);
|
||||
vmovq(Dst, rax);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 2: {
|
||||
movzx(rax, word [STATE + Op->Offset]);
|
||||
vmovq(GetDst(Node), rax);
|
||||
vmovq(Dst, rax);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 4: {
|
||||
vmovd(GetDst(Node), dword [STATE + Op->Offset]);
|
||||
vmovd(Dst, dword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8: {
|
||||
vmovq(GetDst(Node), qword [STATE + Op->Offset]);
|
||||
vmovq(Dst, qword [STATE + Op->Offset]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 16: {
|
||||
if (Op->Offset % 16 == 0)
|
||||
movaps(GetDst(Node), xword [STATE + Op->Offset]);
|
||||
else
|
||||
movups(GetDst(Node), xword [STATE + Op->Offset]);
|
||||
if (Op->Offset % 16 == 0) {
|
||||
vmovaps(Dst, xword [STATE + Op->Offset]);
|
||||
} else {
|
||||
vmovups(Dst, xword [STATE + Op->Offset]);
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
case 32: {
|
||||
if (Op->Offset % 32 == 0) {
|
||||
vmovaps(ToYMM(Dst), yword [STATE + Op->Offset]);
|
||||
} else {
|
||||
vmovups(ToYMM(Dst), yword [STATE + Op->Offset]);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContext size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreContext) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
mov(byte [STATE + Op->Offset], GetSrc<RA_8>(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
break;
|
||||
|
||||
case 2: {
|
||||
mov(word [STATE + Op->Offset], GetSrc<RA_16>(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 4: {
|
||||
mov(dword [STATE + Op->Offset], GetSrc<RA_32>(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8: {
|
||||
mov(qword [STATE + Op->Offset], GetSrc<RA_64>(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 16:
|
||||
LogMan::Msg::DFmt("Invalid store size of 16");
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
|
||||
case 16: {
|
||||
LOGMAN_MSG_A_FMT("Invalid store size of 16");
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else {
|
||||
const auto Value = GetSrc(Op->Value.ID());
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
pextrb(byte [STATE + Op->Offset], GetSrc(Op->Value.ID()), 0);
|
||||
pextrb(byte [STATE + Op->Offset], Value, 0);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
|
||||
case 2: {
|
||||
pextrw(word [STATE + Op->Offset], GetSrc(Op->Value.ID()), 0);
|
||||
pextrw(word [STATE + Op->Offset], Value, 0);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 4: {
|
||||
vmovd(dword [STATE + Op->Offset], GetSrc(Op->Value.ID()));
|
||||
vmovd(dword [STATE + Op->Offset], Value);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8: {
|
||||
vmovq(qword [STATE + Op->Offset], GetSrc(Op->Value.ID()));
|
||||
vmovq(qword [STATE + Op->Offset], Value);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 16: {
|
||||
if (Op->Offset % 16 == 0)
|
||||
movaps(xword [STATE + Op->Offset], GetSrc(Op->Value.ID()));
|
||||
else
|
||||
movups(xword [STATE + Op->Offset], GetSrc(Op->Value.ID()));
|
||||
if (Op->Offset % 16 == 0) {
|
||||
vmovaps(xword [STATE + Op->Offset], Value);
|
||||
} else {
|
||||
vmovups(xword [STATE + Op->Offset], Value);
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
|
||||
case 32: {
|
||||
if (Op->Offset % 32 == 0) {
|
||||
vmovaps(yword [STATE + Op->Offset], ToYMM(Value));
|
||||
} else {
|
||||
vmovups(yword [STATE + Op->Offset], ToYMM(Value));
|
||||
}
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContext size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
size_t size = IROp->Size;
|
||||
Reg index = GetSrc<RA_64>(Op->Index.ID());
|
||||
const auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const Reg Index = GetSrc<RA_64>(Op->Index.ID());
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
switch (Op->Stride) {
|
||||
@@ -153,21 +183,21 @@ DEF_OP(LoadContextIndexed) {
|
||||
case 4:
|
||||
case 8: {
|
||||
lea(rax, dword [STATE + Op->BaseOffset]);
|
||||
switch (size) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
movzx(GetDst<RA_32>(Node), byte [rax + index * Op->Stride]);
|
||||
movzx(GetDst<RA_32>(Node), byte [rax + Index * Op->Stride]);
|
||||
break;
|
||||
case 2:
|
||||
movzx(GetDst<RA_32>(Node), word [rax + index * Op->Stride]);
|
||||
movzx(GetDst<RA_32>(Node), word [rax + Index * Op->Stride]);
|
||||
break;
|
||||
case 4:
|
||||
mov(GetDst<RA_32>(Node), dword [rax + index * Op->Stride]);
|
||||
mov(GetDst<RA_32>(Node), dword [rax + Index * Op->Stride]);
|
||||
break;
|
||||
case 8:
|
||||
mov(GetDst<RA_64>(Node), qword [rax + index * Op->Stride]);
|
||||
mov(GetDst<RA_64>(Node), qword [rax + Index * Op->Stride]);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
@@ -186,53 +216,67 @@ DEF_OP(LoadContextIndexed) {
|
||||
case 2:
|
||||
case 4:
|
||||
case 8: {
|
||||
const auto Dst = GetDst(Node);
|
||||
|
||||
lea(rax, dword [STATE + Op->BaseOffset]);
|
||||
switch (size) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
movzx(eax, byte [rax + index * Op->Stride]);
|
||||
vmovd(GetDst(Node), eax);
|
||||
movzx(eax, byte [rax + Index * Op->Stride]);
|
||||
vmovd(Dst, eax);
|
||||
break;
|
||||
case 2:
|
||||
movzx(eax, word [rax + index * Op->Stride]);
|
||||
vmovd(GetDst(Node), eax);
|
||||
movzx(eax, word [rax + Index * Op->Stride]);
|
||||
vmovd(Dst, eax);
|
||||
break;
|
||||
case 4:
|
||||
vmovd(GetDst(Node), dword [rax + index * Op->Stride]);
|
||||
vmovd(Dst, dword [rax + Index * Op->Stride]);
|
||||
break;
|
||||
case 8:
|
||||
vmovq(GetDst(Node), qword [rax + index * Op->Stride]);
|
||||
vmovq(Dst, qword [rax + Index * Op->Stride]);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 16: {
|
||||
mov(rax, index);
|
||||
shl(rax, 4);
|
||||
case 16:
|
||||
case 32: {
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Shift = Op->Stride == 16 ? 4 : 5;
|
||||
|
||||
mov(rax, Index);
|
||||
shl(rax, Shift);
|
||||
lea(rax, dword [rax + Op->BaseOffset]);
|
||||
switch (size) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
pinsrb(GetDst(Node), byte [STATE + rax], 0);
|
||||
pinsrb(Dst, byte [STATE + rax], 0);
|
||||
break;
|
||||
case 2:
|
||||
pinsrw(GetDst(Node), word [STATE + rax], 0);
|
||||
pinsrw(Dst, word [STATE + rax], 0);
|
||||
break;
|
||||
case 4:
|
||||
vmovd(GetDst(Node), dword [STATE + rax]);
|
||||
vmovd(Dst, dword [STATE + rax]);
|
||||
break;
|
||||
case 8:
|
||||
vmovq(GetDst(Node), qword [STATE + rax]);
|
||||
vmovq(Dst, qword [STATE + rax]);
|
||||
break;
|
||||
case 16:
|
||||
if (Op->BaseOffset % 16 == 0)
|
||||
movaps(GetDst(Node), xword [STATE + rax]);
|
||||
else
|
||||
movups(GetDst(Node), xword [STATE + rax]);
|
||||
if (Op->BaseOffset % 16 == 0) {
|
||||
vmovaps(Dst, xword [STATE + rax]);
|
||||
} else {
|
||||
vmovups(Dst, xword [STATE + rax]);
|
||||
}
|
||||
break;
|
||||
case 32:
|
||||
if (Op->BaseOffset % 32 == 0) {
|
||||
vmovaps(ToYMM(Dst), yword [STATE + rax]);
|
||||
} else {
|
||||
vmovups(ToYMM(Dst), yword [STATE + rax]);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", IROp->Size);
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadContextIndexed size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
@@ -245,12 +289,13 @@ DEF_OP(LoadContextIndexed) {
|
||||
}
|
||||
|
||||
DEF_OP(StoreContextIndexed) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
Reg index = GetSrc<RA_64>(Op->Index.ID());
|
||||
size_t size = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_StoreContextIndexed>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const Reg Index = GetSrc<RA_64>(Op->Index.ID());
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
auto value = GetSrc<RA_64>(Op->Value.ID());
|
||||
const auto Value = GetSrc<RA_64>(Op->Value.ID());
|
||||
lea(rax, dword [STATE + Op->BaseOffset]);
|
||||
|
||||
switch (Op->Stride) {
|
||||
@@ -258,10 +303,10 @@ DEF_OP(StoreContextIndexed) {
|
||||
case 2:
|
||||
case 4:
|
||||
case 8: {
|
||||
if (!(size == 1 || size == 2 || size == 4 || size == 8)) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", IROp->Size);
|
||||
if (!(OpSize == 1 || OpSize == 2 || OpSize == 4 || OpSize == 8)) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", OpSize);
|
||||
}
|
||||
mov(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value);
|
||||
mov(AddressFrame(OpSize * 8) [rax + Index * Op->Stride], Value);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
@@ -270,57 +315,68 @@ DEF_OP(StoreContextIndexed) {
|
||||
}
|
||||
}
|
||||
else {
|
||||
auto value = GetSrc(Op->Value.ID());
|
||||
const auto Value = GetSrc(Op->Value.ID());
|
||||
switch (Op->Stride) {
|
||||
case 1:
|
||||
case 2:
|
||||
case 4:
|
||||
case 8: {
|
||||
lea(rax, dword [STATE + Op->BaseOffset]);
|
||||
switch (size) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
pextrb(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value, 0);
|
||||
pextrb(AddressFrame(OpSize * 8) [rax + Index * Op->Stride], Value, 0);
|
||||
break;
|
||||
case 2:
|
||||
pextrw(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value, 0);
|
||||
pextrw(AddressFrame(OpSize * 8) [rax + Index * Op->Stride], Value, 0);
|
||||
break;
|
||||
case 4:
|
||||
vmovd(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value);
|
||||
vmovd(AddressFrame(OpSize * 8) [rax + Index * Op->Stride], Value);
|
||||
break;
|
||||
case 8:
|
||||
vmovq(AddressFrame(IROp->Size * 8) [rax + index * Op->Stride], value);
|
||||
vmovq(AddressFrame(OpSize * 8) [rax + Index * Op->Stride], Value);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", size);
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 16: {
|
||||
mov(rax, index);
|
||||
shl(rax, 4);
|
||||
case 16:
|
||||
case 32: {
|
||||
const auto Shift = Op->Stride == 16 ? 4 : 5;
|
||||
|
||||
mov(rax, Index);
|
||||
shl(rax, Shift);
|
||||
lea(rax, dword [rax + Op->BaseOffset]);
|
||||
switch (size) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
pextrb(AddressFrame(IROp->Size * 8) [STATE + rax], value, 0);
|
||||
pextrb(AddressFrame(OpSize * 8) [STATE + rax], Value, 0);
|
||||
break;
|
||||
case 2:
|
||||
pextrw(AddressFrame(IROp->Size * 8) [STATE + rax], value, 0);
|
||||
pextrw(AddressFrame(OpSize * 8) [STATE + rax], Value, 0);
|
||||
break;
|
||||
case 4:
|
||||
vmovd(AddressFrame(IROp->Size * 8) [STATE + rax], value);
|
||||
vmovd(AddressFrame(OpSize * 8) [STATE + rax], Value);
|
||||
break;
|
||||
case 8:
|
||||
vmovq(AddressFrame(IROp->Size * 8) [STATE + rax], value);
|
||||
vmovq(AddressFrame(OpSize * 8) [STATE + rax], Value);
|
||||
break;
|
||||
case 16:
|
||||
if (Op->BaseOffset % 16 == 0)
|
||||
movaps(xword [STATE + rax], value);
|
||||
else
|
||||
movups(xword [STATE + rax], value);
|
||||
if (Op->BaseOffset % 16 == 0) {
|
||||
vmovaps(xword [STATE + rax], Value);
|
||||
} else {
|
||||
vmovups(xword [STATE + rax], Value);
|
||||
}
|
||||
break;
|
||||
case 32:
|
||||
if (Op->BaseOffset % 32 == 0) {
|
||||
vmovaps(yword [STATE + rax], ToYMM(Value));
|
||||
} else {
|
||||
vmovups(yword [STATE + rax], ToYMM(Value));
|
||||
}
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", size);
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreContextIndexed size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
@@ -333,10 +389,10 @@ DEF_OP(StoreContextIndexed) {
|
||||
}
|
||||
|
||||
DEF_OP(SpillRegister) {
|
||||
auto Op = IROp->C<IR::IROp_SpillRegister>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_SpillRegister>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const uint32_t SlotOffset = Op->Slot * MaxSpillSlotSize;
|
||||
|
||||
uint32_t SlotOffset = Op->Slot * 16;
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
@@ -355,36 +411,44 @@ DEF_OP(SpillRegister) {
|
||||
mov(qword [rsp + SlotOffset], GetSrc<RA_64>(Op->Value.ID()));
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize);
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::FPRClass) {
|
||||
const auto Src = GetSrc(Op->Value.ID());
|
||||
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
movss(dword [rsp + SlotOffset], GetSrc(Op->Value.ID()));
|
||||
movss(dword [rsp + SlotOffset], Src);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
movsd(qword [rsp + SlotOffset], GetSrc(Op->Value.ID()));
|
||||
movsd(qword [rsp + SlotOffset], Src);
|
||||
break;
|
||||
}
|
||||
case 16: {
|
||||
movaps(xword [rsp + SlotOffset], GetSrc(Op->Value.ID()));
|
||||
movaps(xword [rsp + SlotOffset], Src);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize);
|
||||
case 32: {
|
||||
vmovaps(yword [rsp + SlotOffset], ToYMM(Src));
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unhandled SpillRegister class: {}", Op->Class.Val);
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
|
||||
DEF_OP(FillRegister) {
|
||||
auto Op = IROp->C<IR::IROp_FillRegister>();
|
||||
uint8_t OpSize = IROp->Size;
|
||||
const auto Op = IROp->C<IR::IROp_FillRegister>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const uint32_t SlotOffset = Op->Slot * MaxSpillSlotSize;
|
||||
|
||||
uint32_t SlotOffset = Op->Slot * 16;
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
@@ -403,23 +467,33 @@ DEF_OP(FillRegister) {
|
||||
mov(GetDst<RA_64>(Node), qword [rsp + SlotOffset]);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize);
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::FPRClass) {
|
||||
const auto Dst = GetDst(Node);
|
||||
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
movss(GetDst(Node), dword [rsp + SlotOffset]);
|
||||
vmovss(Dst, dword [rsp + SlotOffset]);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
movsd(GetDst(Node), qword [rsp + SlotOffset]);
|
||||
vmovsd(Dst, qword [rsp + SlotOffset]);
|
||||
break;
|
||||
}
|
||||
case 16: {
|
||||
movaps(GetDst(Node), xword [rsp + SlotOffset]);
|
||||
vmovaps(Dst, xword [rsp + SlotOffset]);
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize);
|
||||
case 32: {
|
||||
vmovaps(ToYMM(Dst), yword [rsp + SlotOffset]);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unhandled FillRegister class: {}", Op->Class.Val);
|
||||
@@ -464,118 +538,132 @@ Xbyak::RegExp X86JITCore::GenerateModRM(Xbyak::Reg Base, IR::OrderedNodeWrapper
|
||||
}
|
||||
|
||||
DEF_OP(LoadMem) {
|
||||
auto Op = IROp->C<IR::IROp_LoadMem>();
|
||||
const auto Op = IROp->C<IR::IROp_LoadMem>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
|
||||
auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
const Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
const auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
auto Dst = GetDst<RA_64>(Node);
|
||||
const auto Dst = GetDst<RA_64>(Node);
|
||||
|
||||
switch (IROp->Size) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
movzx (Dst, byte [MemPtr]);
|
||||
movzx(Dst, byte [MemPtr]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 2: {
|
||||
movzx (Dst, word [MemPtr]);
|
||||
movzx(Dst, word [MemPtr]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 4: {
|
||||
mov(Dst.cvt32(), dword [MemPtr]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8: {
|
||||
mov(Dst, qword [MemPtr]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", IROp->Size);
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
auto Dst = GetDst(Node);
|
||||
const auto Dst = GetDst(Node);
|
||||
|
||||
switch (IROp->Size) {
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
movzx(eax, byte [MemPtr]);
|
||||
vmovd(Dst, eax);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 2: {
|
||||
movzx(eax, word [MemPtr]);
|
||||
vmovd(Dst, eax);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 4: {
|
||||
vmovd(Dst, dword [MemPtr]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 8: {
|
||||
vmovq(Dst, qword [MemPtr]);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case 16: {
|
||||
if (IROp->Size == Op->Align)
|
||||
movups(GetDst(Node), xword [MemPtr]);
|
||||
else
|
||||
movups(GetDst(Node), xword [MemPtr]);
|
||||
if (MemoryDebug) {
|
||||
movq(rcx, GetDst(Node));
|
||||
}
|
||||
}
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", IROp->Size);
|
||||
vmovups(Dst, xword [MemPtr]);
|
||||
if (MemoryDebug) {
|
||||
movq(rcx, Dst);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 32: {
|
||||
vmovups(ToYMM(Dst), yword [MemPtr]);
|
||||
if (MemoryDebug) {
|
||||
movq(rcx, Dst);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadMem size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreMem) {
|
||||
auto Op = IROp->C<IR::IROp_StoreMem>();
|
||||
const auto Op = IROp->C<IR::IROp_StoreMem>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
|
||||
auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
const Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
|
||||
const auto MemPtr = GenerateModRM(MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
switch (IROp->Size) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
mov(byte [MemPtr], GetSrc<RA_8>(Op->Value.ID()));
|
||||
break;
|
||||
break;
|
||||
case 2:
|
||||
mov(word [MemPtr], GetSrc<RA_16>(Op->Value.ID()));
|
||||
break;
|
||||
break;
|
||||
case 4:
|
||||
mov(dword [MemPtr], GetSrc<RA_32>(Op->Value.ID()));
|
||||
break;
|
||||
break;
|
||||
case 8:
|
||||
mov(qword [MemPtr], GetSrc<RA_64>(Op->Value.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", IROp->Size);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else {
|
||||
switch (IROp->Size) {
|
||||
const auto Value = GetSrc(Op->Value.ID());
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
pextrb(byte [MemPtr], GetSrc(Op->Value.ID()), 0);
|
||||
break;
|
||||
pextrb(byte [MemPtr], Value, 0);
|
||||
break;
|
||||
case 2:
|
||||
pextrw(word [MemPtr], GetSrc(Op->Value.ID()), 0);
|
||||
break;
|
||||
pextrw(word [MemPtr], Value, 0);
|
||||
break;
|
||||
case 4:
|
||||
vmovd(dword [MemPtr], GetSrc(Op->Value.ID()));
|
||||
break;
|
||||
vmovd(dword [MemPtr], Value);
|
||||
break;
|
||||
case 8:
|
||||
vmovq(qword [MemPtr], GetSrc(Op->Value.ID()));
|
||||
break;
|
||||
vmovq(qword [MemPtr], Value);
|
||||
break;
|
||||
case 16:
|
||||
if (IROp->Size == Op->Align)
|
||||
movups(xword [MemPtr], GetSrc(Op->Value.ID()));
|
||||
else
|
||||
movups(xword [MemPtr], GetSrc(Op->Value.ID()));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", IROp->Size);
|
||||
vmovups(xword [MemPtr], Value);
|
||||
break;
|
||||
case 32:
|
||||
vmovups(yword [MemPtr], ToYMM(Value));
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreMem size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -47,7 +47,7 @@ DEF_OP(Break) {
|
||||
auto Op = IROp->C<IR::IROp_Break>();
|
||||
|
||||
if (SpillSlots) {
|
||||
add(rsp, SpillSlots * 16);
|
||||
add(rsp, SpillSlots * MaxSpillSlotSize);
|
||||
}
|
||||
|
||||
mov(byte [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.FaultToTopAndGeneratedException)], 1);
|
||||
|
||||
@@ -73,17 +73,11 @@ DEF_OP(CreateElementPair) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Mov) {
|
||||
auto Op = IROp->C<IR::IROp_Mov>();
|
||||
mov (GetDst<RA_64>(Node), GetSrc<RA_64>(Op->Value.ID()));
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void X86JITCore::RegisterMoveHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
|
||||
REGISTER_OP(EXTRACTELEMENTPAIR, ExtractElementPair);
|
||||
REGISTER_OP(CREATEELEMENTPAIR, CreateElementPair);
|
||||
REGISTER_OP(MOV, Mov);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
|
||||
+2281
-969
File diff suppressed because it is too large.
Load diff
+359
-60
@@ -126,10 +126,20 @@ void OpDispatchBuilder::ThunkOp(OpcodeArgs) {
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
uint8_t *sha256 = (uint8_t *)(Op->PC + 2);
|
||||
|
||||
_Thunk(
|
||||
_LoadContext(GPRSize, GPRClass, GPROffset(X86State::REG_RDI)),
|
||||
*reinterpret_cast<SHA256Sum*>(sha256)
|
||||
);
|
||||
if (CTX->Config.Is64BitMode) {
|
||||
// x86-64 ABI puts the function argument in RDI
|
||||
_Thunk(
|
||||
_LoadContext(GPRSize, GPRClass, GPROffset(X86State::REG_RDI)),
|
||||
*reinterpret_cast<SHA256Sum*>(sha256)
|
||||
);
|
||||
}
|
||||
else {
|
||||
// x86 fastcall ABI puts the function argument in ECX
|
||||
_Thunk(
|
||||
_LoadContext(GPRSize, GPRClass, GPROffset(X86State::REG_RCX)),
|
||||
*reinterpret_cast<SHA256Sum*>(sha256)
|
||||
);
|
||||
}
|
||||
|
||||
auto Constant = _Constant(GPRSize);
|
||||
auto OldSP = _LoadContext(GPRSize, GPRClass, RSPOffset);
|
||||
@@ -230,8 +240,11 @@ void OpDispatchBuilder::IRETOp(OpcodeArgs) {
|
||||
// RIP (64/32/16 bits)
|
||||
auto NewRIP = _LoadMem(GPRClass, GPRSize, SP, GPRSize);
|
||||
SP = _Add(SP, Constant);
|
||||
//CS (lower 16 used)
|
||||
_StoreContext(2, GPRClass, _LoadMem(GPRClass, GPRSize, SP, GPRSize), offsetof(FEXCore::Core::CPUState, cs));
|
||||
// CS (lower 16 used)
|
||||
auto NewSegmentCS = _LoadMem(GPRClass, GPRSize, SP, GPRSize);
|
||||
_StoreContext(2, GPRClass, NewSegmentCS, offsetof(FEXCore::Core::CPUState, cs_idx));
|
||||
UpdatePrefixFromSegment(NewSegmentCS, FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX);
|
||||
|
||||
SP = _Add(SP, Constant);
|
||||
//eflags (lower 16 used)
|
||||
auto eflags = _LoadMem(GPRClass, GPRSize, SP, GPRSize);
|
||||
@@ -243,8 +256,11 @@ void OpDispatchBuilder::IRETOp(OpcodeArgs) {
|
||||
// FEX doesn't support a CPL mode switch, so don't need to worry about this on 32-bit
|
||||
_StoreContext(GPRSize, GPRClass, _LoadMem(GPRClass, GPRSize, SP, GPRSize), RSPOffset);
|
||||
SP = _Add(SP, Constant);
|
||||
//ss
|
||||
_StoreContext(2, GPRClass, _LoadMem(GPRClass, GPRSize, SP, GPRSize), offsetof(FEXCore::Core::CPUState, ss));
|
||||
// ss
|
||||
auto NewSegmentSS = _LoadMem(GPRClass, GPRSize, SP, GPRSize);
|
||||
_StoreContext(2, GPRClass, NewSegmentSS, offsetof(FEXCore::Core::CPUState, ss_idx));
|
||||
UpdatePrefixFromSegment(NewSegmentSS, FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX);
|
||||
|
||||
SP = _Add(SP, Constant);
|
||||
}
|
||||
else {
|
||||
@@ -562,26 +578,51 @@ void OpDispatchBuilder::PUSHSegmentOp(OpcodeArgs) {
|
||||
_StoreContext(GPRSize, GPRClass, NewSP, RSPOffset);
|
||||
|
||||
OrderedNode *Src{};
|
||||
switch (SegmentReg) {
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX:
|
||||
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, es));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX:
|
||||
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, cs));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX:
|
||||
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, ss));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX:
|
||||
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, ds));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX:
|
||||
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, fs));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX:
|
||||
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, gs));
|
||||
break;
|
||||
default: break; // Do nothing
|
||||
if (!CTX->Config.Is64BitMode()) {
|
||||
switch (SegmentReg) {
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX:
|
||||
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, es_idx));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX:
|
||||
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, cs_idx));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX:
|
||||
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, ss_idx));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX:
|
||||
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, ds_idx));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX:
|
||||
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, fs_idx));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX:
|
||||
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, gs_idx));
|
||||
break;
|
||||
default: break; // Do nothing
|
||||
}
|
||||
}
|
||||
else {
|
||||
switch (SegmentReg) {
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX:
|
||||
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, es_cached));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX:
|
||||
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, cs_cached));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX:
|
||||
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, ss_cached));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX:
|
||||
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, ds_cached));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX:
|
||||
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, fs_cached));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX:
|
||||
Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, gs_cached));
|
||||
break;
|
||||
default: break; // Do nothing
|
||||
}
|
||||
}
|
||||
|
||||
// Store our value to the new stack location
|
||||
@@ -679,25 +720,27 @@ void OpDispatchBuilder::POPSegmentOp(OpcodeArgs) {
|
||||
|
||||
switch (SegmentReg) {
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX:
|
||||
_StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, es));
|
||||
_StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, es_idx));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX:
|
||||
_StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, cs));
|
||||
_StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, cs_idx));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX:
|
||||
_StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, ss));
|
||||
_StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, ss_idx));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX:
|
||||
_StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, ds));
|
||||
_StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, ds_idx));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX:
|
||||
_StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, fs));
|
||||
_StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, fs_idx));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX:
|
||||
_StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, gs));
|
||||
_StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, gs_idx));
|
||||
break;
|
||||
default: break; // Do nothing
|
||||
}
|
||||
|
||||
UpdatePrefixFromSegment(NewSegment, SegmentReg);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::LEAVEOp(OpcodeArgs) {
|
||||
@@ -1581,11 +1624,13 @@ void OpDispatchBuilder::MOVSegOp(OpcodeArgs) {
|
||||
switch (Op->Dest.Data.GPR.GPR) {
|
||||
case 0: // ES
|
||||
case FEXCore::X86State::REG_R8: // ES
|
||||
_StoreContext(2, GPRClass, Src, offsetof(FEXCore::Core::CPUState, es));
|
||||
_StoreContext(2, GPRClass, Src, offsetof(FEXCore::Core::CPUState, es_idx));
|
||||
UpdatePrefixFromSegment(Src, FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX);
|
||||
break;
|
||||
case 1: // DS
|
||||
case FEXCore::X86State::REG_R11: // DS
|
||||
_StoreContext(2, GPRClass, Src, offsetof(FEXCore::Core::CPUState, ds));
|
||||
_StoreContext(2, GPRClass, Src, offsetof(FEXCore::Core::CPUState, ds_idx));
|
||||
UpdatePrefixFromSegment(Src, FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX);
|
||||
break;
|
||||
case 2: // CS
|
||||
case FEXCore::X86State::REG_R9: // CS
|
||||
@@ -1599,12 +1644,14 @@ void OpDispatchBuilder::MOVSegOp(OpcodeArgs) {
|
||||
break;
|
||||
case 3: // SS
|
||||
case FEXCore::X86State::REG_R10: // SS
|
||||
_StoreContext(2, GPRClass, Src, offsetof(FEXCore::Core::CPUState, ss));
|
||||
_StoreContext(2, GPRClass, Src, offsetof(FEXCore::Core::CPUState, ss_idx));
|
||||
UpdatePrefixFromSegment(Src, FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX);
|
||||
break;
|
||||
case 6: // GS
|
||||
case FEXCore::X86State::REG_R13: // GS
|
||||
if (!CTX->Config.Is64BitMode) {
|
||||
_StoreContext(2, GPRClass, Src, offsetof(FEXCore::Core::CPUState, gs));
|
||||
_StoreContext(2, GPRClass, Src, offsetof(FEXCore::Core::CPUState, gs_idx));
|
||||
UpdatePrefixFromSegment(Src, FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX);
|
||||
} else {
|
||||
LogMan::Msg::EFmt("We don't support modifying GS selector in 64bit mode!");
|
||||
DecodeFailure = true;
|
||||
@@ -1613,7 +1660,8 @@ void OpDispatchBuilder::MOVSegOp(OpcodeArgs) {
|
||||
case 7: // FS
|
||||
case FEXCore::X86State::REG_R12: // FS
|
||||
if (!CTX->Config.Is64BitMode) {
|
||||
_StoreContext(2, GPRClass, Src, offsetof(FEXCore::Core::CPUState, fs));
|
||||
_StoreContext(2, GPRClass, Src, offsetof(FEXCore::Core::CPUState, fs_idx));
|
||||
UpdatePrefixFromSegment(Src, FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX);
|
||||
} else {
|
||||
LogMan::Msg::EFmt("We don't support modifying FS selector in 64bit mode!");
|
||||
DecodeFailure = true;
|
||||
@@ -1631,19 +1679,19 @@ void OpDispatchBuilder::MOVSegOp(OpcodeArgs) {
|
||||
switch (Op->Src[0].Data.GPR.GPR) {
|
||||
case 0: // ES
|
||||
case FEXCore::X86State::REG_R8: // ES
|
||||
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, es));
|
||||
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, es_idx));
|
||||
break;
|
||||
case 1: // DS
|
||||
case FEXCore::X86State::REG_R11: // DS
|
||||
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, ds));
|
||||
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, ds_idx));
|
||||
break;
|
||||
case 2: // CS
|
||||
case FEXCore::X86State::REG_R9: // CS
|
||||
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, cs));
|
||||
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, cs_idx));
|
||||
break;
|
||||
case 3: // SS
|
||||
case FEXCore::X86State::REG_R10: // SS
|
||||
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, ss));
|
||||
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, ss_idx));
|
||||
break;
|
||||
case 6: // GS
|
||||
case FEXCore::X86State::REG_R13: // GS
|
||||
@@ -1651,7 +1699,7 @@ void OpDispatchBuilder::MOVSegOp(OpcodeArgs) {
|
||||
Segment = _Constant(0);
|
||||
}
|
||||
else {
|
||||
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, gs));
|
||||
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, gs_idx));
|
||||
}
|
||||
break;
|
||||
case 7: // FS
|
||||
@@ -1660,7 +1708,7 @@ void OpDispatchBuilder::MOVSegOp(OpcodeArgs) {
|
||||
Segment = _Constant(0);
|
||||
}
|
||||
else {
|
||||
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, fs));
|
||||
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, fs_idx));
|
||||
}
|
||||
break;
|
||||
default:
|
||||
@@ -3332,6 +3380,222 @@ void OpDispatchBuilder::PopcountOp(OpcodeArgs) {
|
||||
GenerateFlags_POPCOUNT(Op, Src);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::DAAOp(OpcodeArgs) {
|
||||
CalculateDeferredFlags();
|
||||
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC);
|
||||
auto AF = GetRFLAG(FEXCore::X86State::RFLAG_AF_LOC);
|
||||
auto AL = _LoadContext(1, GPRClass, GPROffset(X86State::REG_RAX));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(0));
|
||||
|
||||
auto Cond = _Or(AF, _Select(FEXCore::IR::COND_UGT, _And(AL, _Constant(0xF)), _Constant(9), _Constant(1), _Constant(0)));
|
||||
auto FalseBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
auto TrueBlock = CreateNewCodeBlockAfter(FalseBlock);
|
||||
auto EndBlock = CreateNewCodeBlockAfter(TrueBlock);
|
||||
|
||||
_CondJump(Cond, TrueBlock, FalseBlock);
|
||||
SetCurrentCodeBlock(FalseBlock);
|
||||
{
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(0));
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(TrueBlock);
|
||||
{
|
||||
auto NewAL = _Add(AL, _Constant(0x6));
|
||||
_StoreContext(1, GPRClass, NewAL, GPROffset(X86State::REG_RAX));
|
||||
CalculateDeferredFlags();
|
||||
auto NewCF = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Or(CF, NewCF));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(1));
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(EndBlock);
|
||||
|
||||
Cond = _Or(CF, _Select(FEXCore::IR::COND_UGT, AL, _Constant(0x99), _Constant(1), _Constant(0)));
|
||||
FalseBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
TrueBlock = CreateNewCodeBlockAfter(FalseBlock);
|
||||
EndBlock = CreateNewCodeBlockAfter(TrueBlock);
|
||||
_CondJump(Cond, TrueBlock, FalseBlock);
|
||||
SetCurrentCodeBlock(FalseBlock);
|
||||
{
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(0));
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(TrueBlock);
|
||||
{
|
||||
AL = _LoadContext(1, GPRClass, GPROffset(X86State::REG_RAX));
|
||||
auto NewAL = _Add(AL, _Constant(0x60));
|
||||
_StoreContext(1, GPRClass, NewAL, GPROffset(X86State::REG_RAX));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(1));
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(EndBlock);
|
||||
// Update Flags
|
||||
AL = _LoadContext(1, GPRClass, GPROffset(X86State::REG_RAX));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(_Select(FEXCore::IR::COND_UGE, _And(AL, _Constant(0x80)), _Constant(0), _Constant(1), _Constant(0)));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(_Select(FEXCore::IR::COND_EQ, _And(AL, _Constant(0xFF)), _Constant(0), _Constant(1), _Constant(0)));
|
||||
auto EightBitMask = _Constant(0xFF);
|
||||
auto PopCountOp = _Popcount(_And(AL, EightBitMask));
|
||||
auto XorOp = _Xor(PopCountOp, _Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(XorOp);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::DASOp(OpcodeArgs) {
|
||||
CalculateDeferredFlags();
|
||||
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC);
|
||||
auto AF = GetRFLAG(FEXCore::X86State::RFLAG_AF_LOC);
|
||||
auto AL = _LoadContext(1, GPRClass, GPROffset(X86State::REG_RAX));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(0));
|
||||
|
||||
auto Cond = _Or(AF, _Select(FEXCore::IR::COND_UGT, _And(AL, _Constant(0xf)), _Constant(9), _Constant(1), _Constant(0)));
|
||||
auto FalseBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
auto TrueBlock = CreateNewCodeBlockAfter(FalseBlock);
|
||||
auto EndBlock = CreateNewCodeBlockAfter(TrueBlock);
|
||||
|
||||
_CondJump(Cond, TrueBlock, FalseBlock);
|
||||
SetCurrentCodeBlock(FalseBlock);
|
||||
{
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(0));
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(TrueBlock);
|
||||
{
|
||||
auto NewAL = _Sub(AL, _Constant(0x6));
|
||||
_StoreContext(1, GPRClass, NewAL, GPROffset(X86State::REG_RAX));
|
||||
CalculateDeferredFlags();
|
||||
auto NewCF = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Or(CF, NewCF));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(1));
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(EndBlock);
|
||||
|
||||
Cond = _Or(CF, _Select(FEXCore::IR::COND_UGT, AL, _Constant(0x99), _Constant(1), _Constant(0)));
|
||||
FalseBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
TrueBlock = CreateNewCodeBlockAfter(FalseBlock);
|
||||
EndBlock = CreateNewCodeBlockAfter(TrueBlock);
|
||||
_CondJump(Cond, TrueBlock, FalseBlock);
|
||||
SetCurrentCodeBlock(FalseBlock);
|
||||
{
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(0));
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(TrueBlock);
|
||||
{
|
||||
AL = _LoadContext(1, GPRClass, GPROffset(X86State::REG_RAX));
|
||||
auto NewAL = _Sub(AL, _Constant(0x60));
|
||||
_StoreContext(1, GPRClass, NewAL, GPROffset(X86State::REG_RAX));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(1));
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(EndBlock);
|
||||
// Update Flags
|
||||
AL = _LoadContext(1, GPRClass, GPROffset(X86State::REG_RAX));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(_Select(FEXCore::IR::COND_UGE, _And(AL, _Constant(0x80)), _Constant(0), _Constant(1), _Constant(0)));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(_Select(FEXCore::IR::COND_EQ, _And(AL, _Constant(0xFF)), _Constant(0), _Constant(1), _Constant(0)));
|
||||
auto EightBitMask = _Constant(0xFF);
|
||||
auto PopCountOp = _Popcount(_And(AL, EightBitMask));
|
||||
auto XorOp = _Xor(PopCountOp, _Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(XorOp);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AAAOp(OpcodeArgs) {
|
||||
auto AF = GetRFLAG(FEXCore::X86State::RFLAG_AF_LOC);
|
||||
auto AL = _LoadContext(1, GPRClass, GPROffset(X86State::REG_RAX));
|
||||
auto AX = _LoadContext(2, GPRClass, GPROffset(X86State::REG_RAX));
|
||||
auto Cond = _Or(AF, _Select(FEXCore::IR::COND_UGT, _And(AL, _Constant(0xF)), _Constant(9), _Constant(1), _Constant(0)));
|
||||
|
||||
auto FalseBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
auto TrueBlock = CreateNewCodeBlockAfter(FalseBlock);
|
||||
auto EndBlock = CreateNewCodeBlockAfter(TrueBlock);
|
||||
_CondJump(Cond, TrueBlock, FalseBlock);
|
||||
|
||||
SetCurrentCodeBlock(FalseBlock);
|
||||
{
|
||||
auto NewAX = _And(AX, _Constant(0xFF0F));
|
||||
_StoreContext(2, GPRClass, NewAX, GPROffset(X86State::REG_RAX));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(0));
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(TrueBlock);
|
||||
{
|
||||
auto NewAX = _Add(AX, _Constant(0x106));
|
||||
auto Result = _And(NewAX, _Constant(0xFF0F));
|
||||
_StoreContext(2, GPRClass, Result, GPROffset(X86State::REG_RAX));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(1));
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(EndBlock);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AASOp(OpcodeArgs) {
|
||||
auto AF = GetRFLAG(FEXCore::X86State::RFLAG_AF_LOC);
|
||||
auto AL = _LoadContext(1, GPRClass, GPROffset(X86State::REG_RAX));
|
||||
auto AX = _LoadContext(2, GPRClass, GPROffset(X86State::REG_RAX));
|
||||
auto Cond = _Or(AF, _Select(FEXCore::IR::COND_UGT, _And(AL, _Constant(0xF)), _Constant(9), _Constant(1), _Constant(0)));
|
||||
|
||||
auto FalseBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
auto TrueBlock = CreateNewCodeBlockAfter(FalseBlock);
|
||||
auto EndBlock = CreateNewCodeBlockAfter(TrueBlock);
|
||||
_CondJump(Cond, TrueBlock, FalseBlock);
|
||||
|
||||
SetCurrentCodeBlock(FalseBlock);
|
||||
{
|
||||
auto NewAX = _And(AX, _Constant(0xFF0F));
|
||||
_StoreContext(2, GPRClass, NewAX, GPROffset(X86State::REG_RAX));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(0));
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(TrueBlock);
|
||||
{
|
||||
auto NewAX = _Sub(AX, _Constant(6));
|
||||
NewAX = _Sub(NewAX, _Constant(0x100));
|
||||
auto Result = _And(NewAX, _Constant(0xFF0F));
|
||||
_StoreContext(2, GPRClass, Result, GPROffset(X86State::REG_RAX));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(_Constant(1));
|
||||
_Jump(EndBlock);
|
||||
}
|
||||
SetCurrentCodeBlock(EndBlock);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AAMOp(OpcodeArgs) {
|
||||
auto AL = _LoadContext(1, GPRClass, GPROffset(X86State::REG_RAX));
|
||||
auto Imm8 = _Constant(Op->Src[0].Data.Literal.Value & 0xFF);
|
||||
auto UDivOp = _UDiv(AL, Imm8);
|
||||
auto URemOp = _URem(AL, Imm8);
|
||||
auto AH = _Lshl(UDivOp, _Constant(8));
|
||||
auto AX = _Add(AH, URemOp);
|
||||
_StoreContext(2, GPRClass, AX, GPROffset(X86State::REG_RAX));
|
||||
// Update Flags
|
||||
AL = _LoadContext(1, GPRClass, GPROffset(X86State::REG_RAX));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(_Select(FEXCore::IR::COND_UGE, _And(AL, _Constant(0x80)), _Constant(0), _Constant(1), _Constant(0)));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(_Select(FEXCore::IR::COND_EQ, _And(AL, _Constant(0xFF)), _Constant(0), _Constant(1), _Constant(0)));
|
||||
auto EightBitMask = _Constant(0xFF);
|
||||
auto PopCountOp = _Popcount(_And(AL, EightBitMask));
|
||||
auto XorOp = _Xor(PopCountOp, _Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(XorOp);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AADOp(OpcodeArgs) {
|
||||
auto AL = _LoadContext(1, GPRClass, GPROffset(X86State::REG_RAX));
|
||||
auto AH = _Lshr(_LoadContext(2, GPRClass, GPROffset(X86State::REG_RAX)), _Constant(8));
|
||||
auto Imm8 = _Constant(Op->Src[0].Data.Literal.Value & 0xFF);
|
||||
auto NewAL = _Add(AL, _Mul(AH, Imm8));
|
||||
auto Result = _And(NewAL, _Constant(0xFF));
|
||||
_StoreContext(2, GPRClass, Result, GPROffset(X86State::REG_RAX));
|
||||
// Update Flags
|
||||
AL = _LoadContext(1, GPRClass, GPROffset(X86State::REG_RAX));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(_Select(FEXCore::IR::COND_UGE, _And(AL, _Constant(0x80)), _Constant(0), _Constant(1), _Constant(0)));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(_Select(FEXCore::IR::COND_EQ, _And(AL, _Constant(0xFF)), _Constant(0), _Constant(1), _Constant(0)));
|
||||
auto EightBitMask = _Constant(0xFF);
|
||||
auto PopCountOp = _Popcount(_And(AL, EightBitMask));
|
||||
auto XorOp = _Xor(PopCountOp, _Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(XorOp);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::XLATOp(OpcodeArgs) {
|
||||
const uint32_t RAXOffset = GPROffset(X86State::REG_RAX);
|
||||
const uint32_t RBXOffset = GPROffset(X86State::REG_RBX);
|
||||
@@ -3350,13 +3614,15 @@ void OpDispatchBuilder::XLATOp(OpcodeArgs) {
|
||||
|
||||
template<OpDispatchBuilder::Segment Seg>
|
||||
void OpDispatchBuilder::ReadSegmentReg(OpcodeArgs) {
|
||||
// 64-bit only
|
||||
// Doesn't hit the segment register optimization
|
||||
auto Size = GetSrcSize(Op);
|
||||
OrderedNode *Src{};
|
||||
if constexpr (Seg == Segment::FS) {
|
||||
Src = _LoadContext(Size, GPRClass, offsetof(FEXCore::Core::CPUState, fs));
|
||||
Src = _LoadContext(Size, GPRClass, offsetof(FEXCore::Core::CPUState, fs_cached));
|
||||
}
|
||||
else {
|
||||
Src = _LoadContext(Size, GPRClass, offsetof(FEXCore::Core::CPUState, gs));
|
||||
Src = _LoadContext(Size, GPRClass, offsetof(FEXCore::Core::CPUState, gs_cached));
|
||||
}
|
||||
|
||||
StoreResult(GPRClass, Op, Src, -1);
|
||||
@@ -3369,10 +3635,10 @@ void OpDispatchBuilder::WriteSegmentReg(OpcodeArgs) {
|
||||
auto Size = GetDstSize(Op);
|
||||
OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
if constexpr (Seg == Segment::FS) {
|
||||
_StoreContext(Size, GPRClass, Src, offsetof(FEXCore::Core::CPUState, fs));
|
||||
_StoreContext(Size, GPRClass, Src, offsetof(FEXCore::Core::CPUState, fs_cached));
|
||||
}
|
||||
else {
|
||||
_StoreContext(Size, GPRClass, Src, offsetof(FEXCore::Core::CPUState, gs));
|
||||
_StoreContext(Size, GPRClass, Src, offsetof(FEXCore::Core::CPUState, gs_cached));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4570,10 +4836,10 @@ OrderedNode *OpDispatchBuilder::AppendSegmentOffset(OrderedNode *Value, uint32_t
|
||||
|
||||
if (CTX->Config.Is64BitMode) {
|
||||
if (Flags & FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX) {
|
||||
Value = _Add(Value, _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, fs)));
|
||||
Value = _Add(Value, _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, fs_cached)));
|
||||
}
|
||||
else if (Flags & FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX) {
|
||||
Value = _Add(Value, _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, gs)));
|
||||
Value = _Add(Value, _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, gs_cached)));
|
||||
}
|
||||
// If there was any other segment in 64bit then it is ignored
|
||||
}
|
||||
@@ -4585,38 +4851,65 @@ OrderedNode *OpDispatchBuilder::AppendSegmentOffset(OrderedNode *Value, uint32_t
|
||||
// Or the argument only uses a specific prefix (with override set)
|
||||
Prefix = DefaultPrefix;
|
||||
}
|
||||
// With the segment register optimization we store the GDT bases directly in the segment register to remove indexed loads
|
||||
switch (Prefix) {
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX:
|
||||
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, es));
|
||||
Segment = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, es_cached));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX:
|
||||
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, cs));
|
||||
Segment = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, cs_cached));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX:
|
||||
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, ss));
|
||||
Segment = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, ss_cached));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX:
|
||||
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, ds));
|
||||
Segment = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, ds_cached));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX:
|
||||
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, fs));
|
||||
Segment = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, fs_cached));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX:
|
||||
Segment = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, gs));
|
||||
Segment = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, gs_cached));
|
||||
break;
|
||||
default: break; // Do nothing
|
||||
}
|
||||
|
||||
if (Segment) {
|
||||
Segment = _Lshr(Segment, _Constant(3));
|
||||
auto data = _LoadContextIndexed(Segment, 4, offsetof(FEXCore::Core::CPUState, gdt[0]), 4, GPRClass);
|
||||
Value = _Add(Value, data);
|
||||
Value = _Add(Value, Segment);
|
||||
}
|
||||
}
|
||||
|
||||
return Value;
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::UpdatePrefixFromSegment(OrderedNode *Segment, uint32_t SegmentReg) {
|
||||
// Use BFE to extract the selector index in bits [15,3] of the segment register.
|
||||
// In some cases the upper 16-bits of the 32-bit GPR contain garbage to ignore.
|
||||
Segment = _Bfe(4, 16 - 3, 3, Segment);
|
||||
auto NewSegment = _LoadContextIndexed(Segment, 4, offsetof(FEXCore::Core::CPUState, gdt[0]), 4, GPRClass);
|
||||
switch (SegmentReg) {
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX:
|
||||
_StoreContext(4, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, es_cached));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX:
|
||||
_StoreContext(4, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, cs_cached));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX:
|
||||
_StoreContext(4, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, ss_cached));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX:
|
||||
_StoreContext(4, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, ds_cached));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX:
|
||||
_StoreContext(4, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, fs_cached));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX:
|
||||
_StoreContext(4, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, gs_cached));
|
||||
break;
|
||||
default: break; // Do nothing
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::LoadSource_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp const& Op, FEXCore::X86Tables::DecodedOperand const& Operand, uint8_t OpSize, uint32_t Flags, int8_t Align, bool LoadData, bool ForceLoad, MemoryAccessType AccessType) {
|
||||
LOGMAN_THROW_A_FMT(Operand.IsGPR() ||
|
||||
Operand.IsLiteral() ||
|
||||
@@ -5495,12 +5788,18 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0x17, 1, &OpDispatchBuilder::POPSegmentOp<FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX>},
|
||||
{0x1E, 1, &OpDispatchBuilder::PUSHSegmentOp<FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX>},
|
||||
{0x1F, 1, &OpDispatchBuilder::POPSegmentOp<FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX>},
|
||||
{0x27, 1, &OpDispatchBuilder::DAAOp},
|
||||
{0x2F, 1, &OpDispatchBuilder::DASOp},
|
||||
{0x37, 1, &OpDispatchBuilder::AAAOp},
|
||||
{0x3F, 1, &OpDispatchBuilder::AASOp},
|
||||
{0x40, 8, &OpDispatchBuilder::INCOp},
|
||||
{0x48, 8, &OpDispatchBuilder::DECOp},
|
||||
|
||||
{0x60, 1, &OpDispatchBuilder::PUSHAOp},
|
||||
{0x61, 1, &OpDispatchBuilder::POPAOp},
|
||||
{0xCE, 1, &OpDispatchBuilder::INTOp},
|
||||
{0xD4, 1, &OpDispatchBuilder::AAMOp},
|
||||
{0xD5, 1, &OpDispatchBuilder::AADOp},
|
||||
};
|
||||
|
||||
constexpr std::tuple<uint8_t, uint8_t, X86Tables::OpDispatchPtr> BaseOpTable_64[] = {
|
||||
|
||||
@@ -278,6 +278,12 @@ public:
|
||||
void NOTOp(OpcodeArgs);
|
||||
void XADDOp(OpcodeArgs);
|
||||
void PopcountOp(OpcodeArgs);
|
||||
void DAAOp(OpcodeArgs);
|
||||
void DASOp(OpcodeArgs);
|
||||
void AAAOp(OpcodeArgs);
|
||||
void AASOp(OpcodeArgs);
|
||||
void AAMOp(OpcodeArgs);
|
||||
void AADOp(OpcodeArgs);
|
||||
void XLATOp(OpcodeArgs);
|
||||
template<bool Reseed>
|
||||
void RDRANDOp(OpcodeArgs);
|
||||
@@ -646,6 +652,7 @@ private:
|
||||
OrderedNode *Current_HeaderNode{};
|
||||
|
||||
OrderedNode *AppendSegmentOffset(OrderedNode *Value, uint32_t Flags, uint32_t DefaultPrefix = 0, bool Override = false);
|
||||
void UpdatePrefixFromSegment(OrderedNode *Segment, uint32_t SegmentReg);
|
||||
|
||||
enum class MemoryAccessType {
|
||||
// Choose TSO or Non-TSO depending on access type
|
||||
|
||||
@@ -769,7 +769,11 @@ void OpDispatchBuilder::CalculcateFlags_ShiftLeftImmediate(uint8_t SrcSize, Orde
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(1, SrcSize * 8 - Shift, Src1));
|
||||
auto OpSize = SrcSize * 8;
|
||||
if (OpSize < Shift) {
|
||||
Shift &= (OpSize - 1);
|
||||
}
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(1, OpSize - Shift, Src1));
|
||||
}
|
||||
|
||||
// PF
|
||||
@@ -934,6 +938,7 @@ void OpDispatchBuilder::CalculcateFlags_RotateRight(uint8_t SrcSize, OrderedNode
|
||||
auto OldOF = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC);
|
||||
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _Xor(_Bfe(1, OpSize - 2, Res), NewCF);
|
||||
|
||||
// If shift == 0, don't update flags
|
||||
@@ -963,7 +968,9 @@ void OpDispatchBuilder::CalculcateFlags_RotateLeft(uint8_t SrcSize, OrderedNode
|
||||
// OF
|
||||
{
|
||||
auto OldOF = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC);
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result
|
||||
// OF is the LSB and MSB XOR'd together.
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result.
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _Xor(_Bfe(1, OpSize - 1, Res), NewCF);
|
||||
|
||||
auto OF = _Select(FEXCore::IR::COND_EQ, Src2, _Constant(0), OldOF, NewOF);
|
||||
@@ -977,8 +984,7 @@ void OpDispatchBuilder::CalculcateFlags_RotateRightImmediate(uint8_t SrcSize, Or
|
||||
if (Shift == 0) return;
|
||||
|
||||
auto OpSize = SrcSize * 8;
|
||||
|
||||
auto NewCF = _Bfe(1, OpSize - Shift, Src1);
|
||||
auto NewCF = _Bfe(1, OpSize - 1, Res);
|
||||
|
||||
// CF
|
||||
{
|
||||
@@ -989,8 +995,10 @@ void OpDispatchBuilder::CalculcateFlags_RotateRightImmediate(uint8_t SrcSize, Or
|
||||
// OF
|
||||
{
|
||||
if (Shift == 1) {
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(_Xor(_Bfe(1, OpSize - 1, Res), NewCF));
|
||||
// OF is the top two MSBs XOR'd together
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _Xor(_Bfe(1, OpSize - 2, Res), NewCF);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(NewOF);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1000,17 +1008,22 @@ void OpDispatchBuilder::CalculcateFlags_RotateLeftImmediate(uint8_t SrcSize, Ord
|
||||
|
||||
auto OpSize = SrcSize * 8;
|
||||
|
||||
auto NewCF = _Bfe(1, 0, Res);
|
||||
// CF
|
||||
{
|
||||
// Extract the last bit shifted in to CF
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(_Bfe(1, Shift, Src1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_CF_LOC>(NewCF);
|
||||
}
|
||||
|
||||
// OF
|
||||
{
|
||||
if (Shift == 1) {
|
||||
// OF is the top two MSBs XOR'd together
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(_Xor(_Bfe(1, OpSize - 1, Src1), _Bfe(1, OpSize - 2, Src1)));
|
||||
// OF is the LSB and MSB XOR'd together.
|
||||
// OF is set to the XOR of the new CF bit and the most significant bit of the result.
|
||||
// OF is architecturally only defined for 1-bit rotate, which is why this only happens when the shift is one.
|
||||
auto NewOF = _Xor(_Bfe(1, OpSize - 1, Res), NewCF);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(NewOF);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -74,13 +74,12 @@ void OpDispatchBuilder::MOVLPOp(OpcodeArgs) {
|
||||
// xmm, xmm is movhlps special case
|
||||
if (Op->Src[0].IsGPR()) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, 8, 16);
|
||||
Src = _VExtractElement(16, 8, Src, 1);
|
||||
auto Result = _VInsScalarElement(16, 8, 0, Dest, Src);
|
||||
auto Result = _VInsElement(16, 8, 0, 1, Dest, Src);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, 16, 16);
|
||||
}
|
||||
else {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, 8, 16);
|
||||
auto Result = _VInsScalarElement(16, 8, 0, Dest, Src);
|
||||
auto Result = _VInsElement(16, 8, 0, 0, Dest, Src);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, 8, 16);
|
||||
}
|
||||
}
|
||||
@@ -112,7 +111,7 @@ void OpDispatchBuilder::MOVSSOp(OpcodeArgs) {
|
||||
// MOVSS xmm1, xmm2
|
||||
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, 16, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 4, Op->Flags, -1);
|
||||
auto Result = _VInsScalarElement(16, 4, 0, Dest, Src);
|
||||
auto Result = _VInsElement(16, 4, 0, 0, Dest, Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
else if (Op->Dest.IsGPR()) {
|
||||
@@ -133,7 +132,7 @@ void OpDispatchBuilder::MOVSDOp(OpcodeArgs) {
|
||||
// xmm1[63:0] <- xmm2[63:0]
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
auto Result = _VInsScalarElement(16, 8, 0, Dest, Src);
|
||||
auto Result = _VInsElement(16, 8, 0, 0, Dest, Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
else if (Op->Dest.IsGPR()) {
|
||||
@@ -335,7 +334,7 @@ void OpDispatchBuilder::VectorScalarALUOp(OpcodeArgs) {
|
||||
|
||||
if (Size != ElementSize) {
|
||||
// Insert the lower bits
|
||||
Result = _VInsScalarElement(Size, ElementSize, 0, Dest, Result);
|
||||
Result = _VInsElement(Size, ElementSize, 0, 0, Dest, Result);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
@@ -381,7 +380,7 @@ void OpDispatchBuilder::VectorUnaryOp(OpcodeArgs) {
|
||||
|
||||
if constexpr (Scalar) {
|
||||
// Insert the lower bits
|
||||
auto Result = _VInsScalarElement(GetSrcSize(Op), ElementSize, 0, Dest, ALUOp);
|
||||
auto Result = _VInsElement(GetSrcSize(Op), ElementSize, 0, 0, Dest, ALUOp);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
else {
|
||||
@@ -978,7 +977,8 @@ void OpDispatchBuilder::PAVGOp<2>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::MOVDDUPOp(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Res = _SplatVector2(Src);
|
||||
OrderedNode *Res = _VDupElement(16, GetSrcSize(Op), Src, 0);
|
||||
|
||||
StoreResult(FPRClass, Op, Res, -1);
|
||||
}
|
||||
|
||||
@@ -992,7 +992,7 @@ void OpDispatchBuilder::CVTGPR_To_FPR(OpcodeArgs) {
|
||||
|
||||
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, 16, Op->Flags, -1);
|
||||
|
||||
Src = _VInsScalarElement(16, DstElementSize, 0, Dest, Src);
|
||||
Src = _VInsElement(16, DstElementSize, 0, 0, Dest, Src);
|
||||
|
||||
StoreResult(FPRClass, Op, Src, -1);
|
||||
}
|
||||
@@ -1091,7 +1091,7 @@ void OpDispatchBuilder::Scalar_CVT_Float_To_Float(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
Src = _Float_FToF(DstElementSize, SrcElementSize, Src);
|
||||
Src = _VInsScalarElement(16, DstElementSize, 0, Dest, Src);
|
||||
Src = _VInsElement(16, DstElementSize, 0, 0, Dest, Src);
|
||||
|
||||
StoreResult(FPRClass, Op, Src, -1);
|
||||
}
|
||||
@@ -1281,7 +1281,7 @@ void OpDispatchBuilder::VFCMPOp(OpcodeArgs) {
|
||||
|
||||
if constexpr (Scalar) {
|
||||
// Insert the lower bits
|
||||
Result = _VInsScalarElement(GetDstSize(Op), ElementSize, 0, Dest, Result);
|
||||
Result = _VInsElement(GetDstSize(Op), ElementSize, 0, 0, Dest, Result);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
@@ -1563,6 +1563,9 @@ void OpDispatchBuilder::PACKSSOp<4>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize, bool Signed>
|
||||
void OpDispatchBuilder::PMULLOp(OpcodeArgs) {
|
||||
static_assert(ElementSize == sizeof(uint32_t),
|
||||
"Currently only handles 32-bit -> 64-bit");
|
||||
|
||||
auto Size = GetSrcSize(Op);
|
||||
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
@@ -1579,17 +1582,8 @@ void OpDispatchBuilder::PMULLOp(OpcodeArgs) {
|
||||
}
|
||||
}
|
||||
else {
|
||||
OrderedNode* Srcs1[2]{};
|
||||
OrderedNode* Srcs2[2]{};
|
||||
|
||||
Srcs1[0] = _VExtr(Size, ElementSize, Src1, Src1, 0);
|
||||
Srcs1[1] = _VExtr(Size, ElementSize, Src1, Src1, 2);
|
||||
|
||||
Srcs2[0] = _VExtr(Size, ElementSize, Src2, Src2, 0);
|
||||
Srcs2[1] = _VExtr(Size, ElementSize, Src2, Src2, 2);
|
||||
|
||||
Src1 = _VInsElement(Size, ElementSize, 1, 0, Srcs1[0], Srcs1[1]);
|
||||
Src2 = _VInsElement(Size, ElementSize, 1, 0, Srcs2[0], Srcs2[1]);
|
||||
Src1 = _VInsElement(Size, ElementSize, 1, 2, Src1, Src1);
|
||||
Src2 = _VInsElement(Size, ElementSize, 1, 2, Src2, Src2);
|
||||
|
||||
if constexpr (Signed) {
|
||||
Res = _VSMull(Size, ElementSize, Src1, Src2);
|
||||
@@ -1714,8 +1708,8 @@ void OpDispatchBuilder::PFNACCOp(OpcodeArgs) {
|
||||
|
||||
OrderedNode *ResSubSrc{};
|
||||
OrderedNode *ResSubDest{};
|
||||
auto UpperSubDest = _VExtractElement(Size, 4, Dest, 1);
|
||||
auto UpperSubSrc = _VExtractElement(Size, 4, Src, 1);
|
||||
auto UpperSubDest = _VDupElement(Size, 4, Dest, 1);
|
||||
auto UpperSubSrc = _VDupElement(Size, 4, Src, 1);
|
||||
|
||||
ResSubDest = _VFSub(4, 4, Dest, UpperSubDest);
|
||||
ResSubSrc = _VFSub(4, 4, Src, UpperSubSrc);
|
||||
@@ -1733,7 +1727,7 @@ void OpDispatchBuilder::PFPNACCOp(OpcodeArgs) {
|
||||
|
||||
OrderedNode *ResAdd{};
|
||||
OrderedNode *ResSub{};
|
||||
auto UpperSubDest = _VExtractElement(Size, 4, Dest, 1);
|
||||
auto UpperSubDest = _VDupElement(Size, 4, Dest, 1);
|
||||
|
||||
ResSub = _VFSub(4, 4, Dest, UpperSubDest);
|
||||
ResAdd = _VFAddP(Size, 4, Src, Src);
|
||||
@@ -1866,8 +1860,6 @@ void OpDispatchBuilder::PMADDWD(OpcodeArgs) {
|
||||
|
||||
if (Size == 8) {
|
||||
Size <<= 1;
|
||||
Src1 = _VBitcast(Size, 2, Src1);
|
||||
Src2 = _VBitcast(Size, 2, Src2);
|
||||
}
|
||||
|
||||
auto Src1_L = _VSXTL(Size, 2, Src1); // [15:0 ], [31:16], [32:47 ], [63:48 ]
|
||||
@@ -1954,9 +1946,6 @@ void OpDispatchBuilder::PMULHW(OpcodeArgs) {
|
||||
|
||||
OrderedNode *Res{};
|
||||
if (Size == 8) {
|
||||
Dest = _VBitcast(Size * 2, 2, Dest);
|
||||
Src = _VBitcast(Size * 2, 2, Src);
|
||||
|
||||
// Implementation is more efficient for 8byte registers
|
||||
if (Signed)
|
||||
Res = _VSMull(Size * 2, 2, Dest, Src);
|
||||
@@ -2333,7 +2322,7 @@ void OpDispatchBuilder::VectorRound(OpcodeArgs) {
|
||||
if constexpr (Scalar) {
|
||||
// Insert the lower bits
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
auto Result = _VInsScalarElement(GetDstSize(Op), ElementSize, 0, Dest, Src);
|
||||
auto Result = _VInsElement(GetDstSize(Op), ElementSize, 0, 0, Dest, Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
else {
|
||||
|
||||
@@ -782,6 +782,12 @@ void OpDispatchBuilder::X87UnaryOp(OpcodeArgs) {
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
|
||||
if constexpr (IROp == IR::OP_F80SIN ||
|
||||
IROp == IR::OP_F80COS) {
|
||||
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
}
|
||||
|
||||
// Write to ST[TOP]
|
||||
_StoreContextIndexed(result, top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
@@ -809,7 +815,8 @@ void OpDispatchBuilder::X87BinaryOp(OpcodeArgs) {
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
|
||||
if constexpr (IROp == IR::OP_F80FPREM) {
|
||||
if constexpr (IROp == IR::OP_F80FPREM ||
|
||||
IROp == IR::OP_F80FPREM1) {
|
||||
//TODO: Set C0 to Q2, C3 to Q1, C1 to Q0
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
}
|
||||
@@ -854,6 +861,9 @@ void OpDispatchBuilder::X87SinCos(OpcodeArgs) {
|
||||
auto sin = _F80SIN(a);
|
||||
auto cos = _F80COS(a);
|
||||
|
||||
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
|
||||
// Write to ST[TOP]
|
||||
_StoreContextIndexed(sin, orig_top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
_StoreContextIndexed(cos, top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
@@ -900,6 +910,9 @@ void OpDispatchBuilder::X87TAN(OpcodeArgs) {
|
||||
OrderedNode *data = _VCastFromGPR(16, 8, low);
|
||||
data = _VInsGPR(16, 8, 1, data, high);
|
||||
|
||||
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
|
||||
// Write to ST[TOP]
|
||||
_StoreContextIndexed(result, orig_top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
_StoreContextIndexed(data, top, 16, MMBaseOffset(), 16, FPRClass);
|
||||
@@ -1185,7 +1198,7 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
// upper 16 bits [79:64]
|
||||
_StoreMem(FPRClass, 8, ST0Location, data, 1);
|
||||
ST0Location = _Add(ST0Location, _Constant(8));
|
||||
auto topBytes = _VExtractElement(16, 2, data, 4);
|
||||
auto topBytes = _VDupElement(16, 2, data, 4);
|
||||
_StoreMem(FPRClass, 2, ST0Location, topBytes, 1);
|
||||
|
||||
// reset to default
|
||||
|
||||
@@ -778,6 +778,12 @@ void OpDispatchBuilder::X87UnaryOpF64(OpcodeArgs) {
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
|
||||
if constexpr (IROp == IR::OP_F64SIN ||
|
||||
IROp == IR::OP_F64COS) {
|
||||
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
}
|
||||
|
||||
// Write to ST[TOP]
|
||||
_StoreContextIndexed(result, top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
}
|
||||
@@ -804,7 +810,8 @@ void OpDispatchBuilder::X87BinaryOpF64(OpcodeArgs) {
|
||||
// Overwrite the op
|
||||
result.first->Header.Op = IROp;
|
||||
|
||||
if constexpr (IROp == IR::OP_F64FPREM) {
|
||||
if constexpr (IROp == IR::OP_F80FPREM ||
|
||||
IROp == IR::OP_F80FPREM1) {
|
||||
//TODO: Set C0 to Q2, C3 to Q1, C1 to Q0
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
}
|
||||
@@ -831,6 +838,9 @@ void OpDispatchBuilder::X87SinCosF64(OpcodeArgs) {
|
||||
auto sin = _F64SIN(a);
|
||||
auto cos = _F64COS(a);
|
||||
|
||||
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
|
||||
// Write to ST[TOP]
|
||||
_StoreContextIndexed(sin, orig_top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
_StoreContextIndexed(cos, top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
@@ -871,6 +881,9 @@ void OpDispatchBuilder::X87TANF64(OpcodeArgs) {
|
||||
|
||||
auto one = _VCastFromGPR(8, 8, _Constant(0x3FF0000000000000));
|
||||
|
||||
// TODO: ACCURACY: should check source is in range –2^63 to +2^63
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
|
||||
// Write to ST[TOP]
|
||||
_StoreContextIndexed(result, orig_top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
_StoreContextIndexed(one, top, 8, MMBaseOffset(), 16, FPRClass);
|
||||
@@ -996,7 +1009,7 @@ void OpDispatchBuilder::X87FNSAVEF64(OpcodeArgs) {
|
||||
// upper 16 bits [79:64]
|
||||
_StoreMem(FPRClass, 8, ST0Location, data, 1);
|
||||
ST0Location = _Add(ST0Location, _Constant(8));
|
||||
auto topBytes = _VExtractElement(16, 2, data, 4);
|
||||
auto topBytes = _VDupElement(16, 2, data, 4);
|
||||
_StoreMem(FPRClass, 2, ST0Location, topBytes, 1);
|
||||
|
||||
// reset to default
|
||||
|
||||
@@ -266,10 +266,10 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0x17, 1, X86InstInfo{"POP SS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_DEF) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0x1E, 1, X86InstInfo{"PUSH DS", TYPE_INST, GenFlagsSrcSize(SIZE_16BIT) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0x1F, 1, X86InstInfo{"POP DS", TYPE_INST, GenFlagsSizes(SIZE_16BIT, SIZE_DEF) | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{0x27, 1, X86InstInfo{"DAA", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x2F, 1, X86InstInfo{"DAS", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x37, 1, X86InstInfo{"AAA", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x3F, 1, X86InstInfo{"AAS", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x27, 1, X86InstInfo{"DAA", TYPE_INST, GenFlagsDstSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 0, nullptr}},
|
||||
{0x2F, 1, X86InstInfo{"DAS", TYPE_INST, GenFlagsDstSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 0, nullptr}},
|
||||
{0x37, 1, X86InstInfo{"AAA", TYPE_INST, GenFlagsDstSize(SIZE_16BIT) | FLAGS_SF_DST_RAX, 0, nullptr}},
|
||||
{0x3F, 1, X86InstInfo{"AAS", TYPE_INST, GenFlagsDstSize(SIZE_16BIT) | FLAGS_SF_DST_RAX, 0, nullptr}},
|
||||
|
||||
{0x40, 8, X86InstInfo{"INC", TYPE_INST, FLAGS_SF_REX_IN_BYTE, 0, nullptr}},
|
||||
{0x48, 8, X86InstInfo{"DEC", TYPE_INST, FLAGS_SF_REX_IN_BYTE, 0, nullptr}},
|
||||
@@ -283,8 +283,8 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0xA1, 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_MEM_OFFSET, 4, nullptr}},
|
||||
{0xA3, 1, X86InstInfo{"MOV", TYPE_INST, FLAGS_SF_SRC_RAX | FLAGS_MEM_OFFSET, 4, nullptr}},
|
||||
{0xCE, 1, X86InstInfo{"INTO", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0xD4, 1, X86InstInfo{"AAM", TYPE_INST, FLAGS_NONE, 1, nullptr}},
|
||||
{0xD5, 1, X86InstInfo{"AAD", TYPE_INST, FLAGS_NONE, 1, nullptr}},
|
||||
{0xD4, 1, X86InstInfo{"AAM", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 1, nullptr}},
|
||||
{0xD5, 1, X86InstInfo{"AAD", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX, 1, nullptr}},
|
||||
{0xEA, 1, X86InstInfo{"JMPF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
};
|
||||
|
||||
|
||||
+31
-59
@@ -302,10 +302,6 @@
|
||||
}
|
||||
},
|
||||
"Moves": {
|
||||
"GPR = Mov GPR:$Value": {
|
||||
"DestSize": "GetOpSize(_Value)"
|
||||
},
|
||||
|
||||
"GPR = ExtractElementPair GPRPair:$Pair, u8:$Element": {
|
||||
"Desc": ["Extracts a register for the register pair"],
|
||||
"DestSize": "GetOpSize(_Pair) >> 1"
|
||||
@@ -359,22 +355,22 @@
|
||||
"DestSize": "ByteSize",
|
||||
"EmitValidation": [
|
||||
"($Class == GPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16)) || $Class == GPRClass"
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16 || #ByteSize == 32)) || $Class == GPRClass"
|
||||
]
|
||||
},
|
||||
|
||||
"StoreContext u8:#ByteSize, RegisterClass:$Class, SSA:$Value, u32:$Offset": {
|
||||
"Desc": ["Stores a value to the context with offset",
|
||||
"Ctx[Offset] = Value",
|
||||
"Zero Extends if value's type is too small",
|
||||
"Truncates if value's type is too large"
|
||||
],
|
||||
"Desc": ["Stores a value to the context with offset",
|
||||
"Ctx[Offset] = Value",
|
||||
"Zero Extends if value's type is too small",
|
||||
"Truncates if value's type is too large"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "ByteSize",
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value) == $Class",
|
||||
"($Class == GPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16)) || $Class == GPRClass"
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16 || #ByteSize == 32)) || $Class == GPRClass"
|
||||
]
|
||||
},
|
||||
|
||||
@@ -385,7 +381,7 @@
|
||||
"DestSize": "ByteSize",
|
||||
"EmitValidation": [
|
||||
"($Class == GPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16)) || $Class == GPRClass"
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16 || #ByteSize == 32)) || $Class == GPRClass"
|
||||
]
|
||||
},
|
||||
"StoreContextIndexed SSA:$Value, GPR:$Index, u8:#ByteSize, u32:$BaseOffset, u32:$Stride, RegisterClass:$Class": {
|
||||
@@ -397,7 +393,7 @@
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value) == $Class",
|
||||
"($Class == GPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16)) || $Class == GPRClass"
|
||||
"($Class == FPRClass && (#ByteSize == 1 || #ByteSize == 2 || #ByteSize == 4 || #ByteSize == 8 || #ByteSize == 16 || #ByteSize == 32)) || $Class == GPRClass"
|
||||
]
|
||||
},
|
||||
|
||||
@@ -845,27 +841,25 @@
|
||||
"DestSize": "std::max<uint8_t>(4, std::max<uint8_t>(GetOpSize(_TrueVal), GetOpSize(_FalseVal)))"
|
||||
},
|
||||
"GPR = Extr GPR:$Upper, GPR:$Lower, u8:$LSB": {
|
||||
"Desc": ["Concats the two GPRs to create a value that is the size of the full two GPRs",
|
||||
"It then extracts a bitfield width that size of a GPR from the LSB",
|
||||
"Valid LSB range is 0-31 for 32bit and 0-63 for 64bit",
|
||||
"<Size * 2> ConcatValue = $Upper:$Lower",
|
||||
"Result = ConcatValue<LSB+Size - 1: LSB>"
|
||||
]
|
||||
"Desc": ["Concats the two GPRs to create a value that is the size of the full two GPRs",
|
||||
"It then extracts a bitfield width that size of a GPR from the LSB",
|
||||
"Valid LSB range is 0-31 for 32bit and 0-63 for 64bit",
|
||||
"<Size * 2> ConcatValue = $Upper:$Lower",
|
||||
"Result = ConcatValue<LSB+Size - 1: LSB>"
|
||||
]
|
||||
},
|
||||
"GPR = PDep GPR:$Input, GPR:$Mask": {
|
||||
"Desc": [
|
||||
"Performs a parallel bit deposit.",
|
||||
"Takes the contiguous low-order bits and deposits them into",
|
||||
"the destination at the locations specified by the Mask."
|
||||
]
|
||||
"Desc": ["Performs a parallel bit deposit.",
|
||||
"Takes the contiguous low-order bits and deposits them into",
|
||||
"the destination at the locations specified by the Mask."
|
||||
]
|
||||
},
|
||||
|
||||
"GPR = PExt GPR:$Input, GPR:$Mask": {
|
||||
"Desc": [
|
||||
"Performs a parallel bit extract.",
|
||||
"Each bit set in the mask will select the corresponding bit in the Input",
|
||||
"and transfers them to the lower contiguous bits in the destination."
|
||||
]
|
||||
"Desc": ["Performs a parallel bit extract.",
|
||||
"Each bit set in the mask will select the corresponding bit in the Input",
|
||||
"and transfers them to the lower contiguous bits in the destination."
|
||||
]
|
||||
},
|
||||
|
||||
"GPR = LDiv GPR:$Lower, GPR:$Upper, GPR:$Divisor": {
|
||||
@@ -924,15 +918,6 @@
|
||||
}
|
||||
},
|
||||
"Vector": {
|
||||
"FPR = SplatVector2 FPR:$Scalar": {
|
||||
"NumElements": "2",
|
||||
"DestSize": "GetOpSize(_Scalar) * 2"
|
||||
},
|
||||
"FPR = SplatVector4 FPR:$Scalar": {
|
||||
"NumElements": "4",
|
||||
"DestSize": "GetOpSize(_Scalar) * 4"
|
||||
},
|
||||
|
||||
"FPR = VMov u8:#RegisterSize, FPR:$Source": {
|
||||
"Desc" : ["Copy vector register",
|
||||
"When Register size is smaller than Source register size,",
|
||||
@@ -941,12 +926,6 @@
|
||||
"DestSize": "RegisterSize"
|
||||
},
|
||||
|
||||
"FPR = VBitcast u8:#RegisterSize, u8:#ElementSize, FPR:$Source": {
|
||||
"Desc": ["Workaround for issue with LLVM breaking when loading scalar elements to vectors"],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
|
||||
"FPR = VectorZero u8:#RegisterSize": {
|
||||
"Desc": ["Generates a vector zero",
|
||||
"Useful to generate a zero vector without any previous dependencies"
|
||||
@@ -1038,9 +1017,6 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VExtractElement u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, u8:$Index": {
|
||||
"DestSize": "ElementSize"
|
||||
},
|
||||
"FPR = VDupElement u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, u8:$Index": {
|
||||
"Desc": ["Duplicates one element from the source register across the whole register"],
|
||||
"DestSize": "RegisterSize",
|
||||
@@ -1089,7 +1065,7 @@
|
||||
},
|
||||
"FPR = VSXTL2 u8:#RegisterSize, u8:#ElementSize, FPR:$Vector": {
|
||||
"Desc": ["Sign extends elements from the source element size to the next size up",
|
||||
"Source elements come from the upper 64bits of the register"
|
||||
"Source elements come from the upper half of the register"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
@@ -1101,7 +1077,7 @@
|
||||
},
|
||||
"FPR = VUXTL2 u8:#RegisterSize, u8:#ElementSize, FPR:$Vector": {
|
||||
"Desc": ["Zero extends elements from the source element size to the next size up",
|
||||
"Source elements come from the upper 64bits of the register"
|
||||
"Source elements come from the upper half of the register"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
@@ -1124,9 +1100,9 @@
|
||||
},
|
||||
|
||||
"FPR = VRev64 u8:#RegisterSize, u8:#ElementSize, FPR:$Vector": {
|
||||
"Desc" : ["Reverses elements in 64-bit halfwords",
|
||||
"Available element size: 1byte, 2 byte, 4 byte"
|
||||
],
|
||||
"Desc" : ["Reverses elements in 64-bit halfwords",
|
||||
"Available element size: 1byte, 2 byte, 4 byte"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -1320,10 +1296,6 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VInsScalarElement u8:#RegisterSize, u8:#ElementSize, u8:$DestIdx, FPR:$DestVector, FPR:$SrcScalar": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VInsGPR u8:#RegisterSize, u8:#ElementSize, u8:$DestIdx, FPR:$DestVector, GPR:$Src": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
@@ -1587,9 +1559,9 @@
|
||||
"DestSize": "16"
|
||||
},
|
||||
"GPR = F80Cmp FPR:$X80Src1, FPR:$X80Src2, u32:$Flags": {
|
||||
"Desc": ["Does a scalar unordered compare and stores the asked for flags in to a GPR",
|
||||
"Ordering flag result is true if either float input is NaN"
|
||||
],
|
||||
"Desc": ["Does a scalar unordered compare and stores the asked for flags in to a GPR",
|
||||
"Ordering flag result is true if either float input is NaN"
|
||||
],
|
||||
"DestSize": "4"
|
||||
},
|
||||
"FPR = F80BCDLoad FPR:$X80Src": {
|
||||
|
||||
@@ -12,6 +12,7 @@ $end_info$
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
class IREmitter;
|
||||
@@ -66,6 +67,8 @@ void PassManager::InsertRegisterAllocationPass(bool OptimizeSRA, bool SupportsAV
|
||||
}
|
||||
|
||||
bool PassManager::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::Run");
|
||||
|
||||
bool Changed = false;
|
||||
for (auto const &Pass : Passes) {
|
||||
Changed |= Pass->Run(IREmit);
|
||||
|
||||
+12
-9
@@ -6,7 +6,7 @@ $end_info$
|
||||
*/
|
||||
|
||||
|
||||
#if defined(_M_ARM_64)
|
||||
#if JIT_ARM64
|
||||
//aarch64 heuristics
|
||||
#include "aarch64/assembler-aarch64.h"
|
||||
#include "aarch64/cpu-aarch64.h"
|
||||
@@ -20,6 +20,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <bit>
|
||||
#include <cstdint>
|
||||
@@ -45,20 +46,20 @@ uint64_t getMask(IROp_Header* Op) {
|
||||
return (~0ULL) >> (64 - NumBits);
|
||||
}
|
||||
|
||||
#ifdef _M_X86_64
|
||||
// very lazy heuristics
|
||||
static bool IsImmLogical(uint64_t imm, unsigned width) { return imm < 0x8000'0000; }
|
||||
static bool IsImmAddSub(uint64_t imm) { return imm < 0x8000'0000; }
|
||||
static bool IsMemoryScale(uint64_t Scale, uint8_t AccessSize) {
|
||||
return Scale == 1 || Scale == 2 || Scale == 4 || Scale == 8;
|
||||
}
|
||||
#elif defined(_M_ARM_64)
|
||||
#if JIT_ARM64
|
||||
//aarch64 heuristics
|
||||
static bool IsImmLogical(uint64_t imm, unsigned width) { if (width < 32) width = 32; return vixl::aarch64::Assembler::IsImmLogical(imm, width); }
|
||||
static bool IsImmAddSub(uint64_t imm) { return vixl::aarch64::Assembler::IsImmAddSub(imm); }
|
||||
static bool IsMemoryScale(uint64_t Scale, uint8_t AccessSize) {
|
||||
return Scale == AccessSize;
|
||||
}
|
||||
#elif JIT_X86_64
|
||||
// very lazy heuristics
|
||||
static bool IsImmLogical(uint64_t imm, unsigned width) { return imm < 0x8000'0000; }
|
||||
static bool IsImmAddSub(uint64_t imm) { return imm < 0x8000'0000; }
|
||||
static bool IsMemoryScale(uint64_t Scale, uint8_t AccessSize) {
|
||||
return Scale == 1 || Scale == 2 || Scale == 4 || Scale == 8;
|
||||
}
|
||||
#else
|
||||
#error No inline constant heuristics for this target
|
||||
#endif
|
||||
@@ -1028,6 +1029,8 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
|
||||
}
|
||||
|
||||
bool ConstProp::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::ConstProp");
|
||||
|
||||
bool Changed = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
auto OriginalWriteCursor = IREmit->GetWriteCursor();
|
||||
|
||||
@@ -9,6 +9,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <memory>
|
||||
|
||||
@@ -22,6 +23,7 @@ private:
|
||||
};
|
||||
|
||||
bool DeadCodeElimination::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::DCE");
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
int NumRemoved = 0;
|
||||
|
||||
|
||||
+134
-66
@@ -13,6 +13,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <array>
|
||||
#include <memory>
|
||||
@@ -76,24 +77,6 @@ namespace {
|
||||
std::vector<ContextMemberInfo> ClassificationInfo;
|
||||
};
|
||||
|
||||
constexpr static std::array<LastAccessType, 15> DefaultAccess = {
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_INVALID, // SSE padding in non-AVX case
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
ACCESS_NONE,
|
||||
};
|
||||
|
||||
static void ClassifyContextStruct(ContextInfo *ContextClassificationInfo, bool SupportsAVX) {
|
||||
auto ContextClassification = &ContextClassificationInfo->ClassificationInfo;
|
||||
|
||||
@@ -102,7 +85,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, rip),
|
||||
sizeof(FEXCore::Core::CPUState::rip),
|
||||
},
|
||||
DefaultAccess[0],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
@@ -112,62 +95,134 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, gregs[0]) + sizeof(FEXCore::Core::CPUState::gregs[0]) * i,
|
||||
FEXCore::Core::CPUState::GPR_REG_SIZE,
|
||||
},
|
||||
DefaultAccess[1],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, es),
|
||||
sizeof(FEXCore::Core::CPUState::es),
|
||||
offsetof(FEXCore::Core::CPUState, es_idx),
|
||||
sizeof(FEXCore::Core::CPUState::es_idx),
|
||||
},
|
||||
DefaultAccess[2],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, cs),
|
||||
sizeof(FEXCore::Core::CPUState::cs),
|
||||
offsetof(FEXCore::Core::CPUState, cs_idx),
|
||||
sizeof(FEXCore::Core::CPUState::cs_idx),
|
||||
},
|
||||
DefaultAccess[3],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, ss),
|
||||
sizeof(FEXCore::Core::CPUState::ss),
|
||||
offsetof(FEXCore::Core::CPUState, ss_idx),
|
||||
sizeof(FEXCore::Core::CPUState::ss_idx),
|
||||
},
|
||||
DefaultAccess[4],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, ds),
|
||||
sizeof(FEXCore::Core::CPUState::ds),
|
||||
offsetof(FEXCore::Core::CPUState, ds_idx),
|
||||
sizeof(FEXCore::Core::CPUState::ds_idx),
|
||||
},
|
||||
DefaultAccess[5],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, gs),
|
||||
sizeof(FEXCore::Core::CPUState::gs),
|
||||
offsetof(FEXCore::Core::CPUState, gs_idx),
|
||||
sizeof(FEXCore::Core::CPUState::gs_idx),
|
||||
},
|
||||
DefaultAccess[6],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, fs),
|
||||
sizeof(FEXCore::Core::CPUState::fs),
|
||||
offsetof(FEXCore::Core::CPUState, fs_idx),
|
||||
sizeof(FEXCore::Core::CPUState::fs_idx),
|
||||
},
|
||||
DefaultAccess[7],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, _pad),
|
||||
sizeof(FEXCore::Core::CPUState::_pad),
|
||||
},
|
||||
ACCESS_INVALID,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, es_cached),
|
||||
sizeof(FEXCore::Core::CPUState::es_cached),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, cs_cached),
|
||||
sizeof(FEXCore::Core::CPUState::cs_cached),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, ss_cached),
|
||||
sizeof(FEXCore::Core::CPUState::ss_cached),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, ds_cached),
|
||||
sizeof(FEXCore::Core::CPUState::ds_cached),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, gs_cached),
|
||||
sizeof(FEXCore::Core::CPUState::gs_cached),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, fs_cached),
|
||||
sizeof(FEXCore::Core::CPUState::fs_cached),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
ContextClassification->emplace_back(ContextMemberInfo{
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, _pad2),
|
||||
sizeof(FEXCore::Core::CPUState::_pad2),
|
||||
},
|
||||
ACCESS_INVALID,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
@@ -178,7 +233,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.avx.data[0][0]) + FEXCore::Core::CPUState::XMM_AVX_REG_SIZE * i,
|
||||
FEXCore::Core::CPUState::XMM_AVX_REG_SIZE,
|
||||
},
|
||||
DefaultAccess[8],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
@@ -189,7 +244,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.sse.data[0][0]) + FEXCore::Core::CPUState::XMM_SSE_REG_SIZE * i,
|
||||
FEXCore::Core::CPUState::XMM_SSE_REG_SIZE,
|
||||
},
|
||||
DefaultAccess[8],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
@@ -199,7 +254,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.sse.pad[0][0]),
|
||||
static_cast<uint16_t>(FEXCore::Core::CPUState::XMM_SSE_REG_SIZE * FEXCore::Core::CPUState::NUM_XMMS),
|
||||
},
|
||||
DefaultAccess[9],
|
||||
ACCESS_INVALID,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
@@ -210,7 +265,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, flags[0]) + sizeof(FEXCore::Core::CPUState::flags[0]) * i,
|
||||
FEXCore::Core::CPUState::FLAG_SIZE,
|
||||
},
|
||||
DefaultAccess[10],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
@@ -221,7 +276,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, mm[0][0]) + sizeof(FEXCore::Core::CPUState::mm[0]) * i,
|
||||
FEXCore::Core::CPUState::MM_REG_SIZE
|
||||
},
|
||||
DefaultAccess[11],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
@@ -233,7 +288,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, gdt[0]) + sizeof(FEXCore::Core::CPUState::gdt[0]) * i,
|
||||
sizeof(FEXCore::Core::CPUState::gdt[0]),
|
||||
},
|
||||
DefaultAccess[12],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
}
|
||||
@@ -244,7 +299,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, FCW),
|
||||
sizeof(FEXCore::Core::CPUState::FCW),
|
||||
},
|
||||
DefaultAccess[13],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
@@ -254,7 +309,7 @@ namespace {
|
||||
offsetof(FEXCore::Core::CPUState, FTW),
|
||||
sizeof(FEXCore::Core::CPUState::FTW),
|
||||
},
|
||||
DefaultAccess[14],
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
@@ -288,40 +343,55 @@ namespace {
|
||||
ContextClassification->at(Offset).StoreNode = nullptr;
|
||||
};
|
||||
size_t Offset = 0;
|
||||
SetAccess(Offset++, DefaultAccess[0]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_GPRS; ++i) {
|
||||
SetAccess(Offset++, DefaultAccess[1]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
}
|
||||
|
||||
SetAccess(Offset++, DefaultAccess[2]);
|
||||
SetAccess(Offset++, DefaultAccess[3]);
|
||||
SetAccess(Offset++, DefaultAccess[4]);
|
||||
SetAccess(Offset++, DefaultAccess[5]);
|
||||
SetAccess(Offset++, DefaultAccess[6]);
|
||||
SetAccess(Offset++, DefaultAccess[7]);
|
||||
// Segment indexes
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
|
||||
// Pad
|
||||
SetAccess(Offset++, ACCESS_INVALID);
|
||||
|
||||
// Segments
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
|
||||
// Pad2
|
||||
SetAccess(Offset++, ACCESS_INVALID);
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; ++i) {
|
||||
SetAccess(Offset++, DefaultAccess[8]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
}
|
||||
|
||||
if (!SupportsAVX) {
|
||||
SetAccess(Offset++, DefaultAccess[9]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_FLAGS; ++i) {
|
||||
SetAccess(Offset++, DefaultAccess[10]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_MMS; ++i) {
|
||||
SetAccess(Offset++, DefaultAccess[11]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_GDTS; ++i) {
|
||||
SetAccess(Offset++, DefaultAccess[12]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
}
|
||||
|
||||
SetAccess(Offset++, DefaultAccess[13]);
|
||||
SetAccess(Offset++, DefaultAccess[14]);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
}
|
||||
|
||||
struct BlockInfo {
|
||||
@@ -449,7 +519,6 @@ void RCLSE::CalculateControlFlowInfo(FEXCore::IR::IREmitter *IREmit) {
|
||||
* %ssa26 i128 = LoadMem %ssa25 i64, 0x10
|
||||
* (%%ssa27) StoreContext %ssa26 i128, 0x10, 0xb0
|
||||
* %ssa28 i128 = LoadContext 0x10, 0x90
|
||||
* %ssa29 i128 = VBitcast %ssa26 i128
|
||||
*
|
||||
* eg.
|
||||
* %ssa6 i128 = LoadContext 0x10, 0x90
|
||||
@@ -462,13 +531,11 @@ void RCLSE::CalculateControlFlowInfo(FEXCore::IR::IREmitter *IREmit) {
|
||||
* eg.
|
||||
* (%%ssa189) StoreContext %ssa188 i128, 0x10, 0xa0
|
||||
* %ssa190 i128 = LoadContext 0x10, 0x90
|
||||
* %ssa191 i128 = VBitcast %ssa188 i128
|
||||
* %ssa192 i128 = VAdd %ssa191 i128, %ssa190 i128, 0x10, 0x4
|
||||
* %ssa192 i128 = VAdd %ssa188 i128, %ssa190 i128, 0x10, 0x4
|
||||
* (%%ssa193) StoreContext %ssa192 i128, 0x10, 0xa0
|
||||
* Converts to
|
||||
* %ssa173 i128 = LoadContext 0x10, 0x90
|
||||
* %ssa174 i128 = VBitcast %ssa172 i128
|
||||
* %ssa175 i128 = VAdd %ssa174 i128, %ssa173 i128, 0x10, 0x4
|
||||
* %ssa175 i128 = VAdd %ssa172 i128, %ssa173 i128, 0x10, 0x4
|
||||
* (%%ssa176) StoreContext %ssa175 i128, 0x10, 0xa0
|
||||
|
||||
*/
|
||||
@@ -698,6 +765,7 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
}
|
||||
|
||||
bool RCLSE::Run(FEXCore::IR::IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::RCLSE");
|
||||
// XXX: We don't do cross-block optimizations yet
|
||||
//CalculateControlFlowInfo(IREmit);
|
||||
bool Changed = false;
|
||||
|
||||
@@ -12,6 +12,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <memory>
|
||||
#include <stddef.h>
|
||||
@@ -154,6 +155,8 @@ struct Info {
|
||||
*
|
||||
*/
|
||||
bool DeadStoreElimination::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::DSE");
|
||||
|
||||
std::unordered_map<OrderedNode*, Info> InfoMap;
|
||||
|
||||
bool Changed = false;
|
||||
|
||||
@@ -13,6 +13,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <cstdint>
|
||||
@@ -52,6 +53,8 @@ IRCompaction::IRCompaction(FEXCore::Utils::IntrusivePooledAllocator &Allocator)
|
||||
}
|
||||
|
||||
bool IRCompaction::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::IRCompaction");
|
||||
|
||||
LocalBuilder.ReownOrClaimBuffer();
|
||||
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
@@ -14,6 +14,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
@@ -32,6 +33,8 @@ IRValidation::~IRValidation() {
|
||||
}
|
||||
|
||||
bool IRValidation::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::IRValidation");
|
||||
|
||||
bool HadError = false;
|
||||
bool HadWarning = false;
|
||||
|
||||
|
||||
@@ -9,6 +9,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <memory>
|
||||
#include <stdint.h>
|
||||
@@ -53,6 +54,8 @@ bool LongDivideEliminationPass::IsSextOp(IREmitter *IREmit, OrderedNodeWrapper L
|
||||
}
|
||||
|
||||
bool LongDivideEliminationPass::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::LDE");
|
||||
|
||||
bool Changed = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
auto OriginalWriteCursor = IREmit->GetWriteCursor();
|
||||
|
||||
@@ -9,6 +9,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
@@ -24,6 +25,8 @@ public:
|
||||
};
|
||||
|
||||
bool PhiValidation::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::PHIValidation");
|
||||
|
||||
bool HadError = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
|
||||
@@ -6,7 +6,7 @@
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <deque>
|
||||
@@ -191,6 +191,8 @@ private:
|
||||
bool RAValidation::Run(IREmitter *IREmit) {
|
||||
if (!Manager->HasPass("RA")) return false;
|
||||
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::RAValidation");
|
||||
|
||||
IR::RegisterAllocationData* RAData = Manager->GetPass<IR::RegisterAllocationPass>("RA")->GetAllocationData();
|
||||
BlockExitState.clear();
|
||||
// BlocksToVisit will already be empty
|
||||
|
||||
+4
@@ -8,6 +8,8 @@ $end_info$
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
#include <array>
|
||||
@@ -32,6 +34,8 @@ public:
|
||||
*
|
||||
*/
|
||||
bool DeadFlagCalculationEliminination::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::DFE");
|
||||
|
||||
std::array<OrderedNode*, 32> LastValidFlagStores{};
|
||||
|
||||
bool Changed = false;
|
||||
|
||||
@@ -15,6 +15,8 @@ $end_info$
|
||||
#include <FEXCore/Utils/BucketList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <FEXHeaderUtils/TypeDefines.h>
|
||||
|
||||
#include <algorithm>
|
||||
@@ -1527,6 +1529,7 @@ namespace {
|
||||
}
|
||||
|
||||
bool ConstrainedRAPass::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::RA");
|
||||
bool Changed = false;
|
||||
|
||||
auto IR = IREmit->ViewIR();
|
||||
|
||||
@@ -11,6 +11,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <memory>
|
||||
#include <stddef.h>
|
||||
@@ -76,6 +77,8 @@ private:
|
||||
*
|
||||
*/
|
||||
bool StaticRegisterAllocationPass::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::SRA");
|
||||
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
|
||||
@@ -11,6 +11,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <memory>
|
||||
#include <stdint.h>
|
||||
@@ -23,6 +24,8 @@ public:
|
||||
};
|
||||
|
||||
bool SyscallOptimization::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::SyscallOpt");
|
||||
|
||||
bool Changed = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
|
||||
@@ -11,6 +11,7 @@ $end_info$
|
||||
#include <FEXCore/IR/IREmitter.h>
|
||||
#include <FEXCore/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <functional>
|
||||
#include <memory>
|
||||
@@ -36,6 +37,8 @@ public:
|
||||
};
|
||||
|
||||
bool ValueDominanceValidation::Run(IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::ValueDominanceValidation");
|
||||
|
||||
bool HadError = false;
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
|
||||
+38
-7
@@ -45,6 +45,8 @@ namespace FEXCore::Allocator {
|
||||
FREE_Hook free {::free};
|
||||
#endif
|
||||
|
||||
uint64_t HostVASize{};
|
||||
|
||||
using GLIBC_MALLOC_Hook = void*(*)(size_t, const void *caller);
|
||||
using GLIBC_REALLOC_Hook = void*(*)(void*, size_t, const void *caller);
|
||||
using GLIBC_FREE_Hook = void(*)(void*, const void *caller);
|
||||
@@ -97,6 +99,10 @@ namespace FEXCore::Allocator {
|
||||
#pragma GCC diagnostic pop
|
||||
|
||||
FEX_DEFAULT_VISIBILITY size_t DetermineVASize() {
|
||||
if (HostVASize) {
|
||||
return HostVASize;
|
||||
}
|
||||
|
||||
static constexpr std::array<uintptr_t, 7> TLBSizes = {
|
||||
57,
|
||||
52,
|
||||
@@ -127,6 +133,7 @@ namespace FEXCore::Allocator {
|
||||
};
|
||||
|
||||
if (Find(Size)) {
|
||||
HostVASize = Bits;
|
||||
return Bits;
|
||||
}
|
||||
}
|
||||
@@ -138,8 +145,10 @@ namespace FEXCore::Allocator {
|
||||
#define STEAL_LOG(...) // fprintf(stderr, __VA_ARGS__)
|
||||
|
||||
std::vector<MemoryRegion> StealMemoryRegion(uintptr_t Begin, uintptr_t End) {
|
||||
void * const StackLocation = alloca(0);
|
||||
const uintptr_t StackLocation_u64 = reinterpret_cast<uintptr_t>(StackLocation);
|
||||
std::vector<MemoryRegion> Regions;
|
||||
|
||||
|
||||
int MapsFD = open("/proc/self/maps", O_RDONLY);
|
||||
LogMan::Throw::AFmt(MapsFD != -1, "Failed to open /proc/self/maps");
|
||||
|
||||
@@ -148,6 +157,8 @@ namespace FEXCore::Allocator {
|
||||
uintptr_t RegionBegin = 0;
|
||||
uintptr_t RegionEnd = 0;
|
||||
|
||||
uintptr_t PreviousMapEnd = 0;
|
||||
|
||||
char Buffer[2048];
|
||||
const char *Cursor;
|
||||
ssize_t Remaining = 0;
|
||||
@@ -155,7 +166,7 @@ namespace FEXCore::Allocator {
|
||||
for(;;) {
|
||||
|
||||
if (Remaining == 0) {
|
||||
do {
|
||||
do {
|
||||
Remaining = read(MapsFD, Buffer, sizeof(Buffer));
|
||||
} while ( Remaining == -1 && errno == EAGAIN);
|
||||
|
||||
@@ -165,8 +176,8 @@ namespace FEXCore::Allocator {
|
||||
if (Remaining == 0 && State == ParseBegin) {
|
||||
STEAL_LOG("[%d] EndOfFile; RegionBegin: %016lX RegionEnd: %016lX\n", __LINE__, RegionBegin, RegionEnd);
|
||||
|
||||
auto MapBegin = std::max(RegionEnd, Begin);
|
||||
auto MapEnd = End;
|
||||
const auto MapBegin = std::max(RegionEnd, Begin);
|
||||
const auto MapEnd = End;
|
||||
|
||||
STEAL_LOG(" MapBegin: %016lX MapEnd: %016lX\n", MapBegin, MapEnd);
|
||||
|
||||
@@ -202,9 +213,12 @@ namespace FEXCore::Allocator {
|
||||
if (c == '-') {
|
||||
STEAL_LOG("[%d] ParseBegin; RegionBegin: %016lX RegionEnd: %016lX\n", __LINE__, RegionBegin, RegionEnd);
|
||||
|
||||
auto MapBegin = std::max(RegionEnd, Begin);
|
||||
auto MapEnd = std::min(RegionBegin, End);
|
||||
|
||||
const auto MapBegin = std::max(RegionEnd, Begin);
|
||||
const auto MapEnd = std::min(RegionBegin, End);
|
||||
|
||||
// Store the location we are going to map.
|
||||
PreviousMapEnd = MapEnd;
|
||||
|
||||
STEAL_LOG(" MapBegin: %016lX MapEnd: %016lX\n", MapBegin, MapEnd);
|
||||
|
||||
if (MapEnd > MapBegin) {
|
||||
@@ -218,6 +232,7 @@ namespace FEXCore::Allocator {
|
||||
|
||||
Regions.push_back({(void*)MapBegin, MapSize});
|
||||
}
|
||||
|
||||
RegionBegin = 0;
|
||||
RegionEnd = 0;
|
||||
State = ParseEnd;
|
||||
@@ -233,6 +248,22 @@ namespace FEXCore::Allocator {
|
||||
STEAL_LOG("[%d] ParseEnd; RegionBegin: %016lX RegionEnd: %016lX\n", __LINE__, RegionBegin, RegionEnd);
|
||||
|
||||
State = ScanEnd;
|
||||
|
||||
// If the previous map's ending and the region we just parsed overlap the stack then we need to save the stack mapping.
|
||||
// Otherwise we will have severely limited stack size which crashes quickly.
|
||||
if (PreviousMapEnd <= StackLocation_u64 && RegionEnd > StackLocation_u64) {
|
||||
auto BelowStackRegion = Regions.back();
|
||||
LOGMAN_THROW_AA_FMT(reinterpret_cast<uint64_t>(BelowStackRegion.Ptr) + BelowStackRegion.Size == PreviousMapEnd,
|
||||
"This needs to match");
|
||||
|
||||
// Allocate the region under the stack as READ | WRITE so the stack can still grow
|
||||
auto Alloc = mmap(BelowStackRegion.Ptr, BelowStackRegion.Size, PROT_READ | PROT_WRITE, MAP_ANONYMOUS | MAP_NORESERVE | MAP_PRIVATE | MAP_FIXED, -1, 0);
|
||||
|
||||
LogMan::Throw::AFmt(Alloc != MAP_FAILED, "mmap({:x},{:x}) failed", BelowStackRegion.Ptr, BelowStackRegion.Size);
|
||||
LogMan::Throw::AFmt(Alloc == BelowStackRegion.Ptr, "mmap({},{:x}) returned {} instead of {:x}", Alloc, BelowStackRegion.Ptr);
|
||||
|
||||
Regions.pop_back();
|
||||
}
|
||||
continue;
|
||||
} else {
|
||||
LogMan::Throw::AFmt(std::isalpha(c) || std::isdigit(c), "Unexpected char '{}' in ParseEnd", c);
|
||||
|
||||
+83
-102
@@ -69,11 +69,14 @@ namespace Alloc::OSAllocator {
|
||||
struct LiveVMARegion {
|
||||
ReservedVMARegion *SlabInfo;
|
||||
uint64_t FreeSpace{};
|
||||
uint64_t NumManagedPages{};
|
||||
uint32_t LastPageAllocation{};
|
||||
bool HadMunmap{};
|
||||
|
||||
// Align UsedPages so it pads to the next page.
|
||||
// Necessary to take advantage of madvise zero page pooling.
|
||||
alignas(4096) FEXCore::FlexBitSet<uint64_t> UsedPages;
|
||||
using FlexBitElementType = uint64_t;
|
||||
alignas(4096) FEXCore::FlexBitSet<FlexBitElementType> UsedPages;
|
||||
|
||||
// This returns the size of the LiveVMARegion in addition to the flex set that tracks the used data
|
||||
// The LiveVMARegion lives at the start of the VMA region which means on initialization we need to set that
|
||||
@@ -85,8 +88,8 @@ namespace Alloc::OSAllocator {
|
||||
// 0x100'0000 Pages
|
||||
// 1 bit per page for tracking means 0x20'0000 (Pages / 8) bytes of flex space
|
||||
// Which is 2MB of tracking
|
||||
uint64_t NumElements = (Size >> FHU::FEX_PAGE_SHIFT) * sizeof(uint64_t);
|
||||
return sizeof(LiveVMARegion) + FEXCore::FlexBitSet<uint64_t>::Size(NumElements);
|
||||
uint64_t NumElements = (Size >> FHU::FEX_PAGE_SHIFT) * sizeof(FlexBitElementType);
|
||||
return sizeof(LiveVMARegion) + FEXCore::FlexBitSet<FlexBitElementType>::Size(NumElements);
|
||||
}
|
||||
|
||||
static void InitializeVMARegionUsed(LiveVMARegion *Region, size_t AdditionalSize) {
|
||||
@@ -95,19 +98,21 @@ namespace Alloc::OSAllocator {
|
||||
|
||||
Region->FreeSpace = Region->SlabInfo->RegionSize - SizePlusManagedData;
|
||||
|
||||
size_t NumPages = SizePlusManagedData >> FHU::FEX_PAGE_SHIFT;
|
||||
size_t NumManagedPages = SizePlusManagedData >> FHU::FEX_PAGE_SHIFT;
|
||||
size_t ManagedSize = NumManagedPages << FHU::FEX_PAGE_SHIFT;
|
||||
|
||||
// Use madvise to set the full tracking region to zero.
|
||||
// This ensures unused pages are zero, while not having the backing pages consuming memory.
|
||||
::madvise(Region->UsedPages.Memory + (NumPages * 4096), (Region->SlabInfo->RegionSize >> FHU::FEX_PAGE_SHIFT) - (NumPages * 4096), MADV_DONTNEED);
|
||||
::madvise(Region->UsedPages.Memory + ManagedSize, (Region->SlabInfo->RegionSize >> FHU::FEX_PAGE_SHIFT) - ManagedSize, MADV_DONTNEED);
|
||||
|
||||
// Use madvise to claim WILLNEED on the beginning pages for initial state tracking.
|
||||
// Improves performance of the following MemClear by not doing a page level fault dance for data necessary to track >170TB of used pages.
|
||||
::madvise(Region->UsedPages.Memory, NumPages * 4096, MADV_WILLNEED);
|
||||
::madvise(Region->UsedPages.Memory, ManagedSize, MADV_WILLNEED);
|
||||
|
||||
// Set our reserved pages
|
||||
Region->UsedPages.MemSet(NumPages);
|
||||
Region->LastPageAllocation = NumPages;
|
||||
Region->UsedPages.MemSet(NumManagedPages);
|
||||
Region->LastPageAllocation = NumManagedPages;
|
||||
Region->NumManagedPages = NumManagedPages;
|
||||
}
|
||||
};
|
||||
|
||||
@@ -129,6 +134,7 @@ namespace Alloc::OSAllocator {
|
||||
ReservedVMARegion *ReservedRegion = *ReservedIterator;
|
||||
|
||||
ReservedRegions->erase(ReservedIterator);
|
||||
|
||||
// mprotect the new region we've allocated
|
||||
size_t SizeOfLiveRegion = FEXCore::AlignUp(LiveVMARegion::GetSizeWithFlexSet(ReservedRegion->RegionSize), FHU::FEX_PAGE_SIZE);
|
||||
size_t SizePlusManagedData = UsedSize + SizeOfLiveRegion;
|
||||
@@ -152,6 +158,9 @@ namespace Alloc::OSAllocator {
|
||||
|
||||
// 32-bit old kernel workarounds
|
||||
std::vector<FEXCore::Allocator::MemoryRegion> Steal32BitIfOldKernel();
|
||||
|
||||
void AllocateMemoryRegions(std::vector<FEXCore::Allocator::MemoryRegion> const &Ranges);
|
||||
LiveVMARegion *FindLiveRegionForAddress(uintptr_t Addr, uintptr_t AddrEnd);
|
||||
};
|
||||
|
||||
void OSAllocator_64Bit::DetermineVASize() {
|
||||
@@ -167,6 +176,42 @@ void OSAllocator_64Bit::DetermineVASize() {
|
||||
UPPER_BOUND_PAGE = UPPER_BOUND / FHU::FEX_PAGE_SIZE;
|
||||
}
|
||||
|
||||
OSAllocator_64Bit::LiveVMARegion *OSAllocator_64Bit::FindLiveRegionForAddress(uintptr_t Addr, uintptr_t AddrEnd) {
|
||||
LiveVMARegion *LiveRegion{};
|
||||
|
||||
// Check active slabs to see if we can fit this
|
||||
for (auto it = LiveRegions->begin(); it != LiveRegions->end(); ++it) {
|
||||
uintptr_t RegionBegin = (*it)->SlabInfo->Base;
|
||||
uintptr_t RegionEnd = RegionBegin + (*it)->SlabInfo->RegionSize;
|
||||
|
||||
if (Addr >= RegionBegin &&
|
||||
Addr < RegionEnd) {
|
||||
LiveRegion = *it;
|
||||
// Leave our loop
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Couldn't find an active region that fit
|
||||
// Check reserved regions
|
||||
if (!LiveRegion) {
|
||||
// Didn't have a slab that fit this range
|
||||
// Check our reserved regions to see if we have one that fits
|
||||
for (auto it = ReservedRegions->begin(); it != ReservedRegions->end(); ++it) {
|
||||
ReservedVMARegion *ReservedRegion = *it;
|
||||
uintptr_t RegionEnd = ReservedRegion->Base + ReservedRegion->RegionSize;
|
||||
if (Addr >= ReservedRegion->Base &&
|
||||
AddrEnd < RegionEnd) {
|
||||
// Found one, let's make it active
|
||||
LiveRegion = MakeRegionActive(it, 0);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return LiveRegion;
|
||||
}
|
||||
|
||||
void *OSAllocator_64Bit::Mmap(void *addr, size_t length, int prot, int flags, int fd, off_t offset) {
|
||||
if (addr != 0 &&
|
||||
addr < reinterpret_cast<void*>(LOWER_BOUND)) {
|
||||
@@ -205,41 +250,13 @@ void *OSAllocator_64Bit::Mmap(void *addr, size_t length, int prot, int flags, in
|
||||
LiveVMARegion *LiveRegion{};
|
||||
|
||||
if (Fixed || Addr != 0) {
|
||||
// Check active slabs to see if we can fit this
|
||||
for (auto it = LiveRegions->begin(); it != LiveRegions->end(); ++it) {
|
||||
uintptr_t RegionBegin = (*it)->SlabInfo->Base;
|
||||
uintptr_t RegionEnd = RegionBegin + (*it)->SlabInfo->RegionSize;
|
||||
|
||||
if (Addr >= RegionBegin &&
|
||||
Addr < RegionEnd) {
|
||||
LiveRegion = *it;
|
||||
// Leave our loop
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Couldn't find an active region that fit
|
||||
// Check reserved regions
|
||||
if (!LiveRegion) {
|
||||
// Didn't have a slab that fit this range
|
||||
// Check our reserved regions to see if we have one that fits
|
||||
for (auto it = ReservedRegions->begin(); it != ReservedRegions->end(); ++it) {
|
||||
ReservedVMARegion *ReservedRegion = *it;
|
||||
uintptr_t RegionEnd = ReservedRegion->Base + ReservedRegion->RegionSize;
|
||||
if (Addr >= ReservedRegion->Base &&
|
||||
AddrEnd < RegionEnd) {
|
||||
// Found one, let's make it active
|
||||
LiveRegion = MakeRegionActive(it, 0);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
LiveRegion = FindLiveRegionForAddress(Addr, AddrEnd);
|
||||
}
|
||||
|
||||
again:
|
||||
|
||||
auto CheckIfRangeFits = [&AllocatedOffset](LiveVMARegion *Region, uint64_t length, int prot, int flags, int fd, off_t offset, uint64_t StartingPosition = 0) -> std::pair<LiveVMARegion*, void*> {
|
||||
uint64_t AllocatedPage{};
|
||||
uint64_t AllocatedPage{~0ULL};
|
||||
uint64_t NumberOfPages = length >> FHU::FEX_PAGE_SHIFT;
|
||||
|
||||
if (Region->FreeSpace >= length) {
|
||||
@@ -249,72 +266,29 @@ void *OSAllocator_64Bit::Mmap(void *addr, size_t length, int prot, int flags, in
|
||||
: Region->LastPageAllocation;
|
||||
size_t RegionNumberOfPages = Region->SlabInfo->RegionSize >> FHU::FEX_PAGE_SHIFT;
|
||||
|
||||
// Backward scan
|
||||
// We need to do a backward scan first to fill any holes
|
||||
// Otherwise we will very quickly run out of VMA regions (65k maximum)
|
||||
for (size_t CurrentPage = LastAllocation;
|
||||
CurrentPage >= NumberOfPages;) {
|
||||
size_t Remaining = NumberOfPages;
|
||||
assert(Remaining <= CurrentPage);
|
||||
|
||||
while (Remaining) {
|
||||
if (Region->UsedPages[CurrentPage - Remaining]) {
|
||||
// Has an intersecting range
|
||||
break;
|
||||
}
|
||||
--Remaining;
|
||||
}
|
||||
if (Region->HadMunmap) {
|
||||
// Backward scan
|
||||
// We need to do a backward scan first to fill any holes
|
||||
// Otherwise we will very quickly run out of VMA regions (65k maximum)
|
||||
auto SearchResult = Region->UsedPages.BackwardScanForRange<true>(LastAllocation, NumberOfPages, Region->NumManagedPages);
|
||||
|
||||
if (Remaining) {
|
||||
// Didn't find a slab range
|
||||
CurrentPage -= Remaining;
|
||||
}
|
||||
else {
|
||||
// We have a slab range
|
||||
CurrentPage -= NumberOfPages;
|
||||
AllocatedPage = SearchResult.FoundElement;
|
||||
|
||||
// Keep scanning backwards to not introduce ANOTHER gap
|
||||
while (CurrentPage >= 1) {
|
||||
if (Region->UsedPages[CurrentPage - 1]) {
|
||||
// Found a used page, we can leave now
|
||||
break;
|
||||
}
|
||||
--CurrentPage;
|
||||
}
|
||||
AllocatedPage = CurrentPage;
|
||||
break;
|
||||
// If we didn't even have a one page free in the backward search, then unclaim HadMunmap.
|
||||
// Switching over to default forward search.
|
||||
if (SearchResult.FoundElement == ~0ULL && !SearchResult.FoundHole) {
|
||||
Region->HadMunmap = false;
|
||||
}
|
||||
}
|
||||
|
||||
// Foward Scan
|
||||
if (AllocatedPage == 0) {
|
||||
for (size_t CurrentPage = LastAllocation;
|
||||
CurrentPage < (RegionNumberOfPages - NumberOfPages);) {
|
||||
// If we have enough free space, check if we have enough free pages that are contiguous
|
||||
size_t Remaining = NumberOfPages;
|
||||
|
||||
assert((CurrentPage + Remaining - 1) < RegionNumberOfPages);
|
||||
while (Remaining) {
|
||||
if (Region->UsedPages[CurrentPage + Remaining - 1]) {
|
||||
// Has an intersecting range
|
||||
break;
|
||||
}
|
||||
--Remaining;
|
||||
}
|
||||
|
||||
if (Remaining) {
|
||||
// Didn't find a slab range
|
||||
CurrentPage += Remaining;
|
||||
}
|
||||
else {
|
||||
// We have a slab range
|
||||
AllocatedPage = CurrentPage;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (AllocatedPage == ~0ULL) {
|
||||
auto SearchResult = Region->UsedPages.ForwardScanForRange<true>(LastAllocation, NumberOfPages, RegionNumberOfPages);
|
||||
AllocatedPage = SearchResult.FoundElement;
|
||||
}
|
||||
|
||||
if (AllocatedPage) {
|
||||
if (AllocatedPage != ~0ULL) {
|
||||
AllocatedOffset = Region->SlabInfo->Base + AllocatedPage * FHU::FEX_PAGE_SIZE;
|
||||
|
||||
// We need to setup protections for this
|
||||
@@ -497,6 +471,8 @@ int OSAllocator_64Bit::Munmap(void *addr, size_t length) {
|
||||
// This will let us more quickly fill holes
|
||||
(*it)->LastPageAllocation = std::min((*it)->LastPageAllocation, SlabPageBegin);
|
||||
|
||||
(*it)->HadMunmap = true;
|
||||
|
||||
// XXX: Move region back to reserved list
|
||||
return 0;
|
||||
}
|
||||
@@ -537,12 +513,7 @@ std::vector<FEXCore::Allocator::MemoryRegion> OSAllocator_64Bit::Steal32BitIfOld
|
||||
return FEXCore::Allocator::StealMemoryRegion(LOWER_BOUND_32, UPPER_BOUND_32);
|
||||
}
|
||||
|
||||
OSAllocator_64Bit::OSAllocator_64Bit() {
|
||||
DetermineVASize();
|
||||
auto LowMem = Steal32BitIfOldKernel();
|
||||
|
||||
auto Ranges = FEXCore::Allocator::StealMemoryRegion(LOWER_BOUND, UPPER_BOUND);
|
||||
|
||||
void OSAllocator_64Bit::AllocateMemoryRegions(std::vector<FEXCore::Allocator::MemoryRegion> const &Ranges) {
|
||||
for (auto [Ptr, AllocationSize]: Ranges) {
|
||||
if (!ObjectAlloc) {
|
||||
auto MaxSize = std::min(size_t(64) * 1024 * 1024, AllocationSize);
|
||||
@@ -564,12 +535,22 @@ OSAllocator_64Bit::OSAllocator_64Bit() {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
ReservedVMARegion *Region = ObjectAlloc->new_construct<ReservedVMARegion>();
|
||||
Region->Base = reinterpret_cast<uint64_t>(Ptr);
|
||||
Region->RegionSize = AllocationSize;
|
||||
ReservedRegions->emplace_back(Region);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
OSAllocator_64Bit::OSAllocator_64Bit() {
|
||||
DetermineVASize();
|
||||
auto LowMem = Steal32BitIfOldKernel();
|
||||
|
||||
auto Ranges = FEXCore::Allocator::StealMemoryRegion(LOWER_BOUND, UPPER_BOUND);
|
||||
|
||||
AllocateMemoryRegions(Ranges);
|
||||
|
||||
FEXCore::Allocator::ReclaimMemoryRegion(LowMem);
|
||||
}
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
@@ -38,6 +39,109 @@ struct FlexBitSet final {
|
||||
memset(Memory, 0xFF, FEXCore::AlignUp(Elements / MinimumSizeBits, MinimumSizeBits));
|
||||
}
|
||||
|
||||
// Range scanning results
|
||||
struct BitsetScanResults {
|
||||
// Which element was found. ~0ULL if not found.
|
||||
size_t FoundElement;
|
||||
// During the scan, found a hole in the allocations that didn't fit.
|
||||
bool FoundHole;
|
||||
};
|
||||
|
||||
// TODO: Make {Forward,Backward}ScanForRange faster
|
||||
// Currently these functions test a single bit at a time, which is fairly costly.
|
||||
// The compiler emits a full element load per iteration, wasting a bunch of time on loads.
|
||||
// If we change these functions to have a pre-amble and post-amble to align the primary loop to the element size then this can go significantly
|
||||
// faster.
|
||||
//
|
||||
// Once the element scanning is aligned to the element size, we can then use native count leading zero(CLZ) and count trailing zero(CTZ)
|
||||
// instructions on a full element to scan uint64_t elements per loop iteration.
|
||||
|
||||
// Implementation details:
|
||||
// Template argument WantUnset
|
||||
// Used to determine if the desired range is for set or unset ranges.
|
||||
// Typically `WantUnset` should be true. Used for finding a unset range inside of a range will set elements.
|
||||
//
|
||||
// @param BeginningElement - The first element in the set to start scanning from.
|
||||
// @param ElementCount - How many elements to find a range for fitting.
|
||||
// @param MinimumElement - Minimum element in the set to search to
|
||||
//
|
||||
// @return The scan results
|
||||
template<bool WantUnset>
|
||||
BitsetScanResults BackwardScanForRange(size_t BeginningElement, size_t ElementCount, size_t MinimumElement) {
|
||||
bool FoundHole {};
|
||||
for (size_t CurrentPage = BeginningElement;
|
||||
CurrentPage >= (MinimumElement + ElementCount);) {
|
||||
size_t Remaining = ElementCount;
|
||||
LOGMAN_THROW_AA_FMT(Remaining <= CurrentPage, "Scanning less than available range");
|
||||
|
||||
while (Remaining) {
|
||||
if (this->Get(CurrentPage - Remaining) == WantUnset) {
|
||||
// Has an intersecting range
|
||||
break;
|
||||
}
|
||||
--Remaining;
|
||||
}
|
||||
|
||||
if (Remaining) {
|
||||
// If we found at least one Element hole then track that
|
||||
if (Remaining != ElementCount) {
|
||||
FoundHole = true;
|
||||
}
|
||||
|
||||
// Didn't find a slab range
|
||||
CurrentPage -= Remaining;
|
||||
}
|
||||
else {
|
||||
// We have a slab range
|
||||
return BitsetScanResults{CurrentPage - ElementCount, FoundHole};
|
||||
}
|
||||
}
|
||||
|
||||
return BitsetScanResults {~0ULL, FoundHole};
|
||||
}
|
||||
|
||||
// @param BeginningElement - The first element in the set to start scanning from.
|
||||
// @param ElementCount - How many elements to find a range for fitting.
|
||||
// @param ElementsInSet - How many elements are in the full set.
|
||||
//
|
||||
// @return The scan results
|
||||
template<bool WantUnset>
|
||||
BitsetScanResults ForwardScanForRange(size_t BeginningElement, size_t ElementCount, size_t ElementsInSet) {
|
||||
bool FoundHole {};
|
||||
|
||||
for (size_t CurrentElement = BeginningElement;
|
||||
CurrentElement < (ElementsInSet - ElementCount);) {
|
||||
// If we have enough free space, check if we have enough free pages that are contiguous
|
||||
size_t Remaining = ElementCount;
|
||||
|
||||
LOGMAN_THROW_AA_FMT((CurrentElement + Remaining - 1) < ElementsInSet, "Scanning less than available range");
|
||||
|
||||
while (Remaining) {
|
||||
if (this->Get(CurrentElement + Remaining - 1) == WantUnset) {
|
||||
// Has an intersecting range
|
||||
break;
|
||||
}
|
||||
--Remaining;
|
||||
}
|
||||
|
||||
if (Remaining) {
|
||||
// If we found at least one Element hole then track that
|
||||
if (Remaining != ElementCount) {
|
||||
FoundHole = true;
|
||||
}
|
||||
|
||||
// Didn't find a slab range
|
||||
CurrentElement += Remaining;
|
||||
}
|
||||
else {
|
||||
// We have a slab range
|
||||
return BitsetScanResults {CurrentElement, FoundHole};
|
||||
}
|
||||
}
|
||||
|
||||
return BitsetScanResults {~0ULL, FoundHole};
|
||||
}
|
||||
|
||||
// This very explicitly doesn't let you take an address
|
||||
// Is only a getter
|
||||
bool operator[](size_t Element) const {
|
||||
|
||||
+116
@@ -0,0 +1,116 @@
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <fcntl.h>
|
||||
#include <limits.h>
|
||||
#include <linux/magic.h>
|
||||
#include <string>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/vfs.h>
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#define BACKEND_OFF 0
|
||||
#define BACKEND_GPUVIS 1
|
||||
|
||||
#ifdef ENABLE_FEXCORE_PROFILER
|
||||
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
|
||||
namespace FEXCore::Profiler {
|
||||
ProfilerBlock::ProfilerBlock(std::string_view const Format)
|
||||
: DurationBegin {GetTime()}
|
||||
, Format {Format} {
|
||||
}
|
||||
|
||||
ProfilerBlock::~ProfilerBlock() {
|
||||
auto Duration = GetTime() - DurationBegin;
|
||||
TraceObject(Format, Duration);
|
||||
}
|
||||
}
|
||||
|
||||
namespace GPUVis {
|
||||
// ftrace FD for writing trace data.
|
||||
// Needs to be a raw FD since we hold this open for the entire application execution.
|
||||
static int TraceFD {-1};
|
||||
|
||||
// Need to search the paths to find the real trace path
|
||||
static std::array<char const*, 2> TraceFSDirectories {
|
||||
"/sys/kernel/tracing",
|
||||
"/sys/kernel/debug/tracing",
|
||||
};
|
||||
|
||||
static bool IsTraceFS(char const* Path) {
|
||||
struct statfs stat;
|
||||
if (statfs(Path, &stat)) {
|
||||
return false;
|
||||
}
|
||||
return stat.f_type == TRACEFS_MAGIC;
|
||||
}
|
||||
|
||||
void Init() {
|
||||
for (auto Path : TraceFSDirectories) {
|
||||
if (IsTraceFS(Path)) {
|
||||
std::string FilePath = fmt::format("{}/trace_marker", Path);
|
||||
TraceFD = open(FilePath.c_str(), O_WRONLY | O_CLOEXEC);
|
||||
if (TraceFD != -1) {
|
||||
// Opened TraceFD, early exit
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void Shutdown() {
|
||||
if (TraceFD != -1) {
|
||||
close(TraceFD);
|
||||
TraceFD = -1;
|
||||
}
|
||||
}
|
||||
|
||||
void TraceObject(std::string_view const Format, uint64_t Duration) {
|
||||
if (TraceFD != -1) {
|
||||
// Print the duration as something that began negative duration ago
|
||||
std::string Event = fmt::format("{} (lduration=-{})\n", Format, Duration);
|
||||
write(TraceFD, Event.c_str(), Event.size());
|
||||
}
|
||||
}
|
||||
|
||||
void TraceObject(std::string_view const Format) {
|
||||
if (TraceFD != -1) {
|
||||
std::string Event = fmt::format("{}\n", Format);
|
||||
write(TraceFD, Format.data(), Format.size());
|
||||
}
|
||||
}
|
||||
}
|
||||
#else
|
||||
#error Unknown profiler backend
|
||||
#endif
|
||||
#endif
|
||||
|
||||
namespace FEXCore::Profiler {
|
||||
#ifdef ENABLE_FEXCORE_PROFILER
|
||||
void Init() {
|
||||
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
|
||||
GPUVis::Init();
|
||||
#endif
|
||||
}
|
||||
|
||||
void Shutdown() {
|
||||
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
|
||||
GPUVis::Shutdown();
|
||||
#endif
|
||||
}
|
||||
|
||||
void TraceObject(std::string_view const Format, uint64_t Duration) {
|
||||
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
|
||||
GPUVis::TraceObject(Format, Duration);
|
||||
#endif
|
||||
}
|
||||
|
||||
void TraceObject(std::string_view const Format) {
|
||||
#if FEXCORE_PROFILER_BACKEND == BACKEND_GPUVIS
|
||||
GPUVis::TraceObject(Format);
|
||||
#endif
|
||||
|
||||
}
|
||||
#endif
|
||||
}
|
||||
@@ -122,6 +122,10 @@ namespace CPU {
|
||||
bool IsAddressInCodeBuffer(uintptr_t Address) const;
|
||||
|
||||
protected:
|
||||
// Max spill slot size in bytes. We need at most 32 bytes
|
||||
// to be able to handle a 256-bit vector store to a slot.
|
||||
constexpr static uint32_t MaxSpillSlotSize = 32;
|
||||
|
||||
FEXCore::Core::InternalThreadState *ThreadState;
|
||||
|
||||
size_t InitialCodeSize, MaxCodeSize;
|
||||
|
||||
+10
-3
@@ -28,9 +28,16 @@ namespace FEXCore::Core {
|
||||
|
||||
uint64_t rip; ///< Current core's RIP. May not be entirely accurate while JIT is active
|
||||
uint64_t gregs[16];
|
||||
uint16_t es, cs, ss, ds;
|
||||
uint64_t gs;
|
||||
uint64_t fs;
|
||||
// Raw segment register indexes
|
||||
uint16_t es_idx, cs_idx, ss_idx, ds_idx;
|
||||
uint16_t gs_idx, fs_idx;
|
||||
uint16_t _pad[2];
|
||||
|
||||
// Segment registers holding base addresses
|
||||
uint32_t es_cached, cs_cached, ss_cached, ds_cached;
|
||||
uint64_t gs_cached;
|
||||
uint64_t fs_cached;
|
||||
uint64_t _pad2[1];
|
||||
XMMRegs xmm;
|
||||
uint8_t flags[48];
|
||||
uint64_t mm[8][2];
|
||||
|
||||
@@ -0,0 +1,55 @@
|
||||
#pragma once
|
||||
#include <cstdint>
|
||||
#include <string_view>
|
||||
#include <time.h>
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
namespace FEXCore::Profiler {
|
||||
#ifdef ENABLE_FEXCORE_PROFILER
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void Init();
|
||||
FEX_DEFAULT_VISIBILITY void Shutdown();
|
||||
FEX_DEFAULT_VISIBILITY void TraceObject(std::string_view const Format);
|
||||
FEX_DEFAULT_VISIBILITY void TraceObject(std::string_view const Format, uint64_t Duration);
|
||||
|
||||
static inline uint64_t GetTime() {
|
||||
// We want the time in the least amount of overhead possible
|
||||
// clock_gettime will do a VDSO call with the least amount of overhead
|
||||
struct timespec ts;
|
||||
clock_gettime(CLOCK_MONOTONIC, &ts);
|
||||
return ts.tv_sec * 1'000'000'000ULL + ts.tv_nsec;
|
||||
}
|
||||
|
||||
// A class that follows scoping rules to generate a profile duration block
|
||||
class ProfilerBlock final {
|
||||
public:
|
||||
ProfilerBlock(std::string_view const Format);
|
||||
|
||||
~ProfilerBlock();
|
||||
|
||||
private:
|
||||
uint64_t DurationBegin;
|
||||
std::string_view const Format;
|
||||
};
|
||||
|
||||
#define UniqueScopeName2(name, line) name ## line
|
||||
#define UniqueScopeName(name, line) UniqueScopeName2(name, line)
|
||||
|
||||
// Declare an instantaneous profiler event.
|
||||
#define FEXCORE_PROFILE_INSTANT(name) FEXCore::Profiler::TraceObject(name)
|
||||
|
||||
// Declare a scoped profile block variable with a fixed name.
|
||||
#define FEXCORE_PROFILE_SCOPED(name) \
|
||||
FEXCore::Profiler::ProfilerBlock UniqueScopeName(ScopedBlock_, __LINE__) (name)
|
||||
|
||||
#else
|
||||
[[maybe_unused]] static void Init() {}
|
||||
[[maybe_unused]] static void Shutdown() {}
|
||||
[[maybe_unused]] static void TraceObject(std::string_view const Format) {}
|
||||
[[maybe_unused]] static void TraceObject(std::string_view const, uint64_t) {}
|
||||
|
||||
#define FEXCORE_PROFILE_INSTANT(...) do {} while(0)
|
||||
#define FEXCORE_PROFILE_SCOPED(...) do {} while(0)
|
||||
#endif
|
||||
}
|
||||
Vendored
+1
-1
Submodule External/Vulkan-Headers updated: 2b55157592...98f440ce68.
Vendored
+1
-1
Submodule External/vixl updated: 423cd04a70...dfcc56f77d.
@@ -1,74 +1,64 @@
|
||||
add_library(FEXHeaderUtils INTERFACE)
|
||||
|
||||
# Check for syscall support here
|
||||
check_cxx_source_compiles(
|
||||
"
|
||||
#include <sched.h>
|
||||
int main() {
|
||||
return ::getcpu(nullptr, nullptr);
|
||||
}"
|
||||
"
|
||||
#include <sched.h>
|
||||
int main() {
|
||||
return ::getcpu(nullptr, nullptr);
|
||||
}"
|
||||
compiles)
|
||||
if (compiles)
|
||||
message(STATUS "Has getcpu helper")
|
||||
add_definitions(-DHAS_SYSCALL_GETCPU=1)
|
||||
target_compile_definitions(FEXHeaderUtils INTERFACE HAS_SYSCALL_GETCPU=1)
|
||||
endif ()
|
||||
|
||||
check_cxx_source_compiles(
|
||||
"
|
||||
#include <unistd.h>
|
||||
int main() {
|
||||
return ::gettid();
|
||||
}"
|
||||
"
|
||||
#include <unistd.h>
|
||||
int main() {
|
||||
return ::gettid();
|
||||
}"
|
||||
compiles)
|
||||
if (compiles)
|
||||
message(STATUS "Has gettid helper")
|
||||
add_definitions(-DHAS_SYSCALL_GETTID=1)
|
||||
target_compile_definitions(FEXHeaderUtils INTERFACE HAS_SYSCALL_GETTID=1)
|
||||
endif ()
|
||||
|
||||
check_cxx_source_compiles(
|
||||
"
|
||||
#include <signal.h>
|
||||
int main() {
|
||||
return ::tgkill(0, 0, 0);
|
||||
}"
|
||||
"
|
||||
#include <signal.h>
|
||||
int main() {
|
||||
return ::tgkill(0, 0, 0);
|
||||
}"
|
||||
compiles)
|
||||
if (compiles)
|
||||
message(STATUS "Has tgkill helper")
|
||||
add_definitions(-DHAS_SYSCALL_TGKILL=1)
|
||||
target_compile_definitions(FEXHeaderUtils INTERFACE HAS_SYSCALL_TGKILL=1)
|
||||
endif ()
|
||||
|
||||
check_cxx_source_compiles(
|
||||
"
|
||||
#include <sys/stat.h>
|
||||
int main() {
|
||||
return ::statx(0, nullptr, 0, 0, nullptr);
|
||||
}"
|
||||
"
|
||||
#include <sys/stat.h>
|
||||
int main() {
|
||||
return ::statx(0, nullptr, 0, 0, nullptr);
|
||||
}"
|
||||
compiles)
|
||||
if (compiles)
|
||||
message(STATUS "Has statx helper")
|
||||
add_definitions(-DHAS_SYSCALL_STATX=1)
|
||||
target_compile_definitions(FEXHeaderUtils INTERFACE HAS_SYSCALL_STATX=1)
|
||||
endif ()
|
||||
|
||||
check_cxx_source_compiles(
|
||||
"
|
||||
#include <stdio.h>
|
||||
int main() {
|
||||
return ::renameat2(0, nullptr, 0, nullptr, 0);
|
||||
}"
|
||||
"
|
||||
#include <stdio.h>
|
||||
int main() {
|
||||
return ::renameat2(0, nullptr, 0, nullptr, 0);
|
||||
}"
|
||||
compiles)
|
||||
if (compiles)
|
||||
message(STATUS "Has renameat2 helper")
|
||||
add_definitions(-DHAS_SYSCALL_RENAMEAT2=1)
|
||||
endif ()
|
||||
|
||||
check_cxx_source_compiles(
|
||||
"
|
||||
#include <stdio.h>
|
||||
#include <syscall.h>
|
||||
int main() {
|
||||
return ::syscall(SYS_pidfd_open, ::getpid(), 0);
|
||||
}"
|
||||
compiles)
|
||||
if (compiles)
|
||||
message(STATUS "Has pidfd_open helper")
|
||||
add_definitions(-DHAS_SYSCALL_PIDFD_OPEN=1)
|
||||
target_compile_definitions(FEXHeaderUtils INTERFACE HAS_SYSCALL_RENAMEAT2=1)
|
||||
endif ()
|
||||
|
||||
target_include_directories(FEXHeaderUtils INTERFACE .)
|
||||
@@ -38,10 +38,15 @@ namespace FHU::Syscalls {
|
||||
#endif
|
||||
#endif
|
||||
|
||||
// Common syscall numbers
|
||||
#ifndef SYS_pidfd_open
|
||||
#define SYS_pidfd_open 434
|
||||
#endif
|
||||
|
||||
inline int32_t getcpu(uint32_t *cpu, uint32_t *node) {
|
||||
// Third argument is unused
|
||||
#if defined(HAS_SYSCALL_GETCPU) && HAS_SYSCALL_GETCPU
|
||||
return ::getcpu(cpu, node, nullptr);
|
||||
return ::getcpu(cpu, node);
|
||||
#else
|
||||
return ::syscall(SYS_getcpu, cpu, node, nullptr);
|
||||
#endif
|
||||
@@ -57,7 +62,7 @@ inline int32_t gettid() {
|
||||
|
||||
inline int32_t tgkill(pid_t tgid, pid_t tid, int sig) {
|
||||
#if defined(HAS_SYSCALL_GETTID) && HAS_SYSCALL_GETTID
|
||||
return ::tgkill(tggid, tid, sig);
|
||||
return ::tgkill(tgid, tid, sig);
|
||||
#else
|
||||
return ::syscall(SYS_tgkill, tgid, tid, sig);
|
||||
#endif
|
||||
@@ -65,7 +70,7 @@ inline int32_t tgkill(pid_t tgid, pid_t tid, int sig) {
|
||||
|
||||
inline int32_t statx(int dirfd, const char *pathname, int32_t flags, uint32_t mask, void *statxbuf) {
|
||||
#if defined(HAS_SYSCALL_STATX) && HAS_SYSCALL_STATX
|
||||
return ::statx(dirfd, pathname, flags, mask, statxbuf);
|
||||
return ::statx(dirfd, pathname, flags, mask, reinterpret_cast<struct statx *__restrict>(statxbuf));
|
||||
#else
|
||||
return ::syscall(SYS_statx, dirfd, pathname, flags, mask, statxbuf);
|
||||
#endif
|
||||
@@ -80,11 +85,7 @@ inline int32_t renameat2(int olddirfd, const char *oldpath, int newdirfd, const
|
||||
}
|
||||
|
||||
inline int32_t pidfd_open(pid_t pid, unsigned int flags) {
|
||||
#if defined(DHAS_SYSCALL_PIDFD_OPEN) && DHAS_SYSCALL_PIDFD_OPEN
|
||||
return ::syscall(SYS_pidfd_open, pid_t pid, unsigned int flags);
|
||||
#else
|
||||
return -1;
|
||||
#endif
|
||||
return ::syscall(SYS_pidfd_open, pid, flags);
|
||||
}
|
||||
|
||||
}
|
||||
@@ -83,6 +83,12 @@ def HashFile(file):
|
||||
|
||||
return int.from_bytes(x.digest(), "big")
|
||||
|
||||
def RemoveRootFSFolder(RootFSPath):
|
||||
print("Removing previous rootfs extraction before copying")
|
||||
shutil.rmtree(RootFSPath, ignore_errors = True)
|
||||
# Recreate the folder
|
||||
os.makedirs(RootFSPath)
|
||||
|
||||
def CheckFilesystemForFS(RootFSMountPath, RootFSPath, DistroFit):
|
||||
# Check if rootfs mount path exists
|
||||
if (not os.path.exists(RootFSMountPath) or
|
||||
@@ -105,6 +111,7 @@ def CheckFilesystemForFS(RootFSMountPath, RootFSPath, DistroFit):
|
||||
MountRootFSImagePath = RootFSMountPath + DistroFit[3]
|
||||
RootFSImagePath = RootFSPath + "/" + os.path.basename(DistroFit[3])
|
||||
NeedsExtraction = False
|
||||
PreviouslyExistingRootFS = False
|
||||
|
||||
if not os.path.exists(MountRootFSImagePath):
|
||||
print("Image {} doesn't exist".format(MountRootFSImagePath))
|
||||
@@ -113,29 +120,39 @@ def CheckFilesystemForFS(RootFSMountPath, RootFSPath, DistroFit):
|
||||
if not os.path.exists(RootFSImagePath):
|
||||
# Copy over
|
||||
print("RootFS image doesn't exist. Copying")
|
||||
shutil.copyfile(MountRootFSImagePath, RootFSImagePath)
|
||||
NeedsExtraction = True
|
||||
|
||||
# Now hash the image
|
||||
RootFSHash = HashFile(RootFSImagePath)
|
||||
if RootFSHash != DistroFit[4]:
|
||||
print("Hash {} did not match {}, copying new image".format(hex(RootFSHash), hex(DistroFit[4])))
|
||||
RemoveRootFSFolder(RootFSPath)
|
||||
shutil.copyfile(MountRootFSImagePath, RootFSImagePath)
|
||||
NeedsExtraction = True
|
||||
|
||||
# Check if the image needs to be extracted
|
||||
if not os.path.exists(RootFSPath + "/usr"):
|
||||
NeedsExtraction = True
|
||||
else:
|
||||
PreviouslyExistingRootFS = True
|
||||
|
||||
# Now hash the image
|
||||
RootFSHash = HashFile(RootFSImagePath)
|
||||
if RootFSHash != DistroFit[4]:
|
||||
print("Hash {} did not match {}, copying new image".format(hex(RootFSHash), hex(DistroFit[4])))
|
||||
|
||||
if PreviouslyExistingRootFS:
|
||||
RemoveRootFSFolder(RootFSPath)
|
||||
|
||||
shutil.copyfile(MountRootFSImagePath, RootFSImagePath)
|
||||
NeedsExtraction = True
|
||||
|
||||
if NeedsExtraction:
|
||||
print("Extracting rootfs")
|
||||
|
||||
CmdResult = subprocess.call(["unsquashfs", "-f", "-d", RootFSPath, RootFSImagePath])
|
||||
if CmdResult != 0:
|
||||
print("Couldn't extract squashfs")
|
||||
print("Couldn't extract squashfs. Removing image file to be safe")
|
||||
os.remove(RootFSImagePath)
|
||||
return False
|
||||
|
||||
if not os.path.exists(RootFSPath + "/usr"):
|
||||
print("Couldn't extract squashfs")
|
||||
print("Couldn't extract squashfs. Removing image file to be safe")
|
||||
os.remove(RootFSImagePath)
|
||||
return False
|
||||
|
||||
print("RootFS successfully checked and extracted")
|
||||
|
||||
@@ -148,13 +148,16 @@ def HandleFunctionDeclCursor(Arch, Cursor):
|
||||
elif (Child.kind == CursorKind.PARM_DECL):
|
||||
# This gives us a parameter type
|
||||
Function.Params.append(Child.type.spelling)
|
||||
elif (Child.kind == CursorKind.UNEXPOSED_ATTR):
|
||||
# Whatever you are we don't care about you
|
||||
return Arch
|
||||
elif (Child.kind == CursorKind.ASM_LABEL_ATTR):
|
||||
# Whatever you are we don't care about you
|
||||
return Arch
|
||||
elif (Child.kind == CursorKind.VISIBILITY_ATTR):
|
||||
elif (Child.kind == CursorKind.WARN_UNUSED_RESULT_ATTR):
|
||||
# Whatever you are we don't care about you
|
||||
return Arch
|
||||
elif (Child.kind == CursorKind.VISIBILITY_ATTR or
|
||||
Child.kind == CursorKind.UNEXPOSED_ATTR or
|
||||
Child.kind == CursorKind.CONST_ATTR or
|
||||
Child.kind == CursorKind.PURE_ATTR):
|
||||
pass
|
||||
else:
|
||||
logging.critical ("\tUnhandled FunctionDeclCursor {0}-{1}-{2}".format(Child.kind, Child.type.spelling, Child.spelling))
|
||||
@@ -165,7 +168,7 @@ def HandleFunctionDeclCursor(Arch, Cursor):
|
||||
|
||||
def PrintFunctionDecls():
|
||||
for Decl in FunctionDecls:
|
||||
print("fn(\"{0} {1}({2})\")".format(Decl.Ret, Decl.Name, ", ".join(Decl.Params)))
|
||||
print("template<> struct fex_gen_config<{}> {{}};".format(Decl.Name))
|
||||
|
||||
def FindClangArguments(OriginalArguments):
|
||||
AddedArguments = ["clang"]
|
||||
|
||||
@@ -10,6 +10,18 @@ import logging
|
||||
logger = logging.getLogger()
|
||||
logger.setLevel(logging.WARNING)
|
||||
|
||||
# These defines are temporarily defined since python3-clang doesn't yet support these.
|
||||
# Once this tool gets switched over to C++ then this won't be an issue.
|
||||
|
||||
# Expression that references a C++20 concept.
|
||||
CursorKind.CONCEPTSPECIALIZATIONEXPR = CursorKind(153),
|
||||
|
||||
# C++2a std::bit_cast expression.
|
||||
CursorKind.BUILTINBITCASTEXPR = CursorKind(280)
|
||||
|
||||
# a concept declaration.
|
||||
CursorKind.CONCEPTDECL = CursorKind(604),
|
||||
|
||||
@dataclass
|
||||
class TypeDefinition:
|
||||
TYPE_UNKNOWN = 0
|
||||
@@ -268,7 +280,7 @@ def HandleTypeDefDeclCursor(Arch, Cursor):
|
||||
if (len(TypeDefName) != 0):
|
||||
HandleTypeDefDecl(Arch, Cursor, TypeDefName)
|
||||
|
||||
# Append namespace
|
||||
# Append namespace
|
||||
Arch.NamespaceScope.append(TypeDefName)
|
||||
SetNamespace(Arch)
|
||||
|
||||
@@ -404,19 +416,20 @@ def HandleCursor(Arch, Cursor):
|
||||
return
|
||||
|
||||
for Child in Cursor.get_children():
|
||||
if (Child.kind == CursorKind.TRANSLATION_UNIT):
|
||||
kind = Child.kind
|
||||
if (kind == CursorKind.TRANSLATION_UNIT):
|
||||
Arch = HandleCursor(Arch, Child)
|
||||
elif (Child.kind == CursorKind.FIELD_DECL):
|
||||
elif (kind == CursorKind.FIELD_DECL):
|
||||
pass
|
||||
elif (Child.kind == CursorKind.UNION_DECL):
|
||||
elif (kind == CursorKind.UNION_DECL):
|
||||
Arch = HandleUnionDeclCursor(Arch, Child)
|
||||
elif (Child.kind == CursorKind.STRUCT_DECL):
|
||||
elif (kind == CursorKind.STRUCT_DECL):
|
||||
Arch = HandleStructDeclCursor(Arch, Child)
|
||||
elif (Child.kind == CursorKind.TYPEDEF_DECL):
|
||||
elif (kind == CursorKind.TYPEDEF_DECL):
|
||||
Arch = HandleTypeDefDeclCursor(Arch, Child)
|
||||
elif (Child.kind == CursorKind.VAR_DECL):
|
||||
elif (kind == CursorKind.VAR_DECL):
|
||||
Arch = HandleVarDeclCursor(Arch, Child)
|
||||
elif (Child.kind == CursorKind.NAMESPACE):
|
||||
elif (kind == CursorKind.NAMESPACE):
|
||||
# Append namespace
|
||||
Arch.NamespaceScope.append(Child.spelling)
|
||||
SetNamespace(Arch)
|
||||
@@ -427,7 +440,7 @@ def HandleCursor(Arch, Cursor):
|
||||
# Pop namespace off
|
||||
Arch.NamespaceScope.pop()
|
||||
SetNamespace(Arch)
|
||||
elif (Child.kind == CursorKind.TYPE_REF):
|
||||
elif (kind == CursorKind.TYPE_REF):
|
||||
# Safe to pass on
|
||||
pass
|
||||
else:
|
||||
@@ -638,25 +651,21 @@ def main():
|
||||
BaseArgs.append(sys.argv[ArgIndex])
|
||||
|
||||
args_x86_32 = [
|
||||
"-I/usr/i686-linux-gnu/include/c++/10/i686-linux-gnu/",
|
||||
"-I/usr/i686-linux-gnu/include/",
|
||||
"-I/usr/i686-linux-gnu/include",
|
||||
"-O2",
|
||||
"-m32",
|
||||
"--target=i686-linux-unknown",
|
||||
]
|
||||
|
||||
args_x86_64 = [
|
||||
"-I/usr/include/x86_64-linux-gnu",
|
||||
"-I/usr/x86_64-linux-gnu/include/c++/10/x86_64-linux-gnu/",
|
||||
"-I/usr/x86_64-linux-gnu/include/",
|
||||
"-I/usr/x86_64-linux-gnu/include",
|
||||
"-O2",
|
||||
"--target=x86_64-linux-unknown",
|
||||
"-D_M_X86_64",
|
||||
]
|
||||
|
||||
args_aarch64 = [
|
||||
"-I/usr/aarch64-linux-gnu/include/c++/10/aarch64-linux-gnu/",
|
||||
"-I/usr/aarch64-linux-gnu/include/",
|
||||
"-I/usr/aarch64-linux-gnu/include",
|
||||
"-O2",
|
||||
"--target=aarch64-linux-unknown",
|
||||
"-D_M_ARM_64",
|
||||
|
||||
@@ -3,44 +3,55 @@ import os
|
||||
import sys
|
||||
import subprocess
|
||||
|
||||
# Args: <Known Failures file> <ExpectedOutputsFile> <DisabledTestsFile> <TestName> <FexExecutable> <FexArgs>...
|
||||
def LoadTestsFile(File):
|
||||
Dict = {}
|
||||
if not os.path.exists(File):
|
||||
return Dict
|
||||
|
||||
with open(File) as dtf:
|
||||
for line in dtf:
|
||||
test = line.split("#")[0].strip() # remove comments and empty spaces
|
||||
if len(test) > 0:
|
||||
Dict[test] = 1
|
||||
|
||||
return Dict
|
||||
|
||||
def LoadTestsFileResults(File):
|
||||
Dict = {}
|
||||
if not os.path.exists(File):
|
||||
return Dict
|
||||
|
||||
with open(File) as dtf:
|
||||
for line in dtf:
|
||||
test = line.split("#")[0].strip() # remove comments and empty spaces
|
||||
if len(test) > 0:
|
||||
parts = line.split(" ")
|
||||
Dict[parts[0]] = int(parts[1])
|
||||
|
||||
return Dict
|
||||
|
||||
|
||||
# Args: <Known Failures file> <ExpectedOutputsFile> <DisabledTestsFile> <FlakeTestsFile> <TestName> <Mode> <FexExecutable> <FexArgs>...
|
||||
|
||||
# fexargs should also include the test executable
|
||||
|
||||
if (len(sys.argv) < 6):
|
||||
if (len(sys.argv) < 7):
|
||||
sys.exit()
|
||||
|
||||
known_failures_file = sys.argv[1]
|
||||
expected_output_file = sys.argv[2]
|
||||
disabled_tests_file = sys.argv[3]
|
||||
test_name = sys.argv[4]
|
||||
mode = sys.argv[5]
|
||||
fexecutable = sys.argv[6]
|
||||
flake_tests_file = sys.argv[4]
|
||||
test_name = sys.argv[5]
|
||||
mode = sys.argv[6]
|
||||
fexecutable = sys.argv[7]
|
||||
StartingFEXArgsOffset = 8
|
||||
|
||||
known_failures = { }
|
||||
expected_output = { }
|
||||
disabled_tests = { }
|
||||
|
||||
# Open the known failures file and add it to a dictionary
|
||||
with open(known_failures_file) as kff:
|
||||
for line in kff:
|
||||
test = line.split("#")[0].strip() # remove comments and empty spaces
|
||||
if len(test) > 0:
|
||||
known_failures[test] = 1
|
||||
|
||||
# Open expected outputs and add it to dictionary
|
||||
with open(expected_output_file) as eof:
|
||||
for line in eof:
|
||||
line = test = line.split("#")[0].strip() # remove comments and empty spaces
|
||||
if len(line) > 0:
|
||||
parts = line.split(" ")
|
||||
expected_output[parts[0]] = int(parts[1])
|
||||
|
||||
with open(disabled_tests_file) as dtf:
|
||||
for line in dtf:
|
||||
test = line.split("#")[0].strip() # remove comments and empty spaces
|
||||
if len(test) > 0:
|
||||
disabled_tests[test] = 1
|
||||
# Open test expected information files and load in to dictionaries.
|
||||
known_failures = LoadTestsFile(known_failures_file)
|
||||
expected_output = LoadTestsFileResults(expected_output_file)
|
||||
disabled_tests = LoadTestsFile(disabled_tests_file)
|
||||
flake_tests = LoadTestsFile(flake_tests_file)
|
||||
|
||||
# run with timeout to avoid locking up
|
||||
RunnerArgs = []
|
||||
@@ -54,25 +65,37 @@ if (mode == "guest"):
|
||||
RunnerArgs.append(ROOTFS_ENV)
|
||||
|
||||
# Add the rest of the arguments
|
||||
for i in range(len(sys.argv) - 7):
|
||||
RunnerArgs.append(sys.argv[7 + i])
|
||||
for i in range(len(sys.argv) - StartingFEXArgsOffset):
|
||||
RunnerArgs.append(sys.argv[StartingFEXArgsOffset + i])
|
||||
|
||||
#print(RunnerArgs)
|
||||
|
||||
ResultCode = 0
|
||||
|
||||
# Handle flakes
|
||||
TryCount = 1
|
||||
if (flake_tests.get(test_name)):
|
||||
TryCount = 5
|
||||
|
||||
if (disabled_tests.get(test_name)):
|
||||
ResultCode = -73
|
||||
else:
|
||||
# Run the test and wait for it to end to get the result
|
||||
Process = subprocess.Popen(RunnerArgs)
|
||||
Process.wait()
|
||||
ResultCode = Process.returncode
|
||||
|
||||
# expect zero by default
|
||||
if (not test_name in expected_output):
|
||||
expected_output[test_name] = 0
|
||||
|
||||
if ResultCode == 0:
|
||||
for Try in range(TryCount):
|
||||
# Run the test and wait for it to end to get the result
|
||||
print(RunnerArgs)
|
||||
Process = subprocess.Popen(RunnerArgs)
|
||||
Process.wait()
|
||||
ResultCode = Process.returncode
|
||||
|
||||
# Break if the expected output is the result code
|
||||
if (expected_output[test_name] == ResultCode):
|
||||
break
|
||||
|
||||
if (expected_output[test_name] != ResultCode):
|
||||
if (test_name in expected_output):
|
||||
print("test failed, expected is", expected_output[test_name], "but got", ResultCode)
|
||||
|
||||
@@ -4,7 +4,7 @@ import subprocess
|
||||
import os.path
|
||||
from os import path
|
||||
|
||||
# Args: <Known Failures file> <DisabledTestsFile> <DisabledTestsTypeFile> <DisabledTestsRunnerFile> <TestName> <Test Harness Executable> <Args>...
|
||||
# Args: <Known Failures file> <Known Failures Type File> <DisabledTestsFile> <DisabledTestsTypeFile> <DisabledTestsRunnerFile> <TestName> <Test Harness Executable> <Args>...
|
||||
|
||||
if (len(sys.argv) < 7):
|
||||
sys.exit()
|
||||
@@ -12,19 +12,25 @@ if (len(sys.argv) < 7):
|
||||
known_failures = {}
|
||||
disabled_tests = {}
|
||||
known_failures_file = sys.argv[1]
|
||||
disabled_tests_file = sys.argv[2]
|
||||
disabled_tests_type_file = sys.argv[3]
|
||||
disabled_tests_runner_file = sys.argv[4]
|
||||
known_failures_type_file = sys.argv[2]
|
||||
disabled_tests_file = sys.argv[3]
|
||||
disabled_tests_type_file = sys.argv[4]
|
||||
disabled_tests_runner_file = sys.argv[5]
|
||||
|
||||
current_test = sys.argv[5]
|
||||
runner = sys.argv[6]
|
||||
args_start_index = 7
|
||||
current_test = sys.argv[6]
|
||||
runner = sys.argv[7]
|
||||
args_start_index = 8
|
||||
|
||||
# Open the known failures file and add it to a dictionary
|
||||
with open(known_failures_file) as kff:
|
||||
for line in kff:
|
||||
known_failures[line.strip()] = 1
|
||||
|
||||
if path.exists(known_failures_type_file):
|
||||
with open(known_failures_type_file) as dtf:
|
||||
for line in dtf:
|
||||
known_failures[line.strip()] = 1
|
||||
|
||||
with open(disabled_tests_file) as dtf:
|
||||
for line in dtf:
|
||||
disabled_tests[line.strip()] = 1
|
||||
|
||||
@@ -8,6 +8,6 @@ set(SRCS
|
||||
StringUtil.cpp)
|
||||
|
||||
add_library(${NAME} STATIC ${SRCS})
|
||||
target_link_libraries(${NAME} FEXCore_Base cpp-optparse json-maker)
|
||||
target_link_libraries(${NAME} FEXCore_Base cpp-optparse json-maker FEXHeaderUtils)
|
||||
target_include_directories(${NAME} PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/External/cpp-optparse/)
|
||||
target_include_directories(${NAME} PRIVATE ${CMAKE_BINARY_DIR}/generated)
|
||||
@@ -51,10 +51,10 @@ if(TERMUX_BUILD)
|
||||
)
|
||||
|
||||
install(
|
||||
CODE "MESSAGE(\"-- Installing: ${CMAKE_INSTALL_PREFIX}/bin/FEXInterpreter\")"
|
||||
CODE "MESSAGE(\"-- Installing: $ENV{DESTDIR}${CMAKE_INSTALL_PREFIX}/bin/FEXInterpreter\")"
|
||||
CODE "
|
||||
EXECUTE_PROCESS(COMMAND cp FEXLoader FEXInterpreter
|
||||
WORKING_DIRECTORY ${CMAKE_INSTALL_PREFIX}/bin/
|
||||
WORKING_DIRECTORY $ENV{DESTDIR}${CMAKE_INSTALL_PREFIX}/bin/
|
||||
)"
|
||||
)
|
||||
else()
|
||||
@@ -64,12 +64,20 @@ else()
|
||||
)
|
||||
|
||||
install(
|
||||
CODE "MESSAGE(\"-- Installing: ${CMAKE_INSTALL_PREFIX}/bin/FEXInterpreter\")"
|
||||
CODE "MESSAGE(\"-- Installing: $ENV{DESTDIR}${CMAKE_INSTALL_PREFIX}/bin/FEXInterpreter\")"
|
||||
CODE "
|
||||
EXECUTE_PROCESS(COMMAND ln -f FEXLoader FEXInterpreter
|
||||
WORKING_DIRECTORY ${CMAKE_INSTALL_PREFIX}/bin/
|
||||
WORKING_DIRECTORY $ENV{DESTDIR}${CMAKE_INSTALL_PREFIX}/bin/
|
||||
)"
|
||||
)
|
||||
|
||||
if(TARGET uninstall)
|
||||
add_custom_target(uninstall_FEXInterpreter
|
||||
COMMAND "rm" "$ENV{DESTDIR}${CMAKE_INSTALL_PREFIX}/bin/FEXInterpreter"
|
||||
)
|
||||
|
||||
add_dependencies(uninstall uninstall_FEXInterpreter)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
install(PROGRAMS "${PROJECT_SOURCE_DIR}/Scripts/FEXUpdateAOTIRCache.sh" DESTINATION bin RENAME FEXUpdateAOTIRCache)
|
||||
@@ -101,6 +109,17 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
echo "binfmt_misc FEX-x86_64 installed"
|
||||
)
|
||||
if(TARGET uninstall)
|
||||
add_custom_target(uninstall_binfmt_misc_32
|
||||
COMMAND update-binfmts --unimport FEX-x86 || (exit 0)
|
||||
)
|
||||
add_custom_target(uninstall_binfmt_misc_64
|
||||
COMMAND update-binfmts --unimport FEX-x86_64 || (exit 0)
|
||||
)
|
||||
|
||||
add_dependencies(uninstall uninstall_binfmt_misc_32)
|
||||
add_dependencies(uninstall uninstall_binfmt_misc_64)
|
||||
endif()
|
||||
else()
|
||||
# In the case of update-binfmts not being available (Arch for example) then we need to install manually
|
||||
add_custom_target(binfmt_misc_32
|
||||
@@ -129,6 +148,20 @@ if (CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64")
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
echo "binfmt_misc FEX-x86_64 installed"
|
||||
)
|
||||
|
||||
if(TARGET uninstall)
|
||||
add_custom_target(uninstall_binfmt_misc_32
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
echo -1 > /proc/sys/fs/binfmt_misc/FEX-x86 || (exit 0)
|
||||
)
|
||||
add_custom_target(uninstall_binfmt_misc_64
|
||||
COMMAND ${CMAKE_COMMAND} -E
|
||||
echo -1 > /proc/sys/fs/binfmt_misc/FEX-x86_64 || (exit 0)
|
||||
)
|
||||
|
||||
add_dependencies(uninstall uninstall_binfmt_misc_32)
|
||||
add_dependencies(uninstall uninstall_binfmt_misc_64)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
add_custom_target(binfmt_misc
|
||||
|
||||
+231
-69
@@ -3,6 +3,7 @@
|
||||
|
||||
#include "Common/Config.h"
|
||||
#include "Common/FDUtils.h"
|
||||
#include "FEXCore/Utils/Allocator.h"
|
||||
#include "Tests/LinuxSyscalls/Syscalls.h"
|
||||
#include "Linux/Utils/ELFParser.h"
|
||||
#include "Linux/Utils/ELFSymbolDatabase.h"
|
||||
@@ -13,14 +14,16 @@
|
||||
#include <cstring>
|
||||
#include <filesystem>
|
||||
#include <fstream>
|
||||
#include <list>
|
||||
#include <random>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include <FEXCore/Core/CodeLoader.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
#include <FEXHeaderUtils/TypeDefines.h>
|
||||
|
||||
#include <elf.h>
|
||||
#include <fcntl.h>
|
||||
@@ -103,7 +106,7 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
|
||||
}
|
||||
|
||||
template <typename TMap, typename TUnmap>
|
||||
std::optional<uintptr_t> LoadElfFile(ELFParser& Elf, uintptr_t *BrkBase, TMap Mapper, TUnmap Unmapper) {
|
||||
std::optional<uintptr_t> LoadElfFile(ELFParser& Elf, uintptr_t *BrkBase, TMap Mapper, TUnmap Unmapper, uint64_t LoadHint = 0) {
|
||||
|
||||
uintptr_t LoadBase = 0;
|
||||
|
||||
@@ -114,14 +117,11 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
|
||||
if (Elf.ehdr.e_type == ET_DYN) {
|
||||
// needs base address
|
||||
auto TotalSize = CalculateTotalElfSize(Elf.phdrs) + (BrkBase ? BRK_SIZE : 0);
|
||||
LoadBase = (uintptr_t)Mapper(0, TotalSize, PROT_NONE, MAP_ANONYMOUS | MAP_PRIVATE, -1, 0);
|
||||
LoadBase = (uintptr_t)Mapper(reinterpret_cast<void*>(LoadHint), TotalSize, PROT_NONE, MAP_ANONYMOUS | MAP_PRIVATE, -1, 0);
|
||||
if ((void*)LoadBase == MAP_FAILED) {
|
||||
return {};
|
||||
}
|
||||
|
||||
if (Unmapper((void*)LoadBase, TotalSize) == -1) {
|
||||
return {};
|
||||
}
|
||||
//fprintf(stderr, "elf %d: %lx-%lx\n", Elf.fd, LoadBase, LoadBase + TotalSize);
|
||||
if (BrkBase) {
|
||||
*BrkBase = LoadBase + (TotalSize - BRK_SIZE);
|
||||
@@ -132,10 +132,10 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
|
||||
if (Header.p_type != PT_LOAD)
|
||||
continue;
|
||||
|
||||
int MapProt = MapFlags(Header);
|
||||
int MapType = MAP_PRIVATE | MAP_DENYWRITE | MAP_FIXED_NOREPLACE;
|
||||
int MapProt = MapFlags(Header);
|
||||
int MapType = MAP_PRIVATE | MAP_DENYWRITE | MAP_FIXED;
|
||||
|
||||
if (!MapFile(Elf, LoadBase, Header, MapProt, MapType, Mapper)) {
|
||||
if (!MapFile(Elf, LoadBase, Header, MapProt, MapType, Mapper)) {
|
||||
return {};
|
||||
}
|
||||
|
||||
@@ -353,40 +353,79 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
|
||||
//
|
||||
// This is still technically a memory leak if the stack grows, but since the primary thread's stack only gets destroyed on process close, this is
|
||||
// fine.
|
||||
StackPointer = reinterpret_cast<uintptr_t>(Mapper(nullptr, StackSize(), PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS | MAP_STACK | MAP_GROWSDOWN, -1, 0));
|
||||
|
||||
// Stacks need to be allocated at the hint location just like on a real x86 system.
|
||||
// These are 128MB regions on both x86-64 and x86.
|
||||
//
|
||||
// These are required to be in the correct location taking up the appropriate 128MB of space, otherwise the wine preloader crashes FEX.
|
||||
// This is due to the wine-preloader hardcoding addresses [0x7FFFFE000000 - 0x7FFFFFFF0000) as a top-down
|
||||
// allocation region. They use mmap with MAP_FIXED, ignoring any previously mapped area at that location and overwriting it.
|
||||
// Wine-preloader is expecting to allocate 32MB out of the total 128MB stack space in this case. Leaving 96MB for the application.
|
||||
//
|
||||
// If FEX doesn't allocate the stack in this region (nullptr mmap hint) then later allocations that FEX does will /eventually/
|
||||
// end up inside of this address space that wine allocates. This usually ends up being a JIT CodeBuffer, which zeroes the memory and faults with a
|
||||
// SIGILL.
|
||||
//
|
||||
// On the upside, this more accurately emulates how the kernel allocates stack space for the application when hinting at the location.
|
||||
//
|
||||
void* StackPointerBase{};
|
||||
uint64_t StackHint = Is64BitMode() ? STACK_HINT_64 : STACK_HINT_32;
|
||||
|
||||
// Allocate the base of the full 128MB stack range.
|
||||
StackPointerBase = Mapper(reinterpret_cast<void*>(StackHint), FULL_STACK_SIZE, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS | MAP_STACK | MAP_GROWSDOWN | MAP_NORESERVE, -1, 0);
|
||||
|
||||
if (StackPointerBase == reinterpret_cast<void*>(~0ULL)) {
|
||||
LogMan::Msg::EFmt("Allocating stack failed");
|
||||
return false;
|
||||
}
|
||||
|
||||
// Allocate with permissions the 8MB of regular stack size.
|
||||
StackPointer = reinterpret_cast<uintptr_t>(Mapper(
|
||||
reinterpret_cast<void*>(reinterpret_cast<uint64_t>(StackPointerBase) + FULL_STACK_SIZE - StackSize()),
|
||||
StackSize(), PROT_READ | PROT_WRITE, MAP_FIXED | MAP_PRIVATE | MAP_ANONYMOUS | MAP_STACK | MAP_GROWSDOWN, -1, 0));
|
||||
|
||||
if (StackPointer == ~0ULL) {
|
||||
LogMan::Msg::EFmt("Allocating stack failed");
|
||||
return false;
|
||||
}
|
||||
|
||||
// load the main elf
|
||||
|
||||
uintptr_t BrkBase = 0;
|
||||
|
||||
uintptr_t LoadBase = 0;
|
||||
|
||||
if (auto elf = LoadElfFile(MainElf, &BrkBase, Mapper, Unmapper)) {
|
||||
LoadBase = *elf;
|
||||
if (MainElf.ehdr.e_type == ET_DYN) {
|
||||
BaseOffset = LoadBase;
|
||||
}
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Failed to load elf file");
|
||||
return false;
|
||||
}
|
||||
|
||||
// XXX Randomise brk?
|
||||
|
||||
BrkStart = (uint64_t)Mapper((void*)BrkBase, BRK_SIZE, PROT_READ | PROT_WRITE, MAP_ANONYMOUS | MAP_PRIVATE | MAP_FIXED_NOREPLACE, -1, 0);
|
||||
|
||||
if ((void*)BrkStart == MAP_FAILED) {
|
||||
LogMan::Msg::EFmt("Failed to allocate BRK @ {:x}, {}\n", BrkBase, errno);
|
||||
return false;
|
||||
}
|
||||
|
||||
MainElfBase = LoadBase + MainElf.phdrs.front().p_vaddr - MainElf.phdrs.front().p_offset;
|
||||
MainElfEntrypoint = LoadBase + MainElf.ehdr.e_entry;
|
||||
// Load the interpreter ELF first.
|
||||
// This allows the top-down allocation of the kernel to put this at the top of the VA space.
|
||||
// This matches behaviour of native execution more closely.
|
||||
//
|
||||
// eg:
|
||||
// 555555554000-555555558000 r--p 00000000 103:0a 1311400 /usr/bin/ls
|
||||
// 555555558000-55555556c000 r-xp 00004000 103:0a 1311400 /usr/bin/ls
|
||||
// 55555556c000-555555574000 r--p 00018000 103:0a 1311400 /usr/bin/ls
|
||||
// 555555575000-555555577000 rw-p 00020000 103:0a 1311400 /usr/bin/ls
|
||||
// 555555577000-555555578000 rw-p 00000000 00:00 0 [heap]
|
||||
// 7ffff7fbb000-7ffff7fbd000 rw-p 00000000 00:00 0
|
||||
// 7ffff7fbd000-7ffff7fc1000 r--p 00000000 00:00 0 [vvar]
|
||||
// 7ffff7fc1000-7ffff7fc3000 r-xp 00000000 00:00 0 [vdso]
|
||||
// 7ffff7fc3000-7ffff7fc5000 r--p 00000000 103:0a 1316948 /usr/lib/x86_64-linux-gnu/ld-linux-x86-64.so.2
|
||||
// 7ffff7fc5000-7ffff7fef000 r-xp 00002000 103:0a 1316948 /usr/lib/x86_64-linux-gnu/ld-linux-x86-64.so.2
|
||||
// 7ffff7fef000-7ffff7ffa000 r--p 0002c000 103:0a 1316948 /usr/lib/x86_64-linux-gnu/ld-linux-x86-64.so.2
|
||||
// 7ffff7ffb000-7ffff7fff000 rw-p 00037000 103:0a 1316948 /usr/lib/x86_64-linux-gnu/ld-linux-x86-64.so.2
|
||||
// 7ffffffdd000-7ffffffff000 rw-p 00000000 00:00 0 [stack]
|
||||
// ffffffffff600000-ffffffffff601000 --xp 00000000 00:00 0 [vsyscall]
|
||||
//
|
||||
// ARM:
|
||||
// 55ccaf8b1000-55ccaf8b5000 r--p 00000000 00:2a 4 /tmp/.FEXMount178532-oiFrTF/usr/bin/ls
|
||||
// 55ccaf8b5000-55ccaf8c9000 r-xp 00004000 00:2a 4 /tmp/.FEXMount178532-oiFrTF/usr/bin/ls
|
||||
// 55ccaf8c9000-55ccaf8d1000 r--p 00018000 00:2a 4 /tmp/.FEXMount178532-oiFrTF/usr/bin/ls
|
||||
// 55ccaf8d1000-55ccaf8d2000 ---p 00000000 00:00 0
|
||||
// 55ccaf8d2000-55ccaf8d4000 rw-p 00020000 00:2a 4 /tmp/.FEXMount178532-oiFrTF/usr/bin/ls
|
||||
// 55ccaf8d4000-55ccb00d5000 rw-p 00000000 00:00 0
|
||||
// <... Snip of misc allocations ...>
|
||||
// 7fffff6c2000-7fffff6c4000 r--p 00000000 00:2a 22 /tmp/.FEXMount178532-oiFrTF/usr/lib/x86_64-linux-gnu/ld-linux-x86-64.so.2
|
||||
// 7fffff6c4000-7fffff6ee000 r-xp 00002000 00:2a 22 /tmp/.FEXMount178532-oiFrTF/usr/lib/x86_64-linux-gnu/ld-linux-x86-64.so.2
|
||||
// 7fffff6ee000-7fffff6f9000 r--p 0002c000 00:2a 22 /tmp/.FEXMount178532-oiFrTF/usr/lib/x86_64-linux-gnu/ld-linux-x86-64.so.2
|
||||
// 7fffff6f9000-7fffff6fa000 ---p 00000000 00:00 0
|
||||
// 7fffff6fa000-7fffff6fe000 rw-p 00037000 00:2a 22 /tmp/.FEXMount178532-oiFrTF/usr/lib/x86_64-linux-gnu/ld-linux-x86-64.so.2
|
||||
// 7fffff7fe000-7fffffffe000 rw-p 00000000 00:00 0
|
||||
// 7fffffffe000-7ffffffff000 r--p 00000000 08:82 7082611 /usr/share/fex-emu/GuestThunks/libVDSO-guest.so
|
||||
// 7ffffffff000-800000000000 rw-p 00000000 00:00 0
|
||||
uint64_t ELFLoadHint = 0;
|
||||
|
||||
if (!MainElf.InterpreterElf.empty()) {
|
||||
uint64_t InterpLoadBase = 0;
|
||||
@@ -399,7 +438,83 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
|
||||
|
||||
InterpeterElfBase = InterpLoadBase + InterpElf.phdrs.front().p_vaddr - InterpElf.phdrs.front().p_offset;
|
||||
Entrypoint = InterpLoadBase + InterpElf.ehdr.e_entry;
|
||||
|
||||
// If the ELF has an interpreter and is dynamic then we should provide a address hint for loading.
|
||||
// The kernel calculates this `load_bias` by dividing the task size by three then multiplying by two.
|
||||
// It then also offsets by a random number for ASLR purposes.
|
||||
//
|
||||
// Random number that gets added to the base needs to be in the number of bits (multiplied by pages):
|
||||
// 64-bit: [28, 32] bits
|
||||
// 32-bit: [8, 16] bits
|
||||
// By default the /minimum/ number of bits is used here.
|
||||
constexpr uint64_t TASK_SIZE_64 = (1ULL << 47);
|
||||
constexpr uint64_t TASK_SIZE_32 = (1ULL << 32);
|
||||
if (Is64BitMode()) {
|
||||
// Ensure that if we are running on a 36-bit VA system, we don't try hinting that an ELF should
|
||||
// live way outside the VA space.
|
||||
uint64_t HostVASize = 1ULL << FEXCore::Allocator::DetermineVASize();
|
||||
ELFLoadHint = std::min(HostVASize, TASK_SIZE_64) / 3 * 2;
|
||||
}
|
||||
else {
|
||||
ELFLoadHint = TASK_SIZE_32 / 3 * 2;
|
||||
}
|
||||
#define ASLR_LOAD
|
||||
#ifdef ASLR_LOAD
|
||||
// Only enable ASLR randomization if the personality has it enabled.
|
||||
uint32_t Personality = personality(~0ULL);
|
||||
bool NoRandomize = (Personality & ADDR_NO_RANDOMIZE) == ADDR_NO_RANDOMIZE;
|
||||
|
||||
if (!NoRandomize) {
|
||||
constexpr uint64_t ASLR_BITS_64 = 28;
|
||||
constexpr uint64_t ASLR_BITS_32 = 8;
|
||||
std::random_device rd;
|
||||
std::uniform_int_distribution<uint64_t> d(0);
|
||||
uint64_t ASLR_Offset = d(rd);
|
||||
|
||||
if (Is64BitMode()) {
|
||||
ASLR_Offset &= (1ULL << ASLR_BITS_64) - 1;
|
||||
}
|
||||
else {
|
||||
ASLR_Offset &= (1ULL << ASLR_BITS_32) - 1;
|
||||
}
|
||||
|
||||
ASLR_Offset <<= FHU::FEX_PAGE_SHIFT;
|
||||
ELFLoadHint += ASLR_Offset;
|
||||
}
|
||||
#endif
|
||||
// Align the mapping
|
||||
ELFLoadHint &= FHU::FEX_PAGE_MASK;
|
||||
}
|
||||
|
||||
// load the main elf
|
||||
|
||||
uintptr_t BrkBase = 0;
|
||||
|
||||
uintptr_t LoadBase = 0;
|
||||
|
||||
if (auto elf = LoadElfFile(MainElf, &BrkBase, Mapper, Unmapper, ELFLoadHint)) {
|
||||
LoadBase = *elf;
|
||||
if (MainElf.ehdr.e_type == ET_DYN) {
|
||||
BaseOffset = LoadBase;
|
||||
}
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Failed to load elf file");
|
||||
return false;
|
||||
}
|
||||
|
||||
// XXX Randomise brk?
|
||||
|
||||
BrkStart = (uint64_t)Mapper((void*)BrkBase, BRK_SIZE, PROT_READ | PROT_WRITE, MAP_ANONYMOUS | MAP_PRIVATE | MAP_FIXED, -1, 0);
|
||||
|
||||
if ((void*)BrkStart == MAP_FAILED) {
|
||||
LogMan::Msg::EFmt("Failed to allocate BRK @ {:x}, {}\n", BrkBase, errno);
|
||||
return false;
|
||||
}
|
||||
|
||||
MainElfBase = LoadBase + MainElf.phdrs.front().p_vaddr - MainElf.phdrs.front().p_offset;
|
||||
MainElfEntrypoint = LoadBase + MainElf.ehdr.e_entry;
|
||||
|
||||
if (MainElf.InterpreterElf.empty()) {
|
||||
InterpeterElfBase = 0;
|
||||
Entrypoint = MainElfEntrypoint;
|
||||
}
|
||||
@@ -413,33 +528,31 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
|
||||
AuxVariables.emplace_back(auxv_t{14, getauxval(AT_EGID)}); // AT_EGID
|
||||
AuxVariables.emplace_back(auxv_t{17, getauxval(AT_CLKTCK)}); // AT_CLKTIK
|
||||
AuxVariables.emplace_back(auxv_t{6, 0x1000}); // AT_PAGESIZE
|
||||
AuxVariables.emplace_back(auxv_t{25, ~0ULL}); // AT_RANDOM
|
||||
AuxRandom = &AuxVariables.emplace_back(auxv_t{25, ~0ULL}); // AT_RANDOM
|
||||
AuxVariables.emplace_back(auxv_t{23, 0}); // AT_SECURE
|
||||
AuxVariables.emplace_back(auxv_t{8, 0}); // AT_FLAGS
|
||||
AuxVariables.emplace_back(auxv_t{5, MainElf.phdrs.size()}); // AT_PHNUM
|
||||
AuxVariables.emplace_back(auxv_t{16, HWCap}); // AT_HWCAP
|
||||
AuxVariables.emplace_back(auxv_t{26, HWCap2}); // AT_HWCAP2
|
||||
AuxPlatform = &AuxVariables.emplace_back(auxv_t{24, ~0ULL}); // AT_PLATFORM
|
||||
|
||||
if (Is64BitMode()) {
|
||||
AuxVariables.emplace_back(auxv_t{4, 0x38}); // AT_PHENT
|
||||
// On x86 this is the value returned from CPUID 01h EDX
|
||||
AuxVariables.emplace_back(auxv_t{16, 0}); // AT_HWCAP
|
||||
|
||||
//AuxVariables.emplace_back(auxv_t{24, ~0ULL}); // AT_PLATFORM
|
||||
// On x86 only allows userspace to check for monitor and fs/gs base writing in CPL3
|
||||
//AuxVariables.emplace_back(auxv_t{26, 0}); // AT_HWCAP2
|
||||
|
||||
// we don't support vsyscall so we don't set those
|
||||
//AuxVariables.emplace_back(auxv_t{32, 0}); // AT_SYSINFO - Entry point to syscall
|
||||
if (VDSOBase) {
|
||||
AuxVariables.emplace_back(auxv_t{33, reinterpret_cast<uint64_t>(VDSOBase)}); // AT_SYSINFO_EHDR - Address of the start of VDSO
|
||||
}
|
||||
}
|
||||
else {
|
||||
AuxVariables.emplace_back(auxv_t{4, 0x20}); // AT_PHENT
|
||||
|
||||
// we don't support vsyscall or vDSO so we don't set those
|
||||
// we don't support vsyscall so we don't set those
|
||||
//AuxVariables.emplace_back(auxv_t{32, 0}); // AT_SYSINFO - Entry point to syscall
|
||||
//AuxVariables.emplace_back(auxv_t{33, 0}); // AT_SYSINFO_EHDR - Address of the start of VDSO
|
||||
}
|
||||
|
||||
if (VDSOBase) {
|
||||
AuxVariables.emplace_back(auxv_t{33, reinterpret_cast<uint64_t>(VDSOBase)}); // AT_SYSINFO_EHDR - Address of the start of VDSO
|
||||
}
|
||||
|
||||
AuxVariables.emplace_back(auxv_t{3, MainElfBase + MainElf.ehdr.e_phoff}); // Program header
|
||||
AuxVariables.emplace_back(auxv_t{7, InterpeterElfBase}); // AT_BASE - Interpreter address
|
||||
AuxVariables.emplace_back(auxv_t{9, MainElfEntrypoint}); // AT_ENTRY
|
||||
@@ -463,10 +576,11 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
|
||||
uint64_t EnvpOffset,
|
||||
const std::vector<std::string> &Args,
|
||||
const std::vector<std::string> &EnvironmentVariables,
|
||||
const std::vector<auxv_t> &AuxVariables,
|
||||
const std::list<auxv_t> &AuxVariables,
|
||||
uint64_t *AuxTabBase,
|
||||
uint64_t *AuxTabSize,
|
||||
PointerType RandomNumberOffset
|
||||
PointerType RandomNumberOffset,
|
||||
PointerType PlatformNameOffset
|
||||
) {
|
||||
// Pointer list offsets
|
||||
PointerType *ArgumentPointers = reinterpret_cast<PointerType*>(StackPointer + PointerSize);
|
||||
@@ -520,20 +634,10 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
|
||||
// Last envp needs to be nullptr
|
||||
EnvpPointers[EnvironmentVariables.size()] = 0;
|
||||
|
||||
for (size_t i = 0; i < AuxVariables.size(); ++i) {
|
||||
if (AuxVariables[i].key == 25) {
|
||||
// Random value is always 128bits
|
||||
AuxType Random{25, static_cast<PointerType>(StackPointer + RandomNumberOffset)};
|
||||
uint64_t *RandomLoc = reinterpret_cast<uint64_t*>(StackPointer + RandomNumberOffset);
|
||||
RandomLoc[0] = 0xDEAD;
|
||||
RandomLoc[1] = 0xDEAD2;
|
||||
AuxVPointers[i].key = Random.key;
|
||||
AuxVPointers[i].val = Random.val;
|
||||
}
|
||||
else {
|
||||
AuxVPointers[i].key = AuxVariables[i].key;
|
||||
AuxVPointers[i].val = AuxVariables[i].val;
|
||||
}
|
||||
for (size_t i = 0; auto const &Variable : AuxVariables) {
|
||||
AuxVPointers[i].key = Variable.key;
|
||||
AuxVPointers[i].val = Variable.val;
|
||||
++i;
|
||||
}
|
||||
|
||||
*AuxTabBase = reinterpret_cast<uint64_t>(AuxVPointers);
|
||||
@@ -569,12 +673,44 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
|
||||
TotalArgumentMemSize += EnvironmentBackingSize;
|
||||
|
||||
// Random number location
|
||||
uint32_t RandomNumberLocation = TotalArgumentMemSize;
|
||||
uint64_t RandomNumberLocation = TotalArgumentMemSize;
|
||||
TotalArgumentMemSize += 16;
|
||||
|
||||
uint64_t PlatformNameLocation = TotalArgumentMemSize;
|
||||
TotalArgumentMemSize += platform_string_max_size;
|
||||
|
||||
// Offset the stack by how much memory we need
|
||||
StackPointer -= TotalArgumentMemSize;
|
||||
|
||||
// Setup our AUXP values that need memory now that the stack is setup
|
||||
AuxPlatform->val = StackPointer + PlatformNameLocation;
|
||||
char *PlatformLoc = reinterpret_cast<char*>(AuxPlatform->val);
|
||||
memset(PlatformLoc, 0, platform_string_max_size);
|
||||
if (Is64BitMode()) {
|
||||
strncpy(PlatformLoc, platform_name_x86_64.data(), platform_string_max_size);
|
||||
}
|
||||
else {
|
||||
strncpy(PlatformLoc, platform_name_i686.data(), platform_string_max_size);
|
||||
}
|
||||
|
||||
// Random value is always 128bits
|
||||
AuxRandom->val = StackPointer + RandomNumberLocation;
|
||||
uint64_t *RandomLoc = reinterpret_cast<uint64_t*>(AuxRandom->val);
|
||||
uint64_t *HostRandom = reinterpret_cast<uint64_t*>(getauxval(AT_RANDOM));
|
||||
if (HostRandom) {
|
||||
// Pass through the host's random values
|
||||
RandomLoc[0] = HostRandom[0];
|
||||
RandomLoc[1] = HostRandom[1];
|
||||
}
|
||||
else {
|
||||
// Nothing provided from the kernel, generate our own random values.
|
||||
std::random_device rd;
|
||||
std::uniform_int_distribution<uint64_t> d(0);
|
||||
|
||||
RandomLoc[0] = d(rd);
|
||||
RandomLoc[1] = d(rd);
|
||||
}
|
||||
|
||||
// Stack setup
|
||||
// [0, 8): Argument Count
|
||||
// [8, 16): Argument Pointer 0
|
||||
@@ -600,7 +736,8 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
|
||||
AuxVariables,
|
||||
&AuxTabBase,
|
||||
&AuxTabSize,
|
||||
RandomNumberLocation
|
||||
RandomNumberLocation,
|
||||
PlatformNameLocation
|
||||
);
|
||||
}
|
||||
else {
|
||||
@@ -614,7 +751,8 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
|
||||
AuxVariables,
|
||||
&AuxTabBase,
|
||||
&AuxTabSize,
|
||||
RandomNumberLocation
|
||||
RandomNumberLocation,
|
||||
PlatformNameLocation
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -647,19 +785,43 @@ class ELFCodeLoader2 final : public FEXCore::CodeLoader {
|
||||
VDSOBase = Base;
|
||||
}
|
||||
|
||||
void CalculateHWCaps(FEXCore::Context::Context *ctx) {
|
||||
// HWCAP is just CPUID function 0x1, the EDX result
|
||||
auto res_1 = FEXCore::Context::RunCPUIDFunction(ctx, 1, 0);
|
||||
HWCap = res_1.edx;
|
||||
|
||||
// HWCAP2 is as follows:
|
||||
// Bits:
|
||||
// 0 - MONITOR/MWAIT available in CPL3
|
||||
// 1 - FSGSBASE instructions available in CPL3
|
||||
HWCap2 = 0;
|
||||
}
|
||||
|
||||
constexpr static uint64_t BRK_SIZE = 8 * 1024 * 1024;
|
||||
constexpr static uint64_t STACK_SIZE = 8 * 1024 * 1024;
|
||||
constexpr static uint64_t FULL_STACK_SIZE = 128 * 1024 * 1024;
|
||||
constexpr static uint64_t STACK_HINT_32 = 0xFFFFE000 - FULL_STACK_SIZE;
|
||||
constexpr static uint64_t STACK_HINT_64 = 0x7FFFFFFFF000 - FULL_STACK_SIZE;
|
||||
|
||||
std::vector<std::string> Args;
|
||||
std::vector<std::string> EnvironmentVariables;
|
||||
std::vector<char const*> LoaderArgs;
|
||||
|
||||
std::vector<auxv_t> AuxVariables;
|
||||
std::list<auxv_t> AuxVariables;
|
||||
uint64_t AuxTabBase, AuxTabSize;
|
||||
uint64_t ArgumentBackingSize{};
|
||||
uint64_t EnvironmentBackingSize{};
|
||||
uint64_t BaseOffset{};
|
||||
void* VDSOBase{};
|
||||
uint64_t HWCap{};
|
||||
uint64_t HWCap2{};
|
||||
|
||||
auxv_t *AuxRandom{};
|
||||
auxv_t *AuxPlatform{};
|
||||
|
||||
static constexpr std::string_view platform_name_x86_64 = "x86_64";
|
||||
static constexpr std::string_view platform_name_i686 = "i686";
|
||||
static constexpr size_t platform_string_max_size = std::max(platform_name_x86_64.size(), platform_name_i686.size());
|
||||
|
||||
FEX_CONFIG_OPT(AdditionalArguments, ADDITIONALARGUMENTS);
|
||||
};
|
||||
@@ -24,6 +24,7 @@ $end_info$
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cerrno>
|
||||
@@ -283,6 +284,7 @@ int main(int argc, char **argv, char **const envp) {
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::Profiler::Init();
|
||||
FEXCore::Telemetry::Initialize();
|
||||
|
||||
RootFSRedirect(&Program.first, LDPath());
|
||||
@@ -328,9 +330,9 @@ int main(int argc, char **argv, char **const envp) {
|
||||
return -ENOEXEC;
|
||||
}
|
||||
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_APP_FILENAME, std::filesystem::canonical(Program.first).string());
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_APP_CONFIG_NAME, Program.second);
|
||||
FEXCore::Config::Set(FEXCore::Config::CONFIG_IS64BIT_MODE, Loader.Is64BitMode() ? "1" : "0");
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_APP_FILENAME, std::filesystem::canonical(Program.first).string());
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_APP_CONFIG_NAME, Program.second);
|
||||
FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_IS64BIT_MODE, Loader.Is64BitMode() ? "1" : "0");
|
||||
|
||||
std::unique_ptr<FEX::HLE::MemAllocator> Allocator;
|
||||
std::vector<FEXCore::Allocator::MemoryRegion> Base48Bit;
|
||||
@@ -390,11 +392,10 @@ int main(int argc, char **argv, char **const envp) {
|
||||
auto Mapper = std::bind_front(&FEX::HLE::SyscallHandler::GuestMmap, SyscallHandler.get());
|
||||
auto Unmapper = std::bind_front(&FEX::HLE::SyscallHandler::GuestMunmap, SyscallHandler.get());
|
||||
|
||||
if (Loader.Is64BitMode()) {
|
||||
// Load VDSO in to memory prior to mapping our ELFs.
|
||||
void* VDSOBase = FEX::VDSO::LoadVDSOThunks(Mapper);
|
||||
Loader.SetVDSOBase(VDSOBase);
|
||||
}
|
||||
// Load VDSO in to memory prior to mapping our ELFs.
|
||||
void* VDSOBase = FEX::VDSO::LoadVDSOThunks(Loader.Is64BitMode(), Mapper);
|
||||
Loader.SetVDSOBase(VDSOBase);
|
||||
Loader.CalculateHWCaps(CTX);
|
||||
|
||||
if (!Loader.MapMemory(Mapper, Unmapper)) {
|
||||
// failed to map
|
||||
@@ -504,6 +505,7 @@ int main(int argc, char **argv, char **const envp) {
|
||||
FEXCore::Allocator::ReclaimMemoryRegion(Base48Bit);
|
||||
// Allocator is now original system allocator
|
||||
FEXCore::Telemetry::Shutdown(Program.second);
|
||||
FEXCore::Profiler::Shutdown();
|
||||
if (ShutdownReason == FEXCore::Context::ExitReason::EXIT_SHUTDOWN) {
|
||||
return ProgramStatus;
|
||||
}
|
||||
|
||||
@@ -127,13 +127,13 @@ namespace FEX::HarnessHelper {
|
||||
|
||||
// GS
|
||||
if (MatchMask & 1) {
|
||||
CheckGPRs("GS", State1.gs, State2.gs);
|
||||
CheckGPRs("GS", State1.gs_cached, State2.gs_cached);
|
||||
}
|
||||
MatchMask >>= 1;
|
||||
|
||||
// FS
|
||||
if (MatchMask & 1) {
|
||||
CheckGPRs("FS", State1.fs, State2.fs);
|
||||
CheckGPRs("FS", State1.fs_cached, State2.fs_cached);
|
||||
}
|
||||
MatchMask >>= 1;
|
||||
|
||||
@@ -233,8 +233,8 @@ namespace FEX::HarnessHelper {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.avx.data[13][0]),
|
||||
offsetof(FEXCore::Core::CPUState, xmm.avx.data[14][0]),
|
||||
offsetof(FEXCore::Core::CPUState, xmm.avx.data[15][0]),
|
||||
offsetof(FEXCore::Core::CPUState, gs),
|
||||
offsetof(FEXCore::Core::CPUState, fs),
|
||||
offsetof(FEXCore::Core::CPUState, gs_cached),
|
||||
offsetof(FEXCore::Core::CPUState, fs_cached),
|
||||
offsetof(FEXCore::Core::CPUState, flags),
|
||||
offsetof(FEXCore::Core::CPUState, mm[0][0]),
|
||||
offsetof(FEXCore::Core::CPUState, mm[1][0]),
|
||||
@@ -280,8 +280,8 @@ namespace FEX::HarnessHelper {
|
||||
offsetof(FEXCore::Core::CPUState, xmm.sse.data[13][0]),
|
||||
offsetof(FEXCore::Core::CPUState, xmm.sse.data[14][0]),
|
||||
offsetof(FEXCore::Core::CPUState, xmm.sse.data[15][0]),
|
||||
offsetof(FEXCore::Core::CPUState, gs),
|
||||
offsetof(FEXCore::Core::CPUState, fs),
|
||||
offsetof(FEXCore::Core::CPUState, gs_cached),
|
||||
offsetof(FEXCore::Core::CPUState, fs_cached),
|
||||
offsetof(FEXCore::Core::CPUState, flags),
|
||||
offsetof(FEXCore::Core::CPUState, mm[0][0]),
|
||||
offsetof(FEXCore::Core::CPUState, mm[1][0]),
|
||||
|
||||
@@ -230,7 +230,7 @@ FileManager::FileManager(FEXCore::Context::Context *ctx)
|
||||
auto LoadThunksDB = [this, ThunkGuestPath](bool *LoadedThunkDatabase, json_t const* ThunksDB) {
|
||||
// If a thunks DB property exists then we pull in data from the thunks database
|
||||
// Load the initial thunks database
|
||||
if (LoadedThunkDatabase) {
|
||||
if (!*LoadedThunkDatabase) {
|
||||
LoadThunkDatabase(true);
|
||||
LoadThunkDatabase(false);
|
||||
*LoadedThunkDatabase = true;
|
||||
@@ -239,60 +239,25 @@ FileManager::FileManager(FEXCore::Context::Context *ctx)
|
||||
// Now load this property
|
||||
for (json_t const* Item = json_getChild(ThunksDB); Item != nullptr; Item = json_getSibling(Item)) {
|
||||
const char *LibraryName = json_getName(Item);
|
||||
int64_t LibraryEnabled = json_getInteger(Item);
|
||||
if (LibraryEnabled != 0) {
|
||||
// If the library is enabled then find it in the DB
|
||||
// Enable the overlay and all the dependencies in one go
|
||||
auto DBObject = ThunkDB.find(LibraryName);
|
||||
if (DBObject != ThunkDB.end() &&
|
||||
DBObject->second.Enabled == false) {
|
||||
|
||||
auto ThunkPath = ThunkGuestPath / DBObject->second.LibraryName;
|
||||
if (std::filesystem::exists(ThunkPath)) {
|
||||
for (auto Overlay : DBObject->second.Overlays) {
|
||||
// Direct full path in guest RootFS to our overlay file
|
||||
ThunkOverlays.emplace(Overlay, ThunkPath);
|
||||
}
|
||||
}
|
||||
DBObject->second.Enabled = true;
|
||||
// Now walk the dependencies and set them up as well
|
||||
// Make sure to enable each one as we go to remove circular dependencies
|
||||
std::function<void(std::unordered_set<std::string> &Depends)> InsertDependencies
|
||||
= [this, &ThunkGuestPath, &InsertDependencies](std::unordered_set<std::string> &Depends) -> void {
|
||||
for (auto &Depend : Depends) {
|
||||
auto DBDepend = ThunkDB.find(Depend);
|
||||
if (DBDepend != ThunkDB.end() &&
|
||||
DBDepend->second.Enabled == false) {
|
||||
|
||||
auto ThunkPath = ThunkGuestPath / DBDepend->second.LibraryName;
|
||||
if (std::filesystem::exists(ThunkPath)) {
|
||||
for (auto Overlay : DBDepend->second.Overlays) {
|
||||
// Direct full path in guest RootFS to our overlay file
|
||||
ThunkOverlays.emplace(Overlay, ThunkPath);
|
||||
}
|
||||
}
|
||||
|
||||
// Enabled, now walk this dependencies
|
||||
DBDepend->second.Enabled = true;
|
||||
InsertDependencies(DBDepend->second.Depends);
|
||||
}
|
||||
}
|
||||
};
|
||||
InsertDependencies(DBObject->second.Depends);
|
||||
}
|
||||
bool LibraryEnabled = json_getInteger(Item) != 0;
|
||||
// If the library is enabled then find it in the DB
|
||||
// Enable the overlay and all the dependencies in one go
|
||||
auto DBObject = ThunkDB.find(LibraryName);
|
||||
if (DBObject != ThunkDB.end()) {
|
||||
DBObject->second.Enabled = LibraryEnabled;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
// We try to load ThunksDB from {FEX global config, FEX user config, AppConfig Global, AppConfig Local, Defined ThunksConfig option}
|
||||
// We try to load ThunksDB from {FEX global config, FEX user config, Defined ThunksConfig option, AppConfig Global, AppConfig Local}
|
||||
// This doesn't support the classic thunks interface.
|
||||
|
||||
std::vector<std::string> ConfigPaths {
|
||||
FEXCore::Config::GetConfigFileLocation(true),
|
||||
FEXCore::Config::GetConfigFileLocation(false),
|
||||
ThunkConfigFile,
|
||||
FEXCore::Config::GetApplicationConfig(AppConfigName(), true),
|
||||
FEXCore::Config::GetApplicationConfig(AppConfigName(), false),
|
||||
ThunkConfigFile,
|
||||
};
|
||||
|
||||
for (const auto &Path : ConfigPaths) {
|
||||
@@ -313,6 +278,40 @@ FileManager::FileManager(FEXCore::Context::Context *ctx)
|
||||
}
|
||||
}
|
||||
|
||||
// Now that we loaded the thunks object, walk through and ensure dependencies are enabled as well.
|
||||
for (auto const &DBObject : ThunkDB) {
|
||||
if (!DBObject.second.Enabled) {
|
||||
continue;
|
||||
}
|
||||
|
||||
// Now walk the dependencies and set them up as well
|
||||
// Make sure to enable each one as we go to remove circular dependencies
|
||||
std::function<void(const std::unordered_set<std::string> &Depends, bool AlreadyEnabled)> InsertDependencies
|
||||
= [this, &ThunkGuestPath, &InsertDependencies](const std::unordered_set<std::string> &Depends, bool AlreadyEnabled) -> void {
|
||||
for (auto const &Depend : Depends) {
|
||||
auto DBDepend = ThunkDB.find(Depend);
|
||||
if (DBDepend != ThunkDB.end() &&
|
||||
(DBDepend->second.Enabled == false || AlreadyEnabled)) {
|
||||
|
||||
auto ThunkPath = ThunkGuestPath / DBDepend->second.LibraryName;
|
||||
if (std::filesystem::exists(ThunkPath)) {
|
||||
for (const auto& Overlay : DBDepend->second.Overlays) {
|
||||
// Direct full path in guest RootFS to our overlay file
|
||||
ThunkOverlays.emplace(Overlay, ThunkPath);
|
||||
}
|
||||
}
|
||||
|
||||
// Enabled, now walk this dependencies
|
||||
DBDepend->second.Enabled = true;
|
||||
InsertDependencies(DBDepend->second.Depends, false);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
InsertDependencies({DBObject.first}, true);
|
||||
InsertDependencies(DBObject.second.Depends, false);
|
||||
}
|
||||
|
||||
// Now clear the thunk database since we're loaded
|
||||
ThunkDB.clear();
|
||||
|
||||
|
||||
@@ -455,7 +455,7 @@ namespace FEX::HLE {
|
||||
// Ignore a non-canonical address
|
||||
return -EPERM;
|
||||
}
|
||||
Frame->State.gs = addr;
|
||||
Frame->State.gs_cached = addr;
|
||||
Result = 0;
|
||||
break;
|
||||
case 0x1002: // ARCH_SET_FS
|
||||
@@ -463,15 +463,15 @@ namespace FEX::HLE {
|
||||
// Ignore a non-canonical address
|
||||
return -EPERM;
|
||||
}
|
||||
Frame->State.fs = addr;
|
||||
Frame->State.fs_cached = addr;
|
||||
Result = 0;
|
||||
break;
|
||||
case 0x1003: // ARCH_GET_FS
|
||||
*reinterpret_cast<uint64_t*>(addr) = Frame->State.fs;
|
||||
*reinterpret_cast<uint64_t*>(addr) = Frame->State.fs_cached;
|
||||
Result = 0;
|
||||
break;
|
||||
case 0x1004: // ARCH_GET_GS
|
||||
*reinterpret_cast<uint64_t*>(addr) = Frame->State.gs;
|
||||
*reinterpret_cast<uint64_t*>(addr) = Frame->State.gs_cached;
|
||||
Result = 0;
|
||||
break;
|
||||
case 0x3001: // ARCH_CET_STATUS
|
||||
|
||||
@@ -341,9 +341,12 @@ void SyscallHandler::TrackShmat(int shmid, uintptr_t Base, int shmflg) {
|
||||
}
|
||||
|
||||
void SyscallHandler::TrackShmdt(uintptr_t Base) {
|
||||
FHU::ScopedSignalMaskWithUniqueLock lk(_SyscallHandler->VMATracking.Mutex);
|
||||
uintptr_t Length = 0;
|
||||
{
|
||||
FHU::ScopedSignalMaskWithUniqueLock lk(_SyscallHandler->VMATracking.Mutex);
|
||||
|
||||
auto Length = VMATracking.ClearShmUnsafe(CTX, Base);
|
||||
Length = VMATracking.ClearShmUnsafe(CTX, Base);
|
||||
}
|
||||
|
||||
if (SMCChecks != FEXCore::Config::CONFIG_SMC_NONE) {
|
||||
// This might over flush if the shm has holes in it
|
||||
|
||||
@@ -68,6 +68,29 @@ namespace FEX::HLE::x32 {
|
||||
// Now we need to update the thread's GDT to handle this change
|
||||
auto GDT = &Frame->State.gdt[u_info->entry_number];
|
||||
GDT->base = u_info->base_addr;
|
||||
|
||||
// With the segment register optimization we need to check all of the segment registers and update.
|
||||
const auto GetEntry = [](auto value) {
|
||||
return value >> 3;
|
||||
};
|
||||
if (GetEntry(Frame->State.cs_idx) == u_info->entry_number) {
|
||||
Frame->State.cs_cached = GDT->base;
|
||||
}
|
||||
if (GetEntry(Frame->State.ds_idx) == u_info->entry_number) {
|
||||
Frame->State.ds_cached = GDT->base;
|
||||
}
|
||||
if (GetEntry(Frame->State.es_idx) == u_info->entry_number) {
|
||||
Frame->State.es_cached = GDT->base;
|
||||
}
|
||||
if (GetEntry(Frame->State.fs_idx) == u_info->entry_number) {
|
||||
Frame->State.fs_cached = GDT->base;
|
||||
}
|
||||
if (GetEntry(Frame->State.gs_idx) == u_info->entry_number) {
|
||||
Frame->State.gs_cached = GDT->base;
|
||||
}
|
||||
if (GetEntry(Frame->State.ss_idx) == u_info->entry_number) {
|
||||
Frame->State.ss_cached = GDT->base;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
@@ -36,7 +36,7 @@ namespace FEX::HLE::x64 {
|
||||
Result = -1;
|
||||
}
|
||||
} else {
|
||||
Result = reinterpret_cast<uint64_t>(FEXCore::Allocator::mmap(reinterpret_cast<void*>(addr), length, prot, flags, fd, offset));
|
||||
Result = reinterpret_cast<uint64_t>(::mmap(reinterpret_cast<void*>(addr), length, prot, flags, fd, offset));
|
||||
}
|
||||
|
||||
if (Result != -1) {
|
||||
@@ -56,7 +56,7 @@ namespace FEX::HLE::x64 {
|
||||
Result = -1;
|
||||
}
|
||||
} else {
|
||||
Result = FEXCore::Allocator::munmap(addr, length);
|
||||
Result = ::munmap(addr, length);
|
||||
}
|
||||
|
||||
if (Result != -1) {
|
||||
|
||||
@@ -24,7 +24,7 @@ $end_info$
|
||||
|
||||
namespace FEX::HLE::x64 {
|
||||
uint64_t SetThreadArea(FEXCore::Core::CpuStateFrame *Frame, void *tls) {
|
||||
Frame->State.fs = reinterpret_cast<uint64_t>(tls);
|
||||
Frame->State.fs_cached = reinterpret_cast<uint64_t>(tls);
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
@@ -1,9 +1,11 @@
|
||||
#include "VDSO_Emulation.h"
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "Tests/LinuxSyscalls/x32/Types.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include <dlfcn.h>
|
||||
#include <fcntl.h>
|
||||
@@ -17,13 +19,13 @@ namespace FEX::VDSO {
|
||||
using GetTimeOfDayType = decltype(::gettimeofday)*;
|
||||
using ClockGetTimeType = decltype(::clock_gettime)*;
|
||||
using ClockGetResType = decltype(::clock_getres)*;
|
||||
using GetCPUType = decltype(::getcpu)*;
|
||||
using GetCPUType = decltype(FHU::Syscalls::getcpu)*;
|
||||
|
||||
TimeType TimePtr = ::time;
|
||||
GetTimeOfDayType GetTimeOfDayPtr = ::gettimeofday;
|
||||
ClockGetTimeType ClockGetTimePtr = ::clock_gettime;
|
||||
ClockGetResType ClockGetResPtr = ::clock_getres;
|
||||
GetCPUType GetCPUPtr = ::getcpu;
|
||||
GetCPUType GetCPUPtr = FHU::Syscalls::getcpu;
|
||||
|
||||
static void time(void* ArgsRV) {
|
||||
struct ArgsRV_t {
|
||||
@@ -74,7 +76,94 @@ namespace FEX::VDSO {
|
||||
args->rv = GetCPUPtr(args->cpu, args->node);
|
||||
}
|
||||
|
||||
namespace x32 {
|
||||
static void time(void* ArgsRV) {
|
||||
struct ArgsRV_t {
|
||||
HLE::x32::compat_ptr<FEX::HLE::x32::old_time32_t> a_0;
|
||||
int rv;
|
||||
} *args = reinterpret_cast<ArgsRV_t*>(ArgsRV);
|
||||
|
||||
time_t Host{};
|
||||
args->rv = TimePtr(&Host);
|
||||
if (args->a_0) {
|
||||
*args->a_0 = Host;
|
||||
}
|
||||
}
|
||||
|
||||
static void gettimeofday(void* ArgsRV) {
|
||||
struct ArgsRV_t {
|
||||
HLE::x32::compat_ptr<FEX::HLE::x32::timeval32> tv;
|
||||
HLE::x32::compat_ptr<struct timezone> tz;
|
||||
int rv;
|
||||
} *args = reinterpret_cast<ArgsRV_t*>(ArgsRV);
|
||||
|
||||
struct timeval tv64{};
|
||||
struct timeval *tv_ptr{};
|
||||
if (args->tv) {
|
||||
tv_ptr = &tv64;
|
||||
}
|
||||
|
||||
args->rv = GetTimeOfDayPtr(tv_ptr, args->tz);
|
||||
|
||||
if (args->tv) {
|
||||
*args->tv = tv64;
|
||||
}
|
||||
}
|
||||
|
||||
static void clock_gettime(void* ArgsRV) {
|
||||
struct ArgsRV_t {
|
||||
clockid_t clk_id;
|
||||
HLE::x32::compat_ptr<HLE::x32::timespec32> tp;
|
||||
int rv;
|
||||
} *args = reinterpret_cast<ArgsRV_t*>(ArgsRV);
|
||||
|
||||
struct timespec tp64{};
|
||||
args->rv = ClockGetTimePtr(args->clk_id, &tp64);
|
||||
|
||||
if (args->tp) {
|
||||
*args->tp = tp64;
|
||||
}
|
||||
}
|
||||
|
||||
static void clock_gettime64(void* ArgsRV) {
|
||||
struct ArgsRV_t {
|
||||
clockid_t clk_id;
|
||||
HLE::x32::compat_ptr<struct timespec> tp;
|
||||
int rv;
|
||||
} *args = reinterpret_cast<ArgsRV_t*>(ArgsRV);
|
||||
|
||||
args->rv = ClockGetTimePtr(args->clk_id, args->tp);
|
||||
}
|
||||
|
||||
static void clock_getres(void* ArgsRV) {
|
||||
struct ArgsRV_t {
|
||||
clockid_t clk_id;
|
||||
HLE::x32::compat_ptr<HLE::x32::timespec32> tp;
|
||||
int rv;
|
||||
} *args = reinterpret_cast<ArgsRV_t*>(ArgsRV);
|
||||
|
||||
struct timespec tp64{};
|
||||
|
||||
args->rv = ClockGetResPtr(args->clk_id, &tp64);
|
||||
|
||||
if (args->tp) {
|
||||
*args->tp = tp64;
|
||||
}
|
||||
}
|
||||
|
||||
static void getcpu(void* ArgsRV) {
|
||||
struct ArgsRV_t {
|
||||
HLE::x32::compat_ptr<uint32_t> cpu;
|
||||
HLE::x32::compat_ptr<uint32_t> node;
|
||||
int rv;
|
||||
} *args = reinterpret_cast<ArgsRV_t*>(ArgsRV);
|
||||
|
||||
args->rv = GetCPUPtr(args->cpu, args->node);
|
||||
}
|
||||
}
|
||||
|
||||
void LoadHostVDSO() {
|
||||
|
||||
void *vdso = dlopen("linux-vdso.so.1", RTLD_LAZY | RTLD_LOCAL | RTLD_NOLOAD);
|
||||
if (!vdso) {
|
||||
vdso = dlopen("linux-gate.so.1", RTLD_LAZY | RTLD_LOCAL | RTLD_NOLOAD);
|
||||
@@ -117,36 +206,67 @@ namespace FEX::VDSO {
|
||||
{
|
||||
// sha256(libVDSO:time)
|
||||
{ 0x37, 0x63, 0x46, 0xb0, 0x79, 0x06, 0x5f, 0x9d, 0x00, 0xb6, 0x8d, 0xfd, 0x9e, 0x4a, 0x62, 0xcd, 0x1e, 0x6c, 0xcc, 0x22, 0xcd, 0xb2, 0xc0, 0x17, 0x7d, 0x42, 0x6a, 0x40, 0xd1, 0xeb, 0xfa, 0xe0 },
|
||||
&FEX::VDSO::time
|
||||
nullptr,
|
||||
},
|
||||
{
|
||||
// sha256(libVDSO:gettimeofday)
|
||||
{ 0x77, 0x2a, 0xde, 0x1c, 0x13, 0x2d, 0xe9, 0x48, 0xaf, 0xe0, 0xba, 0xcc, 0x6a, 0x89, 0xff, 0xca, 0x4a, 0xdc, 0xd5, 0x63, 0x2c, 0xc5, 0x62, 0x8b, 0x5d, 0xde, 0x0b, 0x15, 0x35, 0xc6, 0xc7, 0x14 },
|
||||
&FEX::VDSO::gettimeofday
|
||||
nullptr,
|
||||
},
|
||||
{
|
||||
// sha256(libVDSO:clock_gettime)
|
||||
{ 0x3c, 0x96, 0x9b, 0x2d, 0xc3, 0xad, 0x2b, 0x3b, 0x9c, 0x4e, 0x4d, 0xca, 0x1c, 0xe8, 0x18, 0x4a, 0x12, 0x8a, 0xe4, 0xc1, 0x56, 0x92, 0x73, 0xce, 0x65, 0x85, 0x5f, 0x65, 0x7e, 0x94, 0x26, 0xbe },
|
||||
&FEX::VDSO::clock_gettime
|
||||
nullptr,
|
||||
},
|
||||
|
||||
{
|
||||
// sha256(libVDSO:clock_gettime64)
|
||||
{ 0xba, 0xe9, 0x6d, 0x30, 0xc0, 0x68, 0xc6, 0xd7, 0x59, 0x04, 0xf7, 0x10, 0x06, 0x72, 0x88, 0xfd, 0x4c, 0x57, 0x0f, 0x31, 0xa5, 0xea, 0xa9, 0xb9, 0xd3, 0x8d, 0x03, 0x81, 0x50, 0x16, 0x22, 0x71 },
|
||||
nullptr,
|
||||
},
|
||||
|
||||
{
|
||||
// sha256(libVDSO:clock_getres)
|
||||
{ 0xe4, 0xa1, 0xf6, 0x23, 0x35, 0xae, 0xb7, 0xb6, 0xb0, 0x37, 0xc5, 0xc3, 0xa3, 0xfd, 0xbf, 0xa2, 0xa1, 0xc8, 0x95, 0x78, 0xe5, 0x76, 0x86, 0xdb, 0x3e, 0x6c, 0x54, 0xd5, 0x02, 0x60, 0xd8, 0x6d },
|
||||
&FEX::VDSO::clock_getres
|
||||
nullptr,
|
||||
},
|
||||
{
|
||||
// sha256(libVDSO:getcpu)
|
||||
{ 0x39, 0x83, 0x39, 0x36, 0x0f, 0x68, 0xd6, 0xfc, 0xc2, 0x3a, 0x97, 0x11, 0x85, 0x09, 0xc7, 0x25, 0xbb, 0x50, 0x49, 0x55, 0x6b, 0x0c, 0x9f, 0x50, 0x37, 0xf5, 0x9d, 0xb0, 0x38, 0x58, 0x57, 0x12 },
|
||||
&FEX::VDSO::getcpu
|
||||
nullptr,
|
||||
},
|
||||
};
|
||||
|
||||
void* LoadVDSOThunks(MapperFn Mapper) {
|
||||
void* LoadVDSOThunks(bool Is64Bit, MapperFn Mapper) {
|
||||
void* VDSOBase{};
|
||||
FEX_CONFIG_OPT(ThunkGuestLibs, THUNKGUESTLIBS);
|
||||
FEX_CONFIG_OPT(ThunkGuestLibs32, THUNKGUESTLIBS32);
|
||||
|
||||
std::filesystem::path ThunkGuestPath{};
|
||||
if (Is64Bit) {
|
||||
ThunkGuestPath = std::filesystem::path(ThunkGuestLibs()) / "libVDSO-guest.so";
|
||||
|
||||
// Set the Thunk definition pointers for x86-64
|
||||
VDSODefinitions[0].ThunkFunction = &FEX::VDSO::time;
|
||||
VDSODefinitions[1].ThunkFunction = &FEX::VDSO::gettimeofday;
|
||||
VDSODefinitions[2].ThunkFunction = &FEX::VDSO::clock_gettime;
|
||||
VDSODefinitions[3].ThunkFunction = &FEX::VDSO::clock_gettime;
|
||||
VDSODefinitions[4].ThunkFunction = &FEX::VDSO::clock_getres;
|
||||
VDSODefinitions[5].ThunkFunction = &FEX::VDSO::getcpu;
|
||||
}
|
||||
else {
|
||||
ThunkGuestPath = std::filesystem::path(ThunkGuestLibs32()) / "libVDSO-guest.so";
|
||||
|
||||
// Set the Thunk definition pointers for x86
|
||||
VDSODefinitions[0].ThunkFunction = &FEX::VDSO::x32::time;
|
||||
VDSODefinitions[1].ThunkFunction = &FEX::VDSO::x32::gettimeofday;
|
||||
VDSODefinitions[2].ThunkFunction = &FEX::VDSO::x32::clock_gettime;
|
||||
VDSODefinitions[3].ThunkFunction = &FEX::VDSO::x32::clock_gettime64;
|
||||
VDSODefinitions[4].ThunkFunction = &FEX::VDSO::x32::clock_getres;
|
||||
VDSODefinitions[5].ThunkFunction = &FEX::VDSO::x32::getcpu;
|
||||
}
|
||||
|
||||
// Load VDSO if we can
|
||||
auto ThunkGuestPath = std::filesystem::path(ThunkGuestLibs()) / "libVDSO-guest.so";
|
||||
int VDSOFD = ::open(ThunkGuestPath.string().c_str(), O_RDONLY);
|
||||
|
||||
if (VDSOFD != -1) {
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
|
||||
namespace FEX::VDSO {
|
||||
using MapperFn = std::function<void *(void *addr, size_t length, int prot, int flags, int fd, off_t offset)>;
|
||||
void* LoadVDSOThunks(MapperFn Mapper);
|
||||
void* LoadVDSOThunks(bool Is64Bit, MapperFn Mapper);
|
||||
|
||||
std::vector<FEXCore::IR::ThunkDefinition> const& GetVDSOThunkDefinitions();
|
||||
}
|
||||
@@ -74,6 +74,7 @@ namespace {
|
||||
LoadedConfig->Erase(FEXCore::Config::ConfigOption::CONFIG_IS_INTERPRETER);
|
||||
LoadedConfig->Erase(FEXCore::Config::ConfigOption::CONFIG_INTERPRETER_INSTALLED);
|
||||
LoadedConfig->Erase(FEXCore::Config::ConfigOption::CONFIG_APP_FILENAME);
|
||||
LoadedConfig->Erase(FEXCore::Config::ConfigOption::CONFIG_APP_CONFIG_NAME);
|
||||
LoadedConfig->Erase(FEXCore::Config::ConfigOption::CONFIG_IS64BIT_MODE);
|
||||
}
|
||||
|
||||
@@ -106,6 +107,7 @@ namespace {
|
||||
LoadedConfig->Erase(FEXCore::Config::ConfigOption::CONFIG_IS_INTERPRETER);
|
||||
LoadedConfig->Erase(FEXCore::Config::ConfigOption::CONFIG_INTERPRETER_INSTALLED);
|
||||
LoadedConfig->Erase(FEXCore::Config::ConfigOption::CONFIG_APP_FILENAME);
|
||||
LoadedConfig->Erase(FEXCore::Config::ConfigOption::CONFIG_APP_CONFIG_NAME);
|
||||
LoadedConfig->Erase(FEXCore::Config::ConfigOption::CONFIG_IS64BIT_MODE);
|
||||
|
||||
return true;
|
||||
|
||||
@@ -401,10 +401,29 @@ namespace ProcessPipe {
|
||||
case FEXServerClient::PacketType::TYPE_GET_PID_FD: {
|
||||
int FD = FHU::Syscalls::pidfd_open(::getpid(), 0);
|
||||
|
||||
SendFDSuccessPacket(Socket, FD);
|
||||
if (FD < 0) {
|
||||
// Couldn't get PIDFD due to too old of kernel.
|
||||
// Return a pipe to track the same information.
|
||||
//
|
||||
int fds[2];
|
||||
pipe2(fds, O_CLOEXEC);
|
||||
SendFDSuccessPacket(Socket, fds[0]);
|
||||
|
||||
// Close the FD now since we've sent it
|
||||
close(FD);
|
||||
// Close the read side now, doesn't matter to us
|
||||
close(fds[0]);
|
||||
|
||||
// Check if we need to increase the FD limit.
|
||||
++NumFilesOpened;
|
||||
CheckRaiseFDLimit();
|
||||
|
||||
// Write side will naturally close on process exit, letting the other process know we have exited.
|
||||
}
|
||||
else {
|
||||
SendFDSuccessPacket(Socket, FD);
|
||||
|
||||
// Close the FD now since we've sent it
|
||||
close(FD);
|
||||
}
|
||||
|
||||
CurrentOffset += sizeof(FEXServerClient::FEXServerRequestPacket::Header);
|
||||
break;
|
||||
@@ -412,6 +431,9 @@ namespace ProcessPipe {
|
||||
// Invalid
|
||||
case FEXServerClient::PacketType::TYPE_ERROR:
|
||||
default:
|
||||
// Something sent us an invalid packet. To ensure we don't spin infinitely, consume all the data.
|
||||
LogMan::Msg::EFmt("[FEXServer] InvalidPacket size received 0x{:x} bytes", CurrentRead - CurrentOffset);
|
||||
CurrentOffset = CurrentRead;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
Loaded 100 of 195 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user