mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 18:00:17 +02:00
Compare commits
499
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
ba1b4744c5 | ||
|
|
bd13c02451 | ||
|
|
5902b175f9 | ||
|
|
d0e47f9073 | ||
|
|
bd7215d36f | ||
|
|
f3f134f9de | ||
|
|
cdbf5d57bd | ||
|
|
1ae50bd670 | ||
|
|
d28c9f9843 | ||
|
|
fe7d52aa78 | ||
|
|
fc0907f8c1 | ||
|
|
e57678d7a6 | ||
|
|
45e594e806 | ||
|
|
87e7a0effa | ||
|
|
4fd1a35b2a | ||
|
|
c460cf0678 | ||
|
|
983802da61 | ||
|
|
49273e0d59 | ||
|
|
76b8459cdc | ||
|
|
e269eb6f65 | ||
|
|
7b4774f375 | ||
|
|
70e9a25112 | ||
|
|
9fb83ea56a | ||
|
|
8bb3398376 | ||
|
|
d42fbb3d4d | ||
|
|
53b2245dc1 | ||
|
|
db14975828 | ||
|
|
a5139d2710 | ||
|
|
9f584c8014 | ||
|
|
6f1b98fb52 | ||
|
|
c258a90505 | ||
|
|
eb47ef43a7 | ||
|
|
1876d6b923 | ||
|
|
90c8fcf393 | ||
|
|
6af90575e9 | ||
|
|
994613260c | ||
|
|
1430fa8220 | ||
|
|
33e06058c6 | ||
|
|
b032d1e1f7 | ||
|
|
d6b43b1fe6 | ||
|
|
fc771c8683 | ||
|
|
f91ac09f87 | ||
|
|
e4fa399412 | ||
|
|
952e949e10 | ||
|
|
3d093d66fb | ||
|
|
05fe2893c7 | ||
|
|
6fc17294b6 | ||
|
|
6607921bee | ||
|
|
0b0793438f | ||
|
|
3dd591e760 | ||
|
|
f290d2f899 | ||
|
|
a676ad7193 | ||
|
|
096c408ef6 | ||
|
|
00b65f76b6 | ||
|
|
39fb266282 | ||
|
|
3969d0ac78 | ||
|
|
439c6bb3c0 | ||
|
|
5cedbf9d34 | ||
|
|
427b235eb5 | ||
|
|
92d5ba580f | ||
|
|
bc2f331c8b | ||
|
|
ca58aef676 | ||
|
|
379dc405f6 | ||
|
|
d74b5c42da | ||
|
|
6004971439 | ||
|
|
a12b8927bc | ||
|
|
57e23b289a | ||
|
|
8214ffccf0 | ||
|
|
547135dc2d | ||
|
|
9c72113161 | ||
|
|
8cc967fa22 | ||
|
|
4bd30bb72d | ||
|
|
5eeb4dabbd | ||
|
|
a27c4b3860 | ||
|
|
dfee08f74f | ||
|
|
0e2629bdd4 | ||
|
|
05c8630b07 | ||
|
|
1cffe618d2 | ||
|
|
98c7bb23b5 | ||
|
|
922853cee1 | ||
|
|
a251e61859 | ||
|
|
9c19799023 | ||
|
|
f423b110a8 | ||
|
|
6fd471e652 | ||
|
|
3d69029d33 | ||
|
|
2258f2e424 | ||
|
|
cca5a68e20 | ||
|
|
da5c9bff68 | ||
|
|
5b87f0699b | ||
|
|
a67fe561a1 | ||
|
|
e2f4065376 | ||
|
|
d3bf87f4f4 | ||
|
|
6d351ec47f | ||
|
|
4dc1dd2511 | ||
|
|
5205ae40fa | ||
|
|
32f1dcde7e | ||
|
|
6772581c53 | ||
|
|
387201815b | ||
|
|
24d61e1125 | ||
|
|
04259f031d | ||
|
|
6403da3715 | ||
|
|
90cb76312c | ||
|
|
b6cff01abb | ||
|
|
b34b711161 | ||
|
|
709d767d61 | ||
|
|
e075916154 | ||
|
|
de10154f29 | ||
|
|
ba71e79e54 | ||
|
|
2cc70b8051 | ||
|
|
9d965f94de | ||
|
|
40c2db4744 | ||
|
|
9a7285dca4 | ||
|
|
8c00ac78b1 | ||
|
|
cf4478eeee | ||
|
|
85c8e7f1bb | ||
|
|
efd95efb40 | ||
|
|
99ad7ea45c | ||
|
|
aba0c57f73 | ||
|
|
8c4f6b648e | ||
|
|
3b83bdd88d | ||
|
|
8d71e08b44 | ||
|
|
c31063a8ef | ||
|
|
28d101f1dd | ||
|
|
aaef344ae3 | ||
|
|
42d0324304 | ||
|
|
2a0019347a | ||
|
|
e0305ea1b9 | ||
|
|
105ff47ae3 | ||
|
|
5ee190a41e | ||
|
|
cf37617c25 | ||
|
|
2e9c8f0f51 | ||
|
|
06c2319851 | ||
|
|
da0668c7cc | ||
|
|
b34df334cb | ||
|
|
9e9f2ccae1 | ||
|
|
15b8f75730 | ||
|
|
9d6b9aa574 | ||
|
|
64724886af | ||
|
|
c088369f4a | ||
|
|
39dbf46422 | ||
|
|
3c1b0bb917 | ||
|
|
2227170dbb | ||
|
|
e08f421e1c | ||
|
|
42c58c5420 | ||
|
|
4afbdd9afb | ||
|
|
06b9e13904 | ||
|
|
29473b43cb | ||
|
|
0427d48b98 | ||
|
|
b228746f1d | ||
|
|
0be8485116 | ||
|
|
5d0279ff08 | ||
|
|
5d908d902c | ||
|
|
3a014f80f2 | ||
|
|
fbefd7855c | ||
|
|
11f9135be6 | ||
|
|
7bb0ce810e | ||
|
|
993b832771 | ||
|
|
013ac1e627 | ||
|
|
d7977a02fa | ||
|
|
73a32ff22c | ||
|
|
ee4ae5390b | ||
|
|
f71db11035 | ||
|
|
9497288b97 | ||
|
|
a00260d801 | ||
|
|
cad48e07e4 | ||
|
|
d91e8a4278 | ||
|
|
faf74eee90 | ||
|
|
0b52e1cd14 | ||
|
|
b62890f136 | ||
|
|
de1d37eef8 | ||
|
|
a57c557485 | ||
|
|
3c9f6c845b | ||
|
|
ff25e9a92e | ||
|
|
6a60f72a9e | ||
|
|
94b690df43 | ||
|
|
94edbc3436 | ||
|
|
5eab1e559a | ||
|
|
53db3ad6f2 | ||
|
|
581f3263ed | ||
|
|
1e3c642be6 | ||
|
|
22c3cd553f | ||
|
|
e5743f8dae | ||
|
|
bddc2f227d | ||
|
|
686c04ea93 | ||
|
|
b38369199e | ||
|
|
e862c904a9 | ||
|
|
43d9384b1c | ||
|
|
cb9af0b86a | ||
|
|
b8c17a843c | ||
|
|
7ad7f181d7 | ||
|
|
eb0bf55033 | ||
|
|
f4e3e4ad30 | ||
|
|
5ae82410cc | ||
|
|
2febb524e9 | ||
|
|
b9e452133c | ||
|
|
747ea0a1f7 | ||
|
|
f8c52ca34a | ||
|
|
663fd5a98b | ||
|
|
93e58bc15e | ||
|
|
3ccdf6508e | ||
|
|
fd33cf1ce5 | ||
|
|
2bb64ad1c6 | ||
|
|
e3de62058b | ||
|
|
6ee9984280 | ||
|
|
e1df548ae9 | ||
|
|
79a685c15e | ||
|
|
a61ab2803c | ||
|
|
209ad27332 | ||
|
|
df08981475 | ||
|
|
5ce6039a02 | ||
|
|
6c3fdf723a | ||
|
|
1a617c1eb2 | ||
|
|
3a9b801400 | ||
|
|
6e663acdad | ||
|
|
ae1023bb7a | ||
|
|
9cf25e276d | ||
|
|
8db3670ecc | ||
|
|
baee367532 | ||
|
|
8da4e72d87 | ||
|
|
b4a84a2317 | ||
|
|
43d6347212 | ||
|
|
438501e49c | ||
|
|
c034e99aaf | ||
|
|
d2d0ca2de9 | ||
|
|
cbe2b442b2 | ||
|
|
c5b1cd6e7d | ||
|
|
8d20d1dae3 | ||
|
|
1f2d702c4e | ||
|
|
d2d35d0553 | ||
|
|
1e7f54dd7e | ||
|
|
8430a2f7e6 | ||
|
|
326cde78e6 | ||
|
|
bbc2b0b42f | ||
|
|
231a2c54aa | ||
|
|
36ae4cee73 | ||
|
|
62e5ee2201 | ||
|
|
b89ebd931e | ||
|
|
d1b4ddaf61 | ||
|
|
734a0b236b | ||
|
|
2229c04d4d | ||
|
|
07cff27fa2 | ||
|
|
2cf86998bc | ||
|
|
6ed15a6fd6 | ||
|
|
7610243b0c | ||
|
|
e11349b577 | ||
|
|
9b425697cb | ||
|
|
42596ff91e | ||
|
|
b3c2ff47f3 | ||
|
|
75a0bc79be | ||
|
|
5a002ad08d | ||
|
|
fbac6f86d1 | ||
|
|
ca18bf2a3d | ||
|
|
199effdff7 | ||
|
|
8212f4b7fb | ||
|
|
e1a45a2720 | ||
|
|
bd7edd8651 | ||
|
|
4e1d10a46f | ||
|
|
70b6bc2bae | ||
|
|
46d019fe02 | ||
|
|
d56f689e15 | ||
|
|
90702b4102 | ||
|
|
0cf105b64a | ||
|
|
5c74d9458c | ||
|
|
75391bf834 | ||
|
|
63304a1d88 | ||
|
|
1ab79bd72e | ||
|
|
985bdf2b6c | ||
|
|
b57ea83aea | ||
|
|
d716e22476 | ||
|
|
bbb8e1ccab | ||
|
|
96f20779c6 | ||
|
|
908313e378 | ||
|
|
12e5c60633 | ||
|
|
11946ffc4c | ||
|
|
f44cd9c545 | ||
|
|
9c5ccb13de | ||
|
|
e9bcfd4784 | ||
|
|
fd1e8d4566 | ||
|
|
68dcce0739 | ||
|
|
3c554cd787 | ||
|
|
9e5f2269d9 | ||
|
|
81fc502c6c | ||
|
|
eda8ca5449 | ||
|
|
35dd8972e2 | ||
|
|
643dd56b74 | ||
|
|
b748eab4ed | ||
|
|
f4eaab6977 | ||
|
|
900c114831 | ||
|
|
7937b7e52d | ||
|
|
b40e707771 | ||
|
|
edde5c8516 | ||
|
|
197facb845 | ||
|
|
ca697d0d5d | ||
|
|
32ddf790d5 | ||
|
|
2cfba8c6d0 | ||
|
|
cf6c0765fa | ||
|
|
ad93f27271 | ||
|
|
92428e5cbc | ||
|
|
02c00a87dd | ||
|
|
6c77bc12c1 | ||
|
|
fbb428c249 | ||
|
|
854a741ea4 | ||
|
|
ed1952a79a | ||
|
|
39e8f5122f | ||
|
|
6ee77eb1a1 | ||
|
|
46fc45b952 | ||
|
|
79d90f3c7f | ||
|
|
eb41cb2261 | ||
|
|
87e8a9b6aa | ||
|
|
69f5aa8e35 | ||
|
|
d654f55c3c | ||
|
|
eb1689e79f | ||
|
|
cf6472df90 | ||
|
|
bc6295a78d | ||
|
|
ed5774b88f | ||
|
|
40a29ca9a7 | ||
|
|
a1e1838b11 | ||
|
|
4731dbab2f | ||
|
|
f2841ccb5e | ||
|
|
81a474fe12 | ||
|
|
5af2477d00 | ||
|
|
4cbba94a29 | ||
|
|
2a57428314 | ||
|
|
f9ac57205b | ||
|
|
3c6246f99a | ||
|
|
072e7bd241 | ||
|
|
4080dca816 | ||
|
|
c22fb80129 | ||
|
|
123fcd4bea | ||
|
|
e0c17672f2 | ||
|
|
53d484c5f4 | ||
|
|
a2ea767b77 | ||
|
|
4bd51a4be0 | ||
|
|
474f2dc267 | ||
|
|
033310234e | ||
|
|
10835c4b62 | ||
|
|
3e741df77b | ||
|
|
31d4b4a201 | ||
|
|
678985cbf5 | ||
|
|
e83e9cb6b3 | ||
|
|
48a51679b8 | ||
|
|
987fc925ee | ||
|
|
b6cbb23781 | ||
|
|
cbb9017c12 | ||
|
|
832d1e8d2c | ||
|
|
34af7f942b | ||
|
|
5f390c16be | ||
|
|
f36ac0498d | ||
|
|
18bdd9b665 | ||
|
|
5fd3852fd3 | ||
|
|
764432c77c | ||
|
|
55c43229ca | ||
|
|
dbc4daaaa4 | ||
|
|
acdbbec78d | ||
|
|
3247e777c0 | ||
|
|
c6e60ff3f5 | ||
|
|
b7df102919 | ||
|
|
f5d450b95c | ||
|
|
972aaf9f5f | ||
|
|
b98d5f30c0 | ||
|
|
d2e2e189d8 | ||
|
|
17d110575c | ||
|
|
a967976420 | ||
|
|
7e1e1ccccc | ||
|
|
502452602e | ||
|
|
f8ff46f3e3 | ||
|
|
8f572deaed | ||
|
|
601dd1d2b5 | ||
|
|
673e826e46 | ||
|
|
f1f81f9de2 | ||
|
|
8513c02ec1 | ||
|
|
320c5f1847 | ||
|
|
282f091e85 | ||
|
|
8647033029 | ||
|
|
3b88947cbe | ||
|
|
f05be0120e | ||
|
|
10c3b2fb5d | ||
|
|
74ae89865c | ||
|
|
a2445904e7 | ||
|
|
bd5188e7bc | ||
|
|
d89c119d9d | ||
|
|
802f3bed47 | ||
|
|
9c606a278d | ||
|
|
7aa5bc0503 | ||
|
|
d94b9fae94 | ||
|
|
05fcaaa758 | ||
|
|
9626a64340 | ||
|
|
c1cfd4db83 | ||
|
|
bfda91ac16 | ||
|
|
b7117b86ea | ||
|
|
a94059ad81 | ||
|
|
2cab87fbd0 | ||
|
|
2017100ff3 | ||
|
|
72cb29d36b | ||
|
|
a2b0d594fb | ||
|
|
a16d4ff1f3 | ||
|
|
305b1ecf2d | ||
|
|
3cc0cae249 | ||
|
|
137aa59254 | ||
|
|
5f8cb0dc54 | ||
|
|
45978474f3 | ||
|
|
75206c7a51 | ||
|
|
06ff0a45a7 | ||
|
|
c948d532a7 | ||
|
|
3825483f79 | ||
|
|
0b238d942e | ||
|
|
16b09a9dd3 | ||
|
|
4f6800b768 | ||
|
|
c84801adb3 | ||
|
|
69f990c1f8 | ||
|
|
8a83564678 | ||
|
|
43d93f836f | ||
|
|
7c2c0f7fe5 | ||
|
|
3ff47cf677 | ||
|
|
3165dce89e | ||
|
|
d4fad0a3b0 | ||
|
|
15fcaa794f | ||
|
|
6a1a505336 | ||
|
|
85937a7bfb | ||
|
|
3baa598b9e | ||
|
|
6f83b10af6 | ||
|
|
f52bcb49ab | ||
|
|
9cef8ff7ce | ||
|
|
a499ad6404 | ||
|
|
7c8767ab32 | ||
|
|
38a8559943 | ||
|
|
48c6acbc87 | ||
|
|
b5d93bc4a0 | ||
|
|
a9fffe771b | ||
|
|
69f3ab14b9 | ||
|
|
a7b9a34ec9 | ||
|
|
f45be1f59e | ||
|
|
012c2ba851 | ||
|
|
a0c2ce0068 | ||
|
|
86898035da | ||
|
|
05eeffe0f7 | ||
|
|
70cffce5eb | ||
|
|
0258e914e0 | ||
|
|
c35b81d2c5 | ||
|
|
fc63dcedb6 | ||
|
|
c879650c4b | ||
|
|
c4d019939b | ||
|
|
3841cc4aa5 | ||
|
|
468f2f2d2d | ||
|
|
be5db91c13 | ||
|
|
6e7d0a520a | ||
|
|
e37996873a | ||
|
|
25e851b570 | ||
|
|
d41b21a626 | ||
|
|
d65c54ba21 | ||
|
|
a5d3bf0f48 | ||
|
|
f76e8c7185 | ||
|
|
96145e8e3f | ||
|
|
530821fc4a | ||
|
|
e4a4a529dd | ||
|
|
863d1e0007 | ||
|
|
c64d6f2b36 | ||
|
|
a798880ac8 | ||
|
|
9db9be9612 | ||
|
|
6a19c77184 | ||
|
|
c17b9dea30 | ||
|
|
7564b6c69a | ||
|
|
5858c5041f | ||
|
|
18d0dec416 | ||
|
|
752046108c | ||
|
|
773ddbf5c4 | ||
|
|
88e90a6f3a | ||
|
|
84325d6b5c | ||
|
|
1b8e03e1de | ||
|
|
569d7297f0 | ||
|
|
d9520c9498 | ||
|
|
6b61093037 | ||
|
|
74b09d5581 | ||
|
|
08da8f6f5c | ||
|
|
7f8fffbb24 | ||
|
|
b08e5f9821 | ||
|
|
1758648f56 | ||
|
|
418ce0aed9 | ||
|
|
4e926076c9 | ||
|
|
3f0aaeba68 | ||
|
|
6ecec77a74 | ||
|
|
79917dc9bb | ||
|
|
b0b60fb761 | ||
|
|
7463152f50 | ||
|
|
cbd8217c60 | ||
|
|
4de1762c21 | ||
|
|
73962402df | ||
|
|
d90ccceec7 | ||
|
|
0c5ef1ce1d | ||
|
|
e5680e031d | ||
|
|
eee0cffa57 | ||
|
|
c480b0ba41 | ||
|
|
e7a47a647c | ||
|
|
500c82bbf8 | ||
|
|
008528af91 | ||
|
|
13199ad203 | ||
|
|
533dd1a54b | ||
|
|
b6f2e10416 | ||
|
|
c58db36a7c |
No files matched your search
@@ -7,3 +7,6 @@ FEXCore/Source/Interface/Core/X86Tables/*
|
||||
# Inline headers with list-like content that can't be processed individually
|
||||
Source/Tools/LinuxEmulation/LinuxSyscalls/x*/SyscallsNames.inl
|
||||
Source/Tools/LinuxEmulation/LinuxSyscalls/x*/Ioctl/*.inl
|
||||
|
||||
# Include files in unittests
|
||||
unittests/*ASM/Includes/*.inc
|
||||
@@ -0,0 +1,79 @@
|
||||
name: steamrt4 build
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
pull_request:
|
||||
branches:
|
||||
- main
|
||||
|
||||
env:
|
||||
BUILD_TYPE: Release
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
|
||||
jobs:
|
||||
steamrt4_build:
|
||||
runs-on: ${{ matrix.arch }}
|
||||
strategy:
|
||||
matrix:
|
||||
arch: [[self-hosted, ARM64, distrobox]]
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
|
||||
- name : submodule checkout
|
||||
run: |
|
||||
git submodule sync --recursive
|
||||
git submodule update --init --depth 1
|
||||
|
||||
- name: Clean Build Environment
|
||||
run: |
|
||||
rm -Rf ${{runner.workspace}}/build
|
||||
cmake -E make_directory ${{runner.workspace}}/build
|
||||
|
||||
# Setup everything required.
|
||||
- name : distrobox setup
|
||||
run: |
|
||||
distrobox create -Y -i registry.gitlab.steamos.cloud/steamrt/steamrt4/sdk/arm64:4.0.20251117.183306 steamrt4 || true
|
||||
distrobox upgrade steamrt4
|
||||
distrobox enter --name steamrt4 -- sudo apt-get install -y \
|
||||
git cmake ninja-build ccache \
|
||||
lld clang \
|
||||
libclang-dev llvm-dev \
|
||||
libstdc++-14-dev-i386-cross libgcc-14-dev-i386-cross \
|
||||
libstdc++-14-dev-amd64-cross libgcc-14-dev-amd64-cross
|
||||
|
||||
- name: Create Build Environment
|
||||
run: distrobox enter --name steamrt4 -- cmake -E make_directory ${{runner.workspace}}/build
|
||||
|
||||
- name: Configure CMake
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: distrobox enter --name steamrt4 -- cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DBUILD_STEAM_SUPPORT=True -DENABLE_LTO=True -DENABLE_ASSERTIONS=False -DBUILD_THUNKS=True -DBUILD_FEXCONFIG=False -DBUILD_TESTING=False -DENABLE_CLANG_THUNKS=True -DUSE_LINKER=lld -DCMAKE_INSTALL_PREFIX=/usr
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: distrobox enter --name steamrt4 -- cmake --build . --config $BUILD_TYPE
|
||||
|
||||
- name: install
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
env:
|
||||
DESTDIR: ${{runner.workspace}}/install
|
||||
run: distrobox enter --name steamrt4 -- cmake --build . --config $BUILD_TYPE -t install
|
||||
- name: Upload libraries
|
||||
uses: 'actions/upload-artifact@v4'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
overwrite: true
|
||||
name: steamrt4_steampipe_depot
|
||||
path: ${{runner.workspace}}/install/*
|
||||
retention-days: 1
|
||||
compression-level: 9
|
||||
+15
-3
@@ -33,6 +33,7 @@ option(ENABLE_FEXCORE_PROFILER "Enables use of the FEXCore timeline profiling ca
|
||||
set (FEXCORE_PROFILER_BACKEND "gpuvis" CACHE STRING "Set which backend to use for the FEXCore profiler (gpuvis, tracy)")
|
||||
option(ENABLE_GLIBC_ALLOCATOR_HOOK_FAULT "Enables glibc memory allocation hooking with fault for CI testing")
|
||||
option(USE_PDB_DEBUGINFO "Builds debug info in PDB format" FALSE)
|
||||
option(BUILD_STEAM_SUPPORT "Builds FEX for integration into Steam" FALSE)
|
||||
|
||||
set (X86_32_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/toolchain_x86_32.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting i686")
|
||||
set (X86_64_TOOLCHAIN_FILE "${CMAKE_CURRENT_SOURCE_DIR}/Data/CMake/toolchain_x86_64.cmake" CACHE FILEPATH "Toolchain file for the (cross-)compiler targeting x86_64")
|
||||
@@ -64,6 +65,10 @@ if (NOT MINGW_BUILD)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (BUILD_STEAM_SUPPORT)
|
||||
add_definitions(-DFEX_STEAM_SUPPORT=1)
|
||||
endif()
|
||||
|
||||
if (ENABLE_FEXCORE_PROFILER)
|
||||
add_definitions(-DENABLE_FEXCORE_PROFILER=1)
|
||||
string(TOUPPER "${FEXCORE_PROFILER_BACKEND}" FEXCORE_PROFILER_BACKEND)
|
||||
@@ -319,7 +324,7 @@ if (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
|
||||
endif()
|
||||
|
||||
find_package(PkgConfig REQUIRED)
|
||||
find_package(Python 3.0 REQUIRED COMPONENTS Interpreter)
|
||||
find_package(Python 3.9 REQUIRED COMPONENTS Interpreter)
|
||||
|
||||
set(BUILD_SHARED_LIBS OFF)
|
||||
|
||||
@@ -476,13 +481,16 @@ add_subdirectory(FEXHeaderUtils/)
|
||||
add_subdirectory(CodeEmitter/)
|
||||
add_subdirectory(FEXCore/)
|
||||
|
||||
if (_M_ARM_64 AND NOT MINGW_BUILD)
|
||||
if (_M_ARM_64 AND NOT MINGW_BUILD AND NOT BUILD_STEAM_SUPPORT)
|
||||
# Binfmt_misc files must be installed prior to Source/ installs
|
||||
add_subdirectory(Data/binfmts/)
|
||||
endif()
|
||||
|
||||
add_subdirectory(Source/)
|
||||
add_subdirectory(Data/AppConfig/)
|
||||
|
||||
if (NOT BUILD_STEAM_SUPPORT)
|
||||
add_subdirectory(Data/AppConfig/)
|
||||
endif()
|
||||
|
||||
# Install the ThunksDB file
|
||||
file(GLOB CONFIG_SOURCES CONFIGURE_DEPENDS ${CMAKE_CURRENT_SOURCE_DIR}/Data/*.json)
|
||||
@@ -582,6 +590,10 @@ if (BUILD_THUNKS)
|
||||
add_dependencies(uninstall uninstall_guest-libs-32)
|
||||
endif()
|
||||
|
||||
if (BUILD_STEAM_SUPPORT)
|
||||
add_subdirectory(Source/Steam/)
|
||||
endif()
|
||||
|
||||
set(FEX_VERSION_MAJOR "0")
|
||||
set(FEX_VERSION_MINOR "0")
|
||||
set(FEX_VERSION_PATCH "0")
|
||||
|
||||
@@ -38,9 +38,7 @@ public:
|
||||
|
||||
[[nodiscard]] BranchEncodeSucceeded adr(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(IsADRRange(Imm), "Unscaled offset too large");
|
||||
|
||||
if (IsADRRange(Imm)) [[likely]] {
|
||||
if (IsADRRange(Imm)) {
|
||||
constexpr uint32_t Op = 0b0001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
@@ -73,9 +71,8 @@ public:
|
||||
|
||||
[[nodiscard]] BranchEncodeSucceeded adrp(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
int64_t Imm = reinterpret_cast<int64_t>(Label->Location) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
|
||||
LOGMAN_THROW_A_FMT(IsADRPRange(Imm) && IsADRPAligned(Imm), "Unscaled offset too large");
|
||||
|
||||
if (IsADRPRange(Imm) && IsADRPAligned(Imm)) [[likely]] {
|
||||
if (IsADRPRange(Imm) && IsADRPAligned(Imm)) {
|
||||
constexpr uint32_t Op = 0b1001'0000 << 24;
|
||||
DataProcessing_PCRel_Imm(Op, rd, Imm);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
@@ -103,16 +100,22 @@ public:
|
||||
}
|
||||
|
||||
[[nodiscard]] BranchEncodeSucceeded LongAddressGen(ARMEmitter::Register rd, const BackwardLabel* Label) {
|
||||
int64_t Imm = reinterpret_cast<int64_t>(Label->Location) - (GetCursorAddress<int64_t>());
|
||||
const auto SLocation = reinterpret_cast<int64_t>(Label->Location);
|
||||
const auto ULocation = std::bit_cast<uint64_t>(SLocation);
|
||||
|
||||
const int64_t Imm = SLocation - (GetCursorAddress<int64_t>());
|
||||
const auto UImm = std::bit_cast<uint64_t>(Imm);
|
||||
|
||||
if (IsADRRange(Imm)) {
|
||||
// If the range is in ADR range then we can just use ADR.
|
||||
return adr(rd, Label);
|
||||
} else if (IsADRPRange(Imm)) {
|
||||
int64_t ADRPImm = (reinterpret_cast<int64_t>(Label->Location) & ~0xFFFLL) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
|
||||
}
|
||||
if (IsADRPRange(Imm)) {
|
||||
const int64_t ADRPImm = (SLocation & ~0xFFFLL) - (GetCursorAddress<int64_t>() & ~0xFFFLL);
|
||||
|
||||
// If the range is in the ADRP range then we can use ADRP.
|
||||
bool NeedsOffset = !IsADRPAligned(reinterpret_cast<uint64_t>(Label->Location));
|
||||
uint64_t AlignedOffset = reinterpret_cast<uint64_t>(Label->Location) & 0xFFFULL;
|
||||
const bool NeedsOffset = !IsADRPAligned(ULocation);
|
||||
const uint64_t AlignedOffset = ULocation & 0xFFFULL;
|
||||
|
||||
// First emit ADRP
|
||||
adrp(rd, ADRPImm >> 12);
|
||||
@@ -125,14 +128,19 @@ public:
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
|
||||
// Can't encode.
|
||||
return BranchEncodeSucceeded::Failure;
|
||||
// Stinky path, we need to load the address as a sequence of movz+movk+movk
|
||||
movz(ARMEmitter::Size::i64Bit, rd, (UImm >> 32) & 0xFFFF, 32);
|
||||
movk(ARMEmitter::Size::i64Bit, rd, (UImm >> 16) & 0xFFFF, 16);
|
||||
movk(ARMEmitter::Size::i64Bit, rd, UImm & 0xFFFF);
|
||||
|
||||
return BranchEncodeSucceeded::Success;
|
||||
}
|
||||
[[nodiscard]] BranchEncodeSucceeded LongAddressGen(ARMEmitter::Register rd, ForwardLabel* Label) {
|
||||
AddLocationToLabel(Label, ForwardLabel::Reference {.Location = GetCursorAddress<uint8_t*>(), .Type = ForwardLabel::InstType::LONG_ADDRESS_GEN});
|
||||
// Emit a register index and a nop. These will be backpatched.
|
||||
// Emit a register index and two nops. These will be backpatched.
|
||||
dc32(rd.Idx());
|
||||
nop();
|
||||
nop();
|
||||
|
||||
// Forward label doesn't know if it can encode until Bind.
|
||||
return BranchEncodeSucceeded::Success;
|
||||
|
||||
@@ -22,7 +22,7 @@ public:
|
||||
}
|
||||
[[nodiscard]] BranchEncodeSucceeded b(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) [[likely]] {
|
||||
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) {
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 0, Cond, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
@@ -55,7 +55,7 @@ public:
|
||||
}
|
||||
[[nodiscard]] BranchEncodeSucceeded bc(ARMEmitter::Condition Cond, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) [[likely]] {
|
||||
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) {
|
||||
constexpr uint32_t Op = 0b0101'010 << 25;
|
||||
Branch_Conditional(Op, 0, 1, Cond, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
@@ -116,7 +116,7 @@ public:
|
||||
}
|
||||
[[nodiscard]] BranchEncodeSucceeded b(const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
if (Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0)) [[likely]] {
|
||||
if (Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0)) {
|
||||
constexpr uint32_t Op = 0b0001'01 << 26;
|
||||
UnconditionalBranch(Op, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
@@ -151,7 +151,7 @@ public:
|
||||
|
||||
[[nodiscard]] BranchEncodeSucceeded bl(const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
if (Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0)) [[likely]] {
|
||||
if (Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0)) {
|
||||
constexpr uint32_t Op = 0b1001'01 << 26;
|
||||
UnconditionalBranch(Op, Imm >> 2);
|
||||
|
||||
@@ -189,7 +189,7 @@ public:
|
||||
[[nodiscard]] BranchEncodeSucceeded cbz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
|
||||
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) [[likely]] {
|
||||
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) {
|
||||
constexpr uint32_t Op = 0b0011'0100 << 24;
|
||||
CompareAndBranch(Op, s, rt, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
@@ -227,7 +227,7 @@ public:
|
||||
[[nodiscard]] BranchEncodeSucceeded cbnz(ARMEmitter::Size s, ARMEmitter::Register rt, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
|
||||
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) [[likely]] {
|
||||
if (Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0)) {
|
||||
constexpr uint32_t Op = 0b0011'0101 << 24;
|
||||
CompareAndBranch(Op, s, rt, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
@@ -265,7 +265,7 @@ public:
|
||||
[[nodiscard]] BranchEncodeSucceeded tbz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
|
||||
if (Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0)) [[likely]] {
|
||||
if (Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0)) {
|
||||
constexpr uint32_t Op = 0b0011'0110 << 24;
|
||||
TestAndBranch(Op, rt, Bit, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
@@ -301,9 +301,8 @@ public:
|
||||
}
|
||||
[[nodiscard]] BranchEncodeSucceeded tbnz(ARMEmitter::Register rt, uint32_t Bit, const BackwardLabel* Label) {
|
||||
int32_t Imm = static_cast<int32_t>(Label->Location - GetCursorAddress<uint8_t*>());
|
||||
LOGMAN_THROW_A_FMT(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0), "Unscaled offset too large");
|
||||
|
||||
if (Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0)) [[likely]] {
|
||||
if (Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0)) {
|
||||
constexpr uint32_t Op = 0b0011'0111 << 24;
|
||||
TestAndBranch(Op, rt, Bit, Imm >> 2);
|
||||
return BranchEncodeSucceeded::Success;
|
||||
|
||||
@@ -586,6 +586,10 @@ concept IsXOrWRegister = std::is_same_v<T, XRegister> || std::is_same_v<T, WRegi
|
||||
template<typename T>
|
||||
concept IsQOrDRegister = std::is_same_v<T, QRegister> || std::is_same_v<T, DRegister>;
|
||||
|
||||
template<typename T>
|
||||
concept IsLabel = std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>;
|
||||
|
||||
enum class BranchEncodeSucceeded {
|
||||
Success,
|
||||
Failure,
|
||||
@@ -658,7 +662,7 @@ public:
|
||||
case ForwardLabel::InstType::ADR: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
if (!IsADRRange(Imm)) [[unlikely]] {
|
||||
if (!IsADRRange(Imm)) {
|
||||
// Can't bind.
|
||||
return false;
|
||||
}
|
||||
@@ -674,7 +678,7 @@ public:
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
|
||||
if (!(IsADRPRange(Imm) && IsADRPAligned(Imm))) [[unlikely]] {
|
||||
if (!(IsADRPRange(Imm) && IsADRPAligned(Imm))) {
|
||||
// Can't bind.
|
||||
return false;
|
||||
}
|
||||
@@ -691,7 +695,7 @@ public:
|
||||
case ForwardLabel::InstType::B: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
if (!(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0))) [[unlikely]] {
|
||||
if (!(Imm >= -134217728 && Imm <= 134217724 && ((Imm & 0b11) == 0))) {
|
||||
// Can't bind.
|
||||
return false;
|
||||
}
|
||||
@@ -707,7 +711,7 @@ public:
|
||||
case ForwardLabel::InstType::TEST_BRANCH: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
if (!(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0))) [[unlikely]] {
|
||||
if (!(Imm >= -32768 && Imm <= 32764 && ((Imm & 0b11) == 0))) {
|
||||
// Can't bind.
|
||||
return false;
|
||||
}
|
||||
@@ -724,7 +728,7 @@ public:
|
||||
case ForwardLabel::InstType::RELATIVE_LOAD: {
|
||||
uint32_t* Instruction = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t Imm = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(Instruction);
|
||||
if (!(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0))) [[unlikely]] {
|
||||
if (!(Imm >= -1048576 && Imm <= 1048575 && ((Imm & 0b11) == 0))) {
|
||||
// Can't bind.
|
||||
return false;
|
||||
}
|
||||
@@ -737,38 +741,44 @@ public:
|
||||
break;
|
||||
}
|
||||
case ForwardLabel::InstType::LONG_ADDRESS_GEN: {
|
||||
uint32_t* Instructions = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
int64_t ImmInstOne = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[0]);
|
||||
int64_t ImmInstTwo = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[1]);
|
||||
auto OriginalOffset = GetCursorOffset();
|
||||
const auto* Instructions = reinterpret_cast<uint32_t*>(Label->Location);
|
||||
const auto ImmInstOne = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[0]);
|
||||
const auto ImmInstTwo = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[1]);
|
||||
const auto ImmInstThree = reinterpret_cast<int64_t>(CurrentAddress) - reinterpret_cast<int64_t>(&Instructions[2]);
|
||||
const auto OriginalOffset = GetCursorOffset();
|
||||
|
||||
auto InstOffset = GetCursorOffsetFromAddress(Instructions);
|
||||
const auto InstOffset = GetCursorOffsetFromAddress(Instructions);
|
||||
SetCursorOffset(InstOffset);
|
||||
|
||||
// We encoded the destination register in to the first instruction space.
|
||||
// Read it back.
|
||||
ARMEmitter::Register DestReg(Instructions[0]);
|
||||
|
||||
if (IsADRRange(ImmInstTwo)) {
|
||||
// If within ADR range from the second instruction, then we can emit NOP+ADR
|
||||
if (IsADRRange(ImmInstThree)) {
|
||||
// If within ADR range from the third instruction, then we can emit NOP+NOP+ADR
|
||||
nop();
|
||||
adr(DestReg, static_cast<uint32_t>(ImmInstTwo) & 0x7FFF);
|
||||
} else if (IsADRPRange(ImmInstOne)) {
|
||||
nop();
|
||||
adr(DestReg, static_cast<uint32_t>(ImmInstThree) & 0x7FFF);
|
||||
} else if (IsADRPRange(ImmInstTwo)) {
|
||||
|
||||
// If within ADRP range from the first instruction, then we are /definitely/ in range for the second instruction.
|
||||
// First check if we are in non-offset range for second instruction.
|
||||
if (IsADRPAligned(reinterpret_cast<uint64_t>(CurrentAddress))) {
|
||||
// We can emit nop + adrp
|
||||
// We can emit nop + nop + adrp
|
||||
nop();
|
||||
nop();
|
||||
adrp(DestReg, static_cast<uint32_t>(ImmInstThree >> 12) & 0x7FFF);
|
||||
} else {
|
||||
// Not aligned, need nop + adrp + add
|
||||
nop();
|
||||
adrp(DestReg, static_cast<uint32_t>(ImmInstTwo >> 12) & 0x7FFF);
|
||||
} else {
|
||||
// Not aligned, need adrp + add
|
||||
adrp(DestReg, static_cast<uint32_t>(ImmInstOne >> 12) & 0x7FFF);
|
||||
add(ARMEmitter::Size::i64Bit, DestReg, DestReg, ImmInstOne & 0xFFF);
|
||||
add(ARMEmitter::Size::i64Bit, DestReg, DestReg, ImmInstTwo & 0xFFF);
|
||||
}
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unscaled offset is too large");
|
||||
FEX_UNREACHABLE;
|
||||
// Stinky path, we need to emit a movz+movk+movk sequence.
|
||||
movz(ARMEmitter::Size::i64Bit, DestReg, uint32_t(ImmInstOne >> 32) & 0x7FFF, 32);
|
||||
movk(ARMEmitter::Size::i64Bit, DestReg, uint32_t(ImmInstOne >> 16) & 0xFFFF, 16);
|
||||
movk(ARMEmitter::Size::i64Bit, DestReg, uint32_t(ImmInstOne) & 0xFFFF);
|
||||
}
|
||||
|
||||
SetCursorOffset(OriginalOffset);
|
||||
|
||||
@@ -5125,7 +5125,7 @@ private:
|
||||
requires (std::is_same_v<T, float> || std::is_same_v<T, double>)
|
||||
[[nodiscard]]
|
||||
static bool IsValidFPValueForImm8(T value) {
|
||||
const uint64_t bits = FEXCore::BitCast<FloatToEquivalentUInt<T>>(value);
|
||||
const uint64_t bits = std::bit_cast<FloatToEquivalentUInt<T>>(value);
|
||||
const uint64_t datasize_idx = FEXCore::ilog2(sizeof(T)) - 1;
|
||||
|
||||
static constexpr std::array mantissa_masks {
|
||||
@@ -5171,7 +5171,7 @@ protected:
|
||||
LOGMAN_THROW_A_FMT(IsValidFPValueForImm8(value), "Value ({}) cannot be encoded into an 8-bit immediate", value);
|
||||
#endif
|
||||
|
||||
const auto bits = FEXCore::BitCast<uint32_t>(value);
|
||||
const auto bits = std::bit_cast<uint32_t>(value);
|
||||
const auto sign = (bits & 0x80000000) >> 24;
|
||||
const auto expb2 = (bits & 0x20000000) >> 23;
|
||||
const auto b5_to_0 = (bits >> 19) & 0x3F;
|
||||
@@ -5184,7 +5184,7 @@ protected:
|
||||
LOGMAN_THROW_A_FMT(IsValidFPValueForImm8(value), "Value ({}) cannot be encoded into an 8-bit immediate", value);
|
||||
#endif
|
||||
|
||||
const auto bits = FEXCore::BitCast<uint64_t>(value);
|
||||
const auto bits = std::bit_cast<uint64_t>(value);
|
||||
const auto sign = (bits & 0x80000000'00000000) >> 56;
|
||||
const auto expb2 = (bits & 0x20000000'00000000) >> 55;
|
||||
const auto b5_to_0 = (bits >> 48) & 0x3F;
|
||||
|
||||
@@ -2,8 +2,8 @@
|
||||
|
||||
let
|
||||
toolchain = pkgs.fetchzip {
|
||||
url = "https://github.com/bylaws/llvm-mingw/releases/download/20250305/llvm-mingw-20250305-ucrt-ubuntu-20.04-aarch64.tar.xz";
|
||||
sha256 = "sha256-cA03/ab9O61eO9+S2JzIXD4V0HzTXK5/AYyxW2d73Po=";
|
||||
url = "https://github.com/bylaws/llvm-mingw/releases/download/20250920/llvm-mingw-20250920-ucrt-ubuntu-22.04-aarch64.tar.xz";
|
||||
sha256 = "sha256-LaojKjC8KzY+soW5u6eoDoXE3qtYk9Ejr7M3enTqRAE=";
|
||||
};
|
||||
|
||||
cmakeToolchainFile = pkgs.substitute {
|
||||
|
||||
Vendored
+1
-1
Submodule External/Catch2 updated: 8ac8190e49...b3fb4b9fea.
Vendored
+1
-1
Submodule External/drm-headers updated: 0675d2f291...3e49836995.
Vendored
+1
-1
Submodule External/fmt updated: 20c8fdad06...407c905e45.
Vendored
+1
-1
Submodule External/xxhash updated: bbb27a5efb...e626a72bc2.
@@ -74,9 +74,11 @@ add_compile_options($<$<COMPILE_LANGUAGE:CXX>:-fno-strict-aliasing> $<$<COMPILE_
|
||||
|
||||
add_subdirectory(Source/)
|
||||
|
||||
install (DIRECTORY include/FEXCore ${CMAKE_BINARY_DIR}/include/FEXCore
|
||||
DESTINATION include
|
||||
COMPONENT Development)
|
||||
if (NOT BUILD_STEAM_SUPPORT)
|
||||
install (DIRECTORY include/FEXCore ${CMAKE_BINARY_DIR}/include/FEXCore
|
||||
DESTINATION include
|
||||
COMPONENT Development)
|
||||
endif()
|
||||
|
||||
if (BUILD_TESTING)
|
||||
add_subdirectory(unittests/)
|
||||
|
||||
@@ -200,6 +200,15 @@ def print_man_environment_tail():
|
||||
],
|
||||
"''", True)
|
||||
|
||||
print_man_env_option(
|
||||
"APP_CACHE_LOCATION",
|
||||
[
|
||||
"Allows the user to override where FEX stores and loads cache files",
|
||||
"By default FEX will look in $XDG_CACHE_HOME/fex-emu/ or $HOME/.cache/fex-emu/",
|
||||
"This will override the full path, trailing forward-slash is expected to exist",
|
||||
],
|
||||
"''", True)
|
||||
|
||||
def print_man_header():
|
||||
header ='''.Dd {0}
|
||||
.Dt FEX
|
||||
|
||||
@@ -58,10 +58,10 @@ class OpDefinition:
|
||||
JITDispatch: bool
|
||||
JITDispatchOverride: str
|
||||
TiedSource: int
|
||||
Inline: list
|
||||
Arguments: list
|
||||
EmitValidation: list
|
||||
Desc: list
|
||||
Inline: list[str]
|
||||
Arguments: list[OpArgument]
|
||||
EmitValidation: list[str]
|
||||
Desc: list[str]
|
||||
|
||||
def __init__(self):
|
||||
self.Name = None
|
||||
@@ -92,19 +92,14 @@ class OpDefinition:
|
||||
attrs = vars(self)
|
||||
print(", ".join("%s: %s" % item for item in attrs.items()))
|
||||
|
||||
IRTypesToCXX = {}
|
||||
CXXTypeToIR = {}
|
||||
IROps = []
|
||||
IRTypesToCXX: dict[str, IRType] = {}
|
||||
CXXTypeToIR: dict[str, IRType] = {}
|
||||
IROps: list[OpDefinition] = []
|
||||
|
||||
IROpNameMap = {}
|
||||
IROpNameSet: set[str] = set()
|
||||
|
||||
def is_ssa_type(type):
|
||||
if (type == "SSA" or
|
||||
type == "GPR" or
|
||||
type == "GPRPair" or
|
||||
type == "FPR"):
|
||||
return True
|
||||
return False
|
||||
def is_ssa_type(op_type: str):
|
||||
return op_type in {"SSA", "GPR", "GPRPair", "FPR"}
|
||||
|
||||
def parse_irtypes(irtypes):
|
||||
for op_key, op_val in irtypes.items():
|
||||
@@ -219,11 +214,8 @@ def parse_ops(ops):
|
||||
OpArg.DefaultInitializer = DefaultInit[1][:-1]
|
||||
|
||||
# If SSA type then we can generate validation for this op
|
||||
if (OpArg.IsSSA and
|
||||
(OpArg.Type == "GPR" or
|
||||
OpArg.Type == "GPRPair" or
|
||||
OpArg.Type == "FPR")):
|
||||
OpDef.EmitValidation.append(f"GetOpRegClass({ArgName}) == InvalidClass || WalkFindRegClass({ArgName}) == {OpArg.Type}Class")
|
||||
if OpArg.IsSSA and OpArg.Type in {"GPR", "GPRPair", "FPR"}:
|
||||
OpDef.EmitValidation.append(f"GetOpRegClass({ArgName}) == RegClass::Invalid || WalkFindRegClass({ArgName}) == RegClass::{OpArg.Type}")
|
||||
|
||||
OpArg.Name = ArgName
|
||||
OpArg.NameWithPrefix = NameWithPrefix
|
||||
@@ -296,22 +288,29 @@ def parse_ops(ops):
|
||||
#OpDef.print()
|
||||
|
||||
# Error on duplicate op
|
||||
if OpDef.Name in IROpNameMap:
|
||||
if OpDef.Name in IROpNameSet:
|
||||
ExitError("Duplicate Op defined! {}".format(OpDef.Name))
|
||||
|
||||
IROps.append(OpDef)
|
||||
IROpNameMap[OpDef.Name] = 1
|
||||
IROpNameSet.add(OpDef.Name)
|
||||
|
||||
# Print out enum values
|
||||
def print_enums():
|
||||
def print_enums(enums):
|
||||
output_file.write("#ifdef IROP_ENUM\n")
|
||||
output_file.write("enum IROps : uint16_t {\n")
|
||||
|
||||
for op in IROps:
|
||||
output_file.write("\tOP_{},\n" .format(op.Name.upper()))
|
||||
|
||||
output_file.write("};\n")
|
||||
|
||||
for name, members in enums.items():
|
||||
output_file.write(f"enum {name} {{\n")
|
||||
for member in members:
|
||||
if member:
|
||||
output_file.write(f"\t{member}\n")
|
||||
else:
|
||||
output_file.write("\n")
|
||||
output_file.write("};\n\n")
|
||||
|
||||
output_file.write("#undef IROP_ENUM\n")
|
||||
output_file.write("#endif\n\n")
|
||||
|
||||
@@ -408,7 +407,7 @@ def print_ir_sizes():
|
||||
[[nodiscard, gnu::const]] std::string_view const& GetName(IROps Op);
|
||||
[[nodiscard, gnu::const]] uint8_t GetArgs(IROps Op);
|
||||
[[nodiscard, gnu::const]] uint8_t GetRAArgs(IROps Op);
|
||||
[[nodiscard, gnu::const]] FEXCore::IR::RegisterClassType GetRegClass(IROps Op);
|
||||
[[nodiscard, gnu::const]] FEXCore::IR::RegClass GetRegClass(IROps Op);
|
||||
[[nodiscard, gnu::const]] bool HasSideEffects(IROps Op);
|
||||
[[nodiscard, gnu::const]] bool ImplicitFlagClobber(IROps Op);
|
||||
[[nodiscard, gnu::const]] bool GetHasDest(IROps Op);
|
||||
@@ -422,30 +421,29 @@ def print_ir_sizes():
|
||||
def print_ir_reg_classes():
|
||||
output_file.write("#ifdef IROP_REG_CLASSES_IMPL\n")
|
||||
|
||||
output_file.write("constexpr std::array<FEXCore::IR::RegisterClassType, IROps::OP_LAST + 1> IRRegClasses = {\n")
|
||||
output_file.write("constexpr std::array<FEXCore::IR::RegClass, IROps::OP_LAST + 1> IRRegClasses = {\n")
|
||||
for op in IROps:
|
||||
if op.Name == "Last":
|
||||
output_file.write("\tFEXCore::IR::InvalidClass,\n")
|
||||
output_file.write("\tRegClass::Invalid,\n")
|
||||
else:
|
||||
Class = "Invalid"
|
||||
if op.HasDest and op.DestType == None:
|
||||
if op.HasDest and op.DestType is None:
|
||||
ExitError("IR op {} has destination with no destination class".format(op.Name))
|
||||
|
||||
if op.HasDest and op.DestType == "SSA": # Special case SSA type
|
||||
output_file.write("\tFEXCore::IR::ComplexClass,\n")
|
||||
output_file.write("\tRegClass::Complex,\n")
|
||||
elif op.HasDest:
|
||||
output_file.write("\tFEXCore::IR::{}Class,\n".format(op.DestType))
|
||||
output_file.write("\tRegClass::{},\n".format(op.DestType))
|
||||
else:
|
||||
# No destination so it has an invalid destination class
|
||||
output_file.write("\tFEXCore::IR::InvalidClass, // No destination\n")
|
||||
output_file.write("\tRegClass::Invalid, // No destination\n")
|
||||
|
||||
|
||||
output_file.write("};\n\n")
|
||||
|
||||
output_file.write("// Make sure our array maps directly to the IROps enum\n")
|
||||
output_file.write("static_assert(IRRegClasses[IROps::OP_LAST] == FEXCore::IR::InvalidClass);\n\n")
|
||||
output_file.write("static_assert(IRRegClasses[IROps::OP_LAST] == RegClass::Invalid);\n\n")
|
||||
|
||||
output_file.write("FEXCore::IR::RegisterClassType GetRegClass(IROps Op) { return IRRegClasses[Op]; }\n\n")
|
||||
output_file.write("FEXCore::IR::RegClass GetRegClass(IROps Op) { return IRRegClasses[Op]; }\n\n")
|
||||
|
||||
output_file.write("#undef IROP_REG_CLASSES_IMPL\n")
|
||||
output_file.write("#endif\n\n")
|
||||
@@ -568,9 +566,7 @@ def print_ir_arg_printer():
|
||||
|
||||
SSAArgNum = 0
|
||||
FirstArg = True
|
||||
for i in range(0, len(op.Arguments)):
|
||||
arg = op.Arguments[i]
|
||||
|
||||
for arg in op.Arguments:
|
||||
# No point printing temporaries that we can't recover
|
||||
if arg.Temporary:
|
||||
continue
|
||||
@@ -671,7 +667,7 @@ def print_ir_allocator_helpers():
|
||||
output_file.write("\t\treturn HeaderOp->Op;\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("\tFEXCore::IR::RegisterClassType GetOpRegClass(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\tFEXCore::IR::RegClass GetOpRegClass(const OrderedNode *Op) const {\n")
|
||||
output_file.write("\t\treturn GetRegClass(GetOpType(Op));\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
@@ -685,22 +681,21 @@ def print_ir_allocator_helpers():
|
||||
output_file.write("\tIRPair<IROp_{}> _{}(" .format(op.Name, op.Name))
|
||||
|
||||
# Output SSA args first
|
||||
for i in range(0, len(op.Arguments)):
|
||||
arg = op.Arguments[i]
|
||||
LastArg = len(op.Arguments) - i - 1 == 0
|
||||
for i, arg in enumerate(op.Arguments):
|
||||
LastArg = i == len(op.Arguments) - 1
|
||||
|
||||
if arg.Temporary:
|
||||
CType = IRTypesToCXX[arg.Type].CXXName
|
||||
output_file.write("{} {}".format(CType, arg.Name));
|
||||
output_file.write("{} {}".format(CType, arg.Name))
|
||||
elif arg.IsSSA:
|
||||
# SSA value
|
||||
output_file.write("OrderedNodeWrapper {}".format(arg.Name))
|
||||
else:
|
||||
# User defined op that is stored
|
||||
CType = IRTypesToCXX[arg.Type].CXXName
|
||||
output_file.write("{} {}".format(CType, arg.Name));
|
||||
output_file.write("{} {}".format(CType, arg.Name))
|
||||
|
||||
if arg.DefaultInitializer != None:
|
||||
if arg.DefaultInitializer:
|
||||
output_file.write(" = {}".format(arg.DefaultInitializer))
|
||||
|
||||
if not LastArg:
|
||||
@@ -758,20 +753,19 @@ def print_ir_allocator_helpers():
|
||||
if op.SSAArgNum:
|
||||
output_file.write("\tIRPair<IROp_{}> _{}(" .format(op.Name, op.Name))
|
||||
|
||||
for i in range(0, len(op.Arguments)):
|
||||
arg = op.Arguments[i]
|
||||
LastArg = len(op.Arguments) - i - 1 == 0
|
||||
for i, arg in enumerate(op.Arguments):
|
||||
LastArg = i == len(op.Arguments) - 1
|
||||
|
||||
if arg.Temporary:
|
||||
CType = IRTypesToCXX[arg.Type].CXXName
|
||||
output_file.write("{} {}".format(CType, arg.Name));
|
||||
output_file.write("{} {}".format(CType, arg.Name))
|
||||
elif arg.IsSSA:
|
||||
output_file.write("OrderedNode *{}".format(arg.Name))
|
||||
else:
|
||||
CType = IRTypesToCXX[arg.Type].CXXName
|
||||
output_file.write("{} {}".format(CType, arg.Name));
|
||||
output_file.write("{} {}".format(CType, arg.Name))
|
||||
|
||||
if arg.DefaultInitializer != None:
|
||||
if arg.DefaultInitializer:
|
||||
output_file.write(" = {}".format(arg.DefaultInitializer))
|
||||
|
||||
if not LastArg:
|
||||
@@ -812,16 +806,15 @@ def print_ir_allocator_helpers():
|
||||
print_validation(op)
|
||||
|
||||
output_file.write(f"\t\treturn _{op.Name}(")
|
||||
for i in range(0, len(op.Arguments)):
|
||||
arg = op.Arguments[i]
|
||||
LastArg = len(op.Arguments) - i - 1 == 0
|
||||
for i, arg in enumerate(op.Arguments):
|
||||
LastArg = i == len(op.Arguments) - 1
|
||||
output_file.write(arg.Name)
|
||||
if arg.IsSSA:
|
||||
output_file.write("->Wrapped(ListDataBegin)")
|
||||
if not LastArg:
|
||||
output_file.write(", ")
|
||||
output_file.write(");\n");
|
||||
output_file.write("\t}\n\n");
|
||||
output_file.write(");\n")
|
||||
output_file.write("\t}\n\n")
|
||||
|
||||
output_file.write("#undef IROP_ALLOCATE_HELPERS\n")
|
||||
output_file.write("#endif\n")
|
||||
@@ -852,8 +845,8 @@ def print_ir_dispatcher_dispatch():
|
||||
output_dispatch_file.write("#endif\n")
|
||||
|
||||
|
||||
if (len(sys.argv) < 4):
|
||||
ExitError()
|
||||
if len(sys.argv) < 4:
|
||||
ExitError("Insufficient parameters passed to script")
|
||||
|
||||
output_filename = sys.argv[2]
|
||||
output_dispatcher_filename = sys.argv[3]
|
||||
@@ -865,6 +858,7 @@ json_file.close()
|
||||
json_object = json.loads(json_text)
|
||||
json_object = {k.upper(): v for k, v in json_object.items()}
|
||||
|
||||
enums = json_object["ENUMS"]
|
||||
ops = json_object["OPS"]
|
||||
irtypes = json_object["IRTYPES"]
|
||||
defines = json_object["DEFINES"]
|
||||
@@ -874,7 +868,7 @@ parse_ops(ops)
|
||||
|
||||
output_file = open(output_filename, "w")
|
||||
|
||||
print_enums()
|
||||
print_enums(enums)
|
||||
print_ir_structs(defines)
|
||||
print_ir_sizes()
|
||||
print_ir_reg_classes()
|
||||
|
||||
@@ -31,7 +31,6 @@ set (SRCS
|
||||
Interface/Core/OpcodeDispatcher/X87.cpp
|
||||
Interface/Core/OpcodeDispatcher/X87F64.cpp
|
||||
Interface/Core/OpcodeDispatcher.cpp
|
||||
Interface/Core/X86HelperGen.cpp
|
||||
Interface/Core/ArchHelpers/Arm64Emitter.cpp
|
||||
Interface/Core/Dispatcher/Dispatcher.cpp
|
||||
Interface/Core/Interpreter/Fallbacks/InterpreterFallbacks.cpp
|
||||
@@ -202,8 +201,10 @@ add_custom_target(CONFIG_INC
|
||||
DEPENDS "${OUTPUT_MAN_NAME}"
|
||||
DEPENDS "${OUTPUT_MAN_NAME_COMPRESS}")
|
||||
|
||||
# Install the compressed man page
|
||||
install(FILES ${OUTPUT_MAN_NAME_COMPRESS} COMPONENT Runtime DESTINATION ${MAN_DIR}/man1)
|
||||
if (NOT BUILD_STEAM_SUPPORT)
|
||||
# Install the compressed man page
|
||||
install(FILES ${OUTPUT_MAN_NAME_COMPRESS} COMPONENT Runtime DESTINATION ${MAN_DIR}/man1)
|
||||
endif()
|
||||
|
||||
# Add in diagnostic colours if the option is available.
|
||||
# Ninja code generator will kill colours if this isn't here
|
||||
@@ -290,7 +291,7 @@ AddObject(${PROJECT_NAME}_object OBJECT)
|
||||
AddLibrary(${PROJECT_NAME} STATIC)
|
||||
AddLibrary(${PROJECT_NAME}_shared SHARED)
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
if (NOT MINGW_BUILD AND NOT BUILD_STEAM_SUPPORT)
|
||||
install(TARGETS ${PROJECT_NAME}_shared
|
||||
LIBRARY
|
||||
DESTINATION ${CMAKE_INSTALL_LIBDIR}
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <FEXCore/Utils/TypeDefines.h>
|
||||
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
|
||||
@@ -12,7 +13,7 @@ namespace FEXCore {
|
||||
// Buffered JIT symbol tracking.
|
||||
struct JITSymbolBuffer {
|
||||
// Maximum buffer size to ensure we are a page in size.
|
||||
constexpr static size_t BUFFER_SIZE = 4096 - (8 * 2);
|
||||
constexpr static size_t BUFFER_SIZE = FEXCore::Utils::FEX_PAGE_SIZE - (8 * 2);
|
||||
// Maximum distance until the end of the buffer to do a write.
|
||||
constexpr static size_t NEEDS_WRITE_DISTANCE = BUFFER_SIZE - 64;
|
||||
// Maximum time threshhold to wait before a buffer write occurs.
|
||||
@@ -27,7 +28,7 @@ struct JITSymbolBuffer {
|
||||
size_t Offset {};
|
||||
char Buffer[BUFFER_SIZE] {};
|
||||
};
|
||||
static_assert(sizeof(JITSymbolBuffer) == 4096, "Ensure this is one page in size");
|
||||
static_assert(sizeof(JITSymbolBuffer) == FEXCore::Utils::FEX_PAGE_SIZE, "Ensure this is one page in size");
|
||||
|
||||
class JITSymbols final {
|
||||
public:
|
||||
|
||||
@@ -4,9 +4,9 @@
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXHeaderUtils/BitUtils.h>
|
||||
#include "cephes_128bit.h"
|
||||
|
||||
#include <bit>
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
#include <stdint.h>
|
||||
@@ -501,12 +501,12 @@ struct FEX_PACKED X80SoftFloat {
|
||||
|
||||
float ToF32(softfloat_state* state) const {
|
||||
const float32_t Result = extF80_to_f32(state, *this);
|
||||
return FEXCore::BitCast<float>(Result);
|
||||
return std::bit_cast<float>(Result);
|
||||
}
|
||||
|
||||
double ToF64(softfloat_state* state) const {
|
||||
const float64_t Result = extF80_to_f64(state, *this);
|
||||
return FEXCore::BitCast<double>(Result);
|
||||
return std::bit_cast<double>(Result);
|
||||
}
|
||||
|
||||
FEXCore::VectorRegType ToVector() const {
|
||||
@@ -518,7 +518,7 @@ struct FEX_PACKED X80SoftFloat {
|
||||
BIGFLOAT ToFMax(softfloat_state* state) const {
|
||||
#if BIGFLOATSIZE == 16
|
||||
const float128_t Result = extF80_to_f128(state, *this);
|
||||
return FEXCore::BitCast<BIGFLOAT>(Result);
|
||||
return std::bit_cast<BIGFLOAT>(Result);
|
||||
#else
|
||||
BIGFLOAT result {};
|
||||
memcpy(&result, this, sizeof(result));
|
||||
@@ -577,18 +577,18 @@ struct FEX_PACKED X80SoftFloat {
|
||||
}
|
||||
|
||||
X80SoftFloat(softfloat_state* state, const float rhs) {
|
||||
*this = f32_to_extF80(state, FEXCore::BitCast<float32_t>(rhs));
|
||||
*this = f32_to_extF80(state, std::bit_cast<float32_t>(rhs));
|
||||
}
|
||||
|
||||
X80SoftFloat(softfloat_state* state, const double rhs) {
|
||||
*this = f64_to_extF80(state, FEXCore::BitCast<float64_t>(rhs));
|
||||
*this = f64_to_extF80(state, std::bit_cast<float64_t>(rhs));
|
||||
}
|
||||
|
||||
X80SoftFloat(softfloat_state* state, BIGFLOAT rhs) {
|
||||
#if BIGFLOATSIZE == 16
|
||||
*this = f128_to_extF80(state, FEXCore::BitCast<float128_t>(rhs));
|
||||
*this = f128_to_extF80(state, std::bit_cast<float128_t>(rhs));
|
||||
#else
|
||||
*this = FEXCore::BitCast<long double>(rhs);
|
||||
*this = std::bit_cast<long double>(rhs);
|
||||
#endif
|
||||
}
|
||||
|
||||
|
||||
@@ -2,59 +2,23 @@
|
||||
#pragma once
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <concepts>
|
||||
#include <string_view>
|
||||
#include <optional>
|
||||
|
||||
namespace FEXCore::StrConv {
|
||||
inline bool Conv(std::string_view Value, bool* Result) {
|
||||
*Result = std::strtoull(Value.data(), nullptr, 0);
|
||||
template<std::integral T>
|
||||
bool Conv(std::string_view Value, T* Result) {
|
||||
if constexpr (std::is_signed_v<T>) {
|
||||
*Result = static_cast<T>(std::strtoll(Value.data(), nullptr, 0));
|
||||
} else {
|
||||
*Result = static_cast<T>(std::strtoull(Value.data(), nullptr, 0));
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
inline bool Conv(std::string_view Value, uint8_t* Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
inline bool Conv(std::string_view Value, int8_t* Result) {
|
||||
*Result = std::strtol(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
inline bool Conv(std::string_view Value, uint16_t* Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
inline bool Conv(std::string_view Value, int16_t* Result) {
|
||||
*Result = std::strtol(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
inline bool Conv(std::string_view Value, uint32_t* Result) {
|
||||
*Result = std::strtoul(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
inline bool Conv(std::string_view Value, int32_t* Result) {
|
||||
*Result = std::strtol(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
inline bool Conv(std::string_view Value, uint64_t* Result) {
|
||||
*Result = std::strtoull(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
inline bool Conv(std::string_view Value, int64_t* Result) {
|
||||
*Result = std::strtoll(Value.data(), nullptr, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
template<typename T, typename = std::enable_if<std::is_enum<T>::value, T>>
|
||||
inline bool Conv(std::string_view Value, T* Result) {
|
||||
*Result = static_cast<T>(std::stoull(Value.data(), nullptr, 0));
|
||||
template<typename T, typename = std::enable_if_t<std::is_enum_v<T>, T>>
|
||||
bool Conv(std::string_view Value, T* Result) {
|
||||
*Result = static_cast<T>(std::strtoull(Value.data(), nullptr, 0));
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
@@ -30,14 +30,14 @@ class Context;
|
||||
}
|
||||
|
||||
namespace FEXCore::Config {
|
||||
namespace DefaultValues {
|
||||
namespace detail {
|
||||
#define P(x) x
|
||||
#define OPT_BASE(type, group, enum, json, default) const P(type) P(enum) = P(default);
|
||||
#define OPT_STR(group, enum, json, default) const std::string_view P(enum) = P(default);
|
||||
#define OPT_STRARRAY(group, enum, json, default) OPT_STR(group, enum, json, default)
|
||||
#define OPT_STRENUM(group, enum, json, default) const uint64_t P(enum) = FEXCore::ToUnderlying(P(default));
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
} // namespace DefaultValues
|
||||
} // namespace detail
|
||||
|
||||
enum Paths {
|
||||
PATH_DATA_DIR_LOCAL = 0,
|
||||
@@ -134,7 +134,7 @@ public:
|
||||
void Load();
|
||||
|
||||
template<typename T>
|
||||
requires (!std::is_same_v<fextl::string, T> && !std::is_same_v<DefaultValues::Type::StringArrayType, T>)
|
||||
requires (!std::is_same_v<fextl::string, T> && !std::is_same_v<StringArrayType, T>)
|
||||
std::optional<T> GetConv(ConfigOption Option) {
|
||||
const auto it = OptionMap.find(Option);
|
||||
if (it == OptionMap.end()) {
|
||||
@@ -142,7 +142,7 @@ public:
|
||||
}
|
||||
|
||||
const auto& Value = it->second;
|
||||
LOGMAN_THROW_A_FMT(!std::holds_alternative<DefaultValues::Type::StringArrayType>(Value), "Tried to get config of invalid type!");
|
||||
LOGMAN_THROW_A_FMT(!std::holds_alternative<StringArrayType>(Value), "Tried to get config of invalid type!");
|
||||
|
||||
if (std::holds_alternative<T>(Value)) [[likely]] {
|
||||
return std::get<T>(Value);
|
||||
@@ -165,7 +165,7 @@ public:
|
||||
|
||||
private:
|
||||
void MergeConfigMap(const LayerOptions& Options);
|
||||
void MergeEnvironmentVariables(const ConfigOption& Option, const DefaultValues::Type::StringArrayType& Value);
|
||||
void MergeEnvironmentVariables(const ConfigOption& Option, const StringArrayType& Value);
|
||||
};
|
||||
|
||||
void MetaLayer::Load() {
|
||||
@@ -181,7 +181,7 @@ void MetaLayer::Load() {
|
||||
}
|
||||
|
||||
|
||||
void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const DefaultValues::Type::StringArrayType& Value) {
|
||||
void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const StringArrayType& Value) {
|
||||
// Environment variables need a bit of additional work
|
||||
// We want to merge the arrays rather than overwrite entirely
|
||||
auto MetaEnvironment = OptionMap.find(Option);
|
||||
@@ -193,7 +193,7 @@ void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const Defa
|
||||
|
||||
// If an environment variable exists in both current meta and in the incoming layer then the meta layer value is overwritten
|
||||
fextl::unordered_map<fextl::string, fextl::string> LookupMap;
|
||||
const auto AddToMap = [&LookupMap](const DefaultValues::Type::StringArrayType& Value) {
|
||||
const auto AddToMap = [&LookupMap](const StringArrayType& Value) {
|
||||
for (const auto& EnvVar : Value) {
|
||||
const auto ItEq = EnvVar.find_first_of('=');
|
||||
if (ItEq == fextl::string::npos) {
|
||||
@@ -209,7 +209,7 @@ void MetaLayer::MergeEnvironmentVariables(const ConfigOption& Option, const Defa
|
||||
}
|
||||
};
|
||||
|
||||
AddToMap(std::get<DefaultValues::Type::StringArrayType>(MetaEnvironment->second));
|
||||
AddToMap(std::get<StringArrayType>(MetaEnvironment->second));
|
||||
AddToMap(Value);
|
||||
|
||||
// Now with the two layers merged in the map
|
||||
@@ -225,8 +225,8 @@ void MetaLayer::MergeConfigMap(const LayerOptions& Options) {
|
||||
// Insert this layer's options, overlaying previous options that exist here
|
||||
for (auto& it : Options) {
|
||||
if (it.first == FEXCore::Config::ConfigOption::CONFIG_ENV || it.first == FEXCore::Config::ConfigOption::CONFIG_HOSTENV) {
|
||||
LOGMAN_THROW_A_FMT(std::holds_alternative<DefaultValues::Type::StringArrayType>(it.second), "Tried to get config of invalid type!");
|
||||
MergeEnvironmentVariables(it.first, std::get<DefaultValues::Type::StringArrayType>(it.second));
|
||||
LOGMAN_THROW_A_FMT(std::holds_alternative<StringArrayType>(it.second), "Tried to get config of invalid type!");
|
||||
MergeEnvironmentVariables(it.first, std::get<StringArrayType>(it.second));
|
||||
} else {
|
||||
OptionMap.insert_or_assign(it.first, it.second);
|
||||
}
|
||||
@@ -423,7 +423,7 @@ bool Exists(ConfigOption Option) {
|
||||
return Meta->OptionExists(Option);
|
||||
}
|
||||
|
||||
std::optional<DefaultValues::Type::StringArrayType*> All(ConfigOption Option) {
|
||||
std::optional<StringArrayType*> All(ConfigOption Option) {
|
||||
return Meta->All(Option);
|
||||
}
|
||||
|
||||
@@ -436,6 +436,12 @@ std::optional<T> GetConv(ConfigOption Option) {
|
||||
return Meta->GetConv<T>(Option);
|
||||
}
|
||||
|
||||
template std::optional<bool> GetConv(ConfigOption Option);
|
||||
template std::optional<uint8_t> GetConv(ConfigOption Option);
|
||||
template std::optional<int32_t> GetConv(ConfigOption Option);
|
||||
template std::optional<uint32_t> GetConv(ConfigOption Option);
|
||||
template std::optional<uint64_t> GetConv(ConfigOption Option);
|
||||
|
||||
void Set(ConfigOption Option, std::string_view Data) {
|
||||
Meta->Set(Option, Data);
|
||||
}
|
||||
@@ -491,13 +497,12 @@ template Value<uint8_t>::Value(FEXCore::Config::ConfigOption _Option, uint8_t De
|
||||
template Value<uint64_t>::Value(FEXCore::Config::ConfigOption _Option, uint64_t Default);
|
||||
|
||||
template<typename T>
|
||||
void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, DefaultValues::Type::StringArrayType* List) {
|
||||
void Value<T>::GetListIfExists(FEXCore::Config::ConfigOption Option, StringArrayType* List) {
|
||||
auto Value = FEXCore::Config::All(Option);
|
||||
List->clear();
|
||||
if (Value) {
|
||||
*List = **Value;
|
||||
}
|
||||
}
|
||||
template void Value<DefaultValues::Type::StringArrayType>::GetListIfExists(FEXCore::Config::ConfigOption Option,
|
||||
DefaultValues::Type::StringArrayType* List);
|
||||
template void Value<StringArrayType>::GetListIfExists(FEXCore::Config::ConfigOption Option, StringArrayType* List);
|
||||
} // namespace FEXCore::Config
|
||||
@@ -16,6 +16,13 @@
|
||||
"Maximum number of instruction to store in a block"
|
||||
]
|
||||
},
|
||||
"EnableCodeCachingWIP": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Enable the code caching subsystem"
|
||||
]
|
||||
},
|
||||
"HostFeatures": {
|
||||
"Type": "strenum",
|
||||
"Default": "FEXCore::Config::HostFeatures::OFF",
|
||||
@@ -59,7 +66,9 @@
|
||||
"ENABLEWFXT": "enablewfxt",
|
||||
"DISABLEWFXT": "disablewfxt",
|
||||
"ENABLE3DNOW": "enable3dnow",
|
||||
"DISABLE3DNOW": "disable3dnow"
|
||||
"DISABLE3DNOW": "disable3dnow",
|
||||
"ENABLESSE4A": "enablesse4a",
|
||||
"DISABLESSE4A": "disablesse4a"
|
||||
},
|
||||
"Desc": [
|
||||
"Allows controlling of the CPU features in the JIT.",
|
||||
@@ -82,7 +91,8 @@
|
||||
"\t{enable,disable}svebitperm: Will force enable or disable svebitperm even if the host doesn't support it",
|
||||
"\t{enable,disable}preserveallabi: Will force enable or disable preserve_all abi even if the host doesn't support it",
|
||||
"\t{enable,disable}wfxt: Will force enable or disable wfxt even if the host doesn't support it",
|
||||
"\t{enable,disable}3dnow: Will force enable or disable 3DNow even if the host doesn't support it"
|
||||
"\t{enable,disable}3dnow: Will force enable or disable 3DNow! even if the host doesn't support it",
|
||||
"\t{enable,disable}sse4a: Will force enable or disable SSE4a even if the host doesn't support it"
|
||||
]
|
||||
},
|
||||
"SmallTSCScale": {
|
||||
@@ -91,6 +101,13 @@
|
||||
"Desc": [
|
||||
"Scales the cycle counter on systems that have low frequencies."
|
||||
]
|
||||
},
|
||||
"CPUFeatureRegisters": {
|
||||
"Type": "str",
|
||||
"Default": "",
|
||||
"Desc": [
|
||||
"Allows overriding cpu feature flags for manual testing"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Emulation": {
|
||||
@@ -158,6 +175,44 @@
|
||||
"Desc": [
|
||||
"Allows the user to pass additional arguments to the application"
|
||||
]
|
||||
},
|
||||
"DisableL2Cache": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Disables FEXCore's JIT L2 cache lookup. Saving memory.",
|
||||
"Can potentially introduce more stutters."
|
||||
]
|
||||
},
|
||||
"DynamicL1Cache": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Switches FEXCore's JIT L1 cache to be dynamically sized. Saving memory.",
|
||||
"Can potentially introduce more stutters."
|
||||
]
|
||||
},
|
||||
"DynamicL1CacheIncreaseCountHeuristic": {
|
||||
"Type": "uint64",
|
||||
"Default": "250",
|
||||
"Desc": [
|
||||
"Threshold of lookups per second that the L1 dynamic cache should increase its size.",
|
||||
"Lower numbers means more aggressive scaling upward to the maximum size.",
|
||||
"Higher numbers means more conservative scaling, using less memory.",
|
||||
"Can potentially introduce stutters, more likely the higher the number.",
|
||||
"Don't have this number smaller than the decrease count!"
|
||||
]
|
||||
},
|
||||
"DynamicL1CacheDecreaseCountHeuristic": {
|
||||
"Type": "uint64",
|
||||
"Default": "50",
|
||||
"Desc": [
|
||||
"Threshold of lookups per second that the L1 dynamic cache should decrease its size.",
|
||||
"The higher the number, the more aggressively it reduces the L1 cache size.",
|
||||
"Lower numbers means more conservative memory savings.",
|
||||
"Can potentially introduce more stutters, more likely the higher the number.",
|
||||
"Don't have this number larger than the increase count!"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Debug": {
|
||||
@@ -328,11 +383,11 @@
|
||||
"Requires a supported version of Mangohud to see the results"
|
||||
]
|
||||
},
|
||||
"TraceProfiler": {
|
||||
"EnableGpuvisProfiling": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Enables FEX's trace profiler. Using gpuvis or tracy"
|
||||
"Enables profiling when FEX was built with the gpuvis profiler backend."
|
||||
]
|
||||
}
|
||||
},
|
||||
@@ -388,12 +443,19 @@
|
||||
"This is required to ensure a split-lock doesn't tear inside the process"
|
||||
]
|
||||
},
|
||||
"KernelUnalignedAtomicBackpatching": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"When the kernel unaligned atomic handler is enabled, use backpatching to reduce kernel context switches."
|
||||
]
|
||||
},
|
||||
"VolatileMetadata": {
|
||||
"Type": "bool",
|
||||
"Default": "true",
|
||||
"Desc": [
|
||||
"Use volatile metadata in PE files to inform TSO instructions when available.",
|
||||
"When metadata is unavailable falls back to the currently enabled TSO options."
|
||||
"When metadata is unavailable falls back to the currently enabled TSO options."
|
||||
]
|
||||
},
|
||||
"X87ReducedPrecision": {
|
||||
@@ -403,23 +465,6 @@
|
||||
"Emulates X87 floating point using 64-bit precision. This reduces emulation accuracy and may result in rendering bugs."
|
||||
]
|
||||
},
|
||||
"ABILocalFlags": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"When enabled enables an optimization around flags.",
|
||||
"Assumes flags are not used across cals.",
|
||||
"Hand-written assembly can violate this assumption."
|
||||
]
|
||||
},
|
||||
"ParanoidTSO": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
"Desc": [
|
||||
"Makes TSO operations even more strict.",
|
||||
"Forces vector loadstores to also become atomic."
|
||||
]
|
||||
},
|
||||
"StallProcess": {
|
||||
"Type": "bool",
|
||||
"Default": "false",
|
||||
|
||||
@@ -4,7 +4,6 @@
|
||||
#include "Common/JitSymbols.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include <Interface/IR/IntrusiveIRList.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/Context.h>
|
||||
@@ -29,6 +28,7 @@
|
||||
namespace FEXCore {
|
||||
class SignalDelegator;
|
||||
class ThunkHandler;
|
||||
struct LookupCacheWriteLockToken;
|
||||
|
||||
namespace Core {
|
||||
struct DebugData;
|
||||
@@ -61,7 +61,7 @@ struct CustomIRResult {
|
||||
, Data(Data) {}
|
||||
};
|
||||
|
||||
using BlockDelinkerFunc = void (*)(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record);
|
||||
using BlockDelinkerFunc = void (*)(FEXCore::Context::ExitFunctionLinkData* Record);
|
||||
constexpr uint32_t TSC_SCALE_MAXIMUM = 1'000'000'000; ///< 1Ghz
|
||||
|
||||
class CodeCache : public AbstractCodeCache {
|
||||
@@ -72,12 +72,34 @@ public:
|
||||
ContextImpl& CTX;
|
||||
bool IsGeneratingCache = false;
|
||||
|
||||
uint64_t ComputeCodeMapId(std::string_view Filename, int FD) override;
|
||||
|
||||
void LoadData(Core::InternalThreadState&, std::byte* MappedCacheFile, const ExecutableFileSectionInfo&) override;
|
||||
bool SaveData(Core::InternalThreadState&, int TargetFD, const ExecutableFileSectionInfo&, uint64_t SerializedBaseAddress) override;
|
||||
|
||||
void InitiateCacheGeneration() override {
|
||||
IsGeneratingCache = true;
|
||||
}
|
||||
|
||||
/**
|
||||
* Applies a set of FEX relocations to the given code section.
|
||||
*
|
||||
* FEX relocations describe runtime-dependencies of FEX-generated code.
|
||||
* When loading a code cache, they are used to move cached code to the
|
||||
* dynamically chosen base address of the guest binary.
|
||||
*
|
||||
* Conversely, relocations are applied in reverse when writing code caches
|
||||
* to ensure consistency across generation runs.
|
||||
*
|
||||
* Note that FEX relocations are unrelated to ELF/PE relocations.
|
||||
*
|
||||
* @param GuestDelta Guest address offset to apply to RIP-relative data
|
||||
* @param ForStorage True for serializing data (producing deterministic output); false for de-serializing it (resolving dynamic symbols)
|
||||
*
|
||||
* @return Returns true on success
|
||||
*/
|
||||
[[nodiscard]]
|
||||
bool ApplyCodeRelocations(uint64_t GuestDelta, std::span<std::byte> Code, std::span<const CPU::Relocation> Relocations, bool ForStorage);
|
||||
};
|
||||
|
||||
class ContextImpl final : public FEXCore::Context::Context, public CPU::CodeBufferManager {
|
||||
@@ -154,10 +176,20 @@ public:
|
||||
return CodeCache;
|
||||
}
|
||||
|
||||
void OnCodeBufferAllocated(CPU::CodeBuffer&) override;
|
||||
void SetCodeMapWriter(fextl::unique_ptr<CodeMapWriter> Writer) override {
|
||||
CodeMapWriter = std::move(Writer);
|
||||
}
|
||||
|
||||
void FlushAndCloseCodeMap() override {
|
||||
if (CodeMapWriter) {
|
||||
CodeMapWriter.reset();
|
||||
}
|
||||
}
|
||||
|
||||
void OnCodeBufferAllocated(const std::shared_ptr<CPU::CodeBuffer>&) override;
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer = true) override;
|
||||
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, InvalidatedEntryAccumulator& Accumulator, uint64_t Start,
|
||||
uint64_t Length) override;
|
||||
void InvalidateCodeBuffersCodeRange(uint64_t Start, uint64_t Length) override;
|
||||
void InvalidateThreadCachedCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) override;
|
||||
FEXCore::ForkableSharedMutex& GetCodeInvalidationMutex() override {
|
||||
return CodeInvalidationMutex;
|
||||
}
|
||||
@@ -197,7 +229,6 @@ public:
|
||||
FEX_CONFIG_OPT(TSOEnabled, TSOENABLED);
|
||||
FEX_CONFIG_OPT(VectorTSOEnabled, VECTORTSOENABLED);
|
||||
FEX_CONFIG_OPT(MemcpySetTSOEnabled, MEMCPYSETTSOENABLED);
|
||||
FEX_CONFIG_OPT(ABILocalFlags, ABILOCALFLAGS);
|
||||
FEX_CONFIG_OPT(SMCChecks, SMCCHECKS);
|
||||
FEX_CONFIG_OPT(MaxInstPerBlock, MAXINST);
|
||||
FEX_CONFIG_OPT(RootFSPath, ROOTFS);
|
||||
@@ -205,7 +236,6 @@ public:
|
||||
FEX_CONFIG_OPT(LibraryJITNaming, LIBRARYJITNAMING);
|
||||
FEX_CONFIG_OPT(BlockJITNaming, BLOCKJITNAMING);
|
||||
FEX_CONFIG_OPT(GDBSymbols, GDBSYMBOLS);
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
|
||||
FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY);
|
||||
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
|
||||
@@ -226,14 +256,12 @@ public:
|
||||
FEXCore::ThunkHandler* ThunkHandler {};
|
||||
fextl::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
|
||||
CodeCache CodeCache;
|
||||
fextl::unique_ptr<CodeMapWriter> CodeMapWriter;
|
||||
|
||||
SignalDelegator* SignalDelegation {};
|
||||
X86GeneratedCode X86CodeGen;
|
||||
|
||||
ContextImpl(const FEXCore::HostFeatures& Features);
|
||||
|
||||
static bool ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP);
|
||||
|
||||
static void ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP);
|
||||
|
||||
// This is used as a replacement for the SMC writes in the mono callsite backpatcher that avoids atomic operations
|
||||
@@ -268,9 +296,9 @@ public:
|
||||
|
||||
FEXCore::JITSymbols Symbols;
|
||||
|
||||
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator;
|
||||
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator;
|
||||
FEXCore::Utils::PooledAllocatorVirtual CPUBackendAllocator;
|
||||
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator {"FEXMem_OpDispatcher"};
|
||||
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator {"FEXMem_Frontend"};
|
||||
FEXCore::Utils::PooledAllocatorVirtualWithGuard CPUBackendAllocator {"FEXMem_CPUBackend"};
|
||||
|
||||
// If Atomic-based TSO emulation is enabled or not.
|
||||
bool IsAtomicTSOEnabled() const {
|
||||
@@ -311,10 +339,6 @@ protected:
|
||||
AtomicTSOEmulationEnabled = false;
|
||||
VectorAtomicTSOEmulationEnabled = false;
|
||||
MemcpyAtomicTSOEmulationEnabled = false;
|
||||
} else if (Config.ParanoidTSO) {
|
||||
AtomicTSOEmulationEnabled = true;
|
||||
VectorAtomicTSOEmulationEnabled = true;
|
||||
MemcpyAtomicTSOEmulationEnabled = true;
|
||||
} else {
|
||||
AtomicTSOEmulationEnabled = Config.TSOEnabled;
|
||||
VectorAtomicTSOEmulationEnabled = Config.TSOEnabled && Config.VectorTSOEnabled;
|
||||
@@ -353,5 +377,8 @@ private:
|
||||
|
||||
bool MonoDetected = false;
|
||||
std::atomic<uint64_t> MonoBackpatcherBlock;
|
||||
|
||||
std::mutex CodeBufferListLock;
|
||||
fextl::vector<std::weak_ptr<CPU::CodeBuffer>> CodeBufferList;
|
||||
};
|
||||
} // namespace FEXCore::Context
|
||||
@@ -7,7 +7,7 @@
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage) {
|
||||
Ref LoadEffectiveAddress(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage) {
|
||||
Ref Tmp = A.Base;
|
||||
|
||||
if (A.Offset) {
|
||||
@@ -51,8 +51,8 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, b
|
||||
return Tmp ?: IREmit->Constant(0);
|
||||
}
|
||||
|
||||
AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO, bool Vector,
|
||||
IR::OpSize AccessSize) {
|
||||
AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO,
|
||||
bool Vector, IR::OpSize AccessSize) {
|
||||
const auto Is32Bit = GPRSize == OpSize::i32Bit;
|
||||
const auto GPRSizeMatchesAddrSize = A.AddrSize == GPRSize;
|
||||
const auto OffsetIndexToLargeFor32Bit = Is32Bit && (A.Offset <= -16384 || A.Offset >= 16384);
|
||||
@@ -103,7 +103,7 @@ AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSi
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(IREmit, B, GPRSize, true /* AddSegmentBase */, false),
|
||||
.Index = IREmit->Constant(A.Offset),
|
||||
.IndexType = MEM_OFFSET_SXTX,
|
||||
.IndexType = MemOffsetType::SXTX,
|
||||
.IndexScale = 1,
|
||||
};
|
||||
}
|
||||
@@ -111,15 +111,17 @@ AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSi
|
||||
if (AtomicTSO) {
|
||||
// TODO: LRCPC3 support for vector Imm9.
|
||||
} else if (!Is32Bit && A.Base && (A.Index || A.Segment) && !A.Offset && (A.IndexScale == 1 || A.IndexScale == AccessSizeAsImm)) {
|
||||
AddressMode B = A;
|
||||
|
||||
// ScaledRegisterLoadstore
|
||||
if (A.Index && A.Segment) {
|
||||
A.Base = IREmit->Add(GPRSize, A.Base, A.Segment);
|
||||
} else if (A.Segment) {
|
||||
A.Index = A.Segment;
|
||||
A.IndexScale = 1;
|
||||
if (B.Index && B.Segment) {
|
||||
B.Base = IREmit->Add(GPRSize, B.Base, B.Segment);
|
||||
} else if (B.Segment) {
|
||||
B.Index = B.Segment;
|
||||
B.IndexScale = 1;
|
||||
}
|
||||
|
||||
return A;
|
||||
return B;
|
||||
}
|
||||
|
||||
if (Vector || !AtomicTSO) {
|
||||
@@ -134,7 +136,7 @@ AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSi
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(IREmit, B, GPRSize, true /* AddSegmentBase */, false),
|
||||
.Index = IREmit->Constant(A.Offset),
|
||||
.IndexType = MEM_OFFSET_SXTX,
|
||||
.IndexType = MemOffsetType::SXTX,
|
||||
.IndexScale = 1,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -11,17 +11,18 @@ struct AddressMode {
|
||||
Ref Segment {nullptr};
|
||||
Ref Base {nullptr};
|
||||
Ref Index {nullptr};
|
||||
MemOffsetType IndexType = MEM_OFFSET_SXTX;
|
||||
uint8_t IndexScale = 1;
|
||||
int64_t Offset = 0;
|
||||
|
||||
MemOffsetType IndexType = MemOffsetType::SXTX;
|
||||
uint8_t IndexScale = 1;
|
||||
|
||||
// Size in bytes for the address calculation. 8 for an arm64 hardware mode.
|
||||
IR::OpSize AddrSize;
|
||||
bool NonTSO;
|
||||
};
|
||||
|
||||
Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage = false);
|
||||
AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO, bool Vector,
|
||||
IR::OpSize AccessSize);
|
||||
Ref LoadEffectiveAddress(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool AddSegmentBase, bool AllowUpperGarbage = false);
|
||||
AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO,
|
||||
bool Vector, IR::OpSize AccessSize);
|
||||
|
||||
}; // namespace FEXCore::IR
|
||||
} // namespace FEXCore::IR
|
||||
@@ -1,10 +1,10 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "FEXCore/Core/X86Enums.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
|
||||
@@ -1,30 +1,31 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include "FEXCore/Utils/EnumUtils.h"
|
||||
#include "Interface/Core/JIT/Relocations.h"
|
||||
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
#include <aarch64/disasm-aarch64.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#endif
|
||||
#ifdef VIXL_SIMULATOR
|
||||
#include <aarch64/simulator-aarch64.h>
|
||||
#include <aarch64/simulator-constants-aarch64.h>
|
||||
#endif
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
#include <CodeEmitter/Registers.h>
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <optional>
|
||||
#include <span>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
namespace FEXCore::X86State {
|
||||
enum X86Reg : uint32_t;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
// Contains the address to the currently available CPU state
|
||||
@@ -104,9 +105,12 @@ constexpr ARMEmitter::PRegister PRED_TMP_32B = ARMEmitter::PReg::p7;
|
||||
// This class contains common emitter utility functions that can
|
||||
// be used by both Arm64 JIT and ARM64 Dispatcher
|
||||
class Arm64Emitter : public ARMEmitter::Emitter {
|
||||
protected:
|
||||
public:
|
||||
Arm64Emitter(FEXCore::Context::ContextImpl* ctx, void* EmissionPtr = nullptr, size_t size = 0);
|
||||
|
||||
void LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad = false);
|
||||
|
||||
protected:
|
||||
FEXCore::Context::ContextImpl* EmitterCTX;
|
||||
|
||||
std::span<const ARMEmitter::Register> StaticRegisters {};
|
||||
@@ -116,8 +120,6 @@ protected:
|
||||
std::span<const ARMEmitter::VRegister> GeneralFPRegisters {};
|
||||
uint32_t PairRegisters = 0;
|
||||
|
||||
void LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, bool NOPPad = false);
|
||||
|
||||
void FillSpecialRegs(ARMEmitter::Register TmpReg, ARMEmitter::Register TmpReg2, bool SetFIZ, bool SetPredRegs);
|
||||
|
||||
// Correlate an ARM register back to an x86 register index.
|
||||
|
||||
@@ -1,14 +1,17 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "FEXCore/Utils/AllocatorHooks.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/AllocatorHooks.h>
|
||||
#include <FEXCore/Utils/PrctlUtils.h>
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
#include "LookupCache.h"
|
||||
|
||||
#ifndef _WIN32
|
||||
#include <linux/prctl.h>
|
||||
#include <sys/prctl.h>
|
||||
#endif
|
||||
|
||||
@@ -357,6 +360,8 @@ namespace CPU {
|
||||
LogMan::Msg::EFmt("Failed to mprotect last page of code buffer.");
|
||||
}
|
||||
|
||||
FEXCore::Allocator::VirtualName("FEXMemJIT", reinterpret_cast<void*>(Ptr), Size);
|
||||
|
||||
LookupCache = fextl::make_unique<GuestToHostMap>();
|
||||
}
|
||||
|
||||
@@ -395,7 +400,7 @@ namespace CPU {
|
||||
Latest = Buffer;
|
||||
LatestOffset = 0;
|
||||
|
||||
OnCodeBufferAllocated(*Buffer);
|
||||
OnCodeBufferAllocated(Buffer);
|
||||
|
||||
return Buffer;
|
||||
}
|
||||
|
||||
@@ -81,7 +81,7 @@ namespace CPU {
|
||||
// Protects writes to the latest CodeBuffer and changes to LatestOffset
|
||||
FEXCore::ForkableUniqueMutex CodeBufferWriteMutex;
|
||||
|
||||
virtual void OnCodeBufferAllocated(CodeBuffer&) {};
|
||||
virtual void OnCodeBufferAllocated(const std::shared_ptr<CodeBuffer>&) {};
|
||||
|
||||
private:
|
||||
fextl::shared_ptr<CodeBuffer> Latest;
|
||||
@@ -161,7 +161,7 @@ namespace CPU {
|
||||
virtual CompiledCode CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, bool CheckTF) = 0;
|
||||
|
||||
virtual fextl::vector<FEXCore::CPU::Relocation> TakeRelocations() = 0;
|
||||
virtual fextl::vector<FEXCore::CPU::Relocation> TakeRelocations(uint64_t GuestBaseAddress) = 0;
|
||||
|
||||
virtual void ClearCache() {}
|
||||
|
||||
|
||||
@@ -14,6 +14,7 @@ $end_info$
|
||||
#include <FEXCore/Core/CPUID.h>
|
||||
#include <FEXCore/Core/HostFeatures.h>
|
||||
#include <FEXCore/Utils/FileLoading.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
@@ -88,6 +89,7 @@ namespace ProductNames {
|
||||
static const char ARM_Blizzard_M2Pro[] = "Apple Blizzard (M2 Pro)";
|
||||
static const char ARM_Avalanche_M2Max[] = "Apple Avalanche (M2 Max)";
|
||||
static const char ARM_Blizzard_M2Max[] = "Apple Blizzard (M2 Max)";
|
||||
static const char ARM_AppleSilicon[] = "Apple Silicon";
|
||||
|
||||
static const char ARM_ORYON_1[] = "Oryon-1";
|
||||
static const char ARM_Ampere_1[] = "AmpereOne";
|
||||
@@ -188,6 +190,7 @@ void CPUIDEmu::SetupHostHybridFlag() {
|
||||
{0x61, 0x029, 1, ProductNames::ARM_Firestorm_M1Max}, // Apple Firestorm (M1 Max)
|
||||
{0x61, 0x025, 1, ProductNames::ARM_Firestorm_M1Pro}, // Apple Firestorm (M1 Pro)
|
||||
{0x61, 0x023, 1, ProductNames::ARM_Firestorm_M1}, // Apple Firestorm (M1)
|
||||
{0x61, 0, 1, ProductNames::ARM_AppleSilicon}, // QEmu Apple Silicon
|
||||
|
||||
{0x41, 0xd8c, 1, ProductNames::ARM_C1Ultra}, // C1-Ultra
|
||||
{0x41, 0xd90, 1, ProductNames::ARM_C1Premium}, // C1-Premium
|
||||
@@ -441,10 +444,10 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
|
||||
|
||||
Res.eax = FAMILY_IDENTIFIER;
|
||||
|
||||
Res.ebx = 0 | // Brand index
|
||||
(8 << 8) | // Cache line size in bytes
|
||||
(Cores << 16) | // Number of addressable IDs for the logical cores in the physical CPU
|
||||
(0 << 24); // Local APIC ID
|
||||
Res.ebx = 0 | // Brand index
|
||||
(8 << 8) | // Cache line size in bytes
|
||||
(Cores << 16) | // Number of addressable IDs for the logical cores in the physical CPU
|
||||
(GetCPUID() << 24); // Local APIC ID
|
||||
|
||||
Res.ecx = (1 << 0) | // SSE3
|
||||
(CTX->HostFeatures.SupportsPMULL_128Bit << 1) | // PCLMULQDQ
|
||||
@@ -507,7 +510,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
|
||||
(1 << 25) | // SSE
|
||||
(1 << 26) | // SSE2
|
||||
(0 << 27) | // Self Snoop
|
||||
(1 << 28) | // Max APIC IDs reserved field is valid
|
||||
(0 << 28) | // (HTT) Max APIC IDs reserved field is valid
|
||||
(1 << 29) | // Thermal monitor
|
||||
(0 << 30) | // Reserved
|
||||
(0 << 31); // Pending break enable
|
||||
@@ -909,38 +912,38 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0001h(uint32_t Leaf) con
|
||||
|
||||
Res.eax = FAMILY_IDENTIFIER;
|
||||
|
||||
Res.ecx = (1 << 0) | // LAHF/SAHF
|
||||
(1 << 1) | // 0 = Single core product, 1 = multi core product
|
||||
(0 << 2) | // SVM
|
||||
(1 << 3) | // Extended APIC register space
|
||||
(0 << 4) | // LOCK MOV CR0 means MOV CR8
|
||||
(1 << 5) | // ABM instructions
|
||||
(0 << 6) | // SSE4a
|
||||
(0 << 7) | // Misaligned SSE mode
|
||||
(1 << 8) | // PREFETCHW
|
||||
(0 << 9) | // OS visible workaround support
|
||||
(0 << 10) | // Instruction based sampling support
|
||||
(0 << 11) | // XOP
|
||||
(0 << 12) | // SKINIT
|
||||
(0 << 13) | // Watchdog timer support
|
||||
(0 << 14) | // Reserved
|
||||
(0 << 15) | // Lightweight profiling support
|
||||
(0 << 16) | // FMA4
|
||||
(1 << 17) | // Translation cache extension
|
||||
(0 << 18) | // Reserved
|
||||
(0 << 19) | // Reserved
|
||||
(0 << 20) | // Reserved
|
||||
(0 << 21) | // XOP-TBM
|
||||
(0 << 22) | // Topology extensions support
|
||||
(0 << 23) | // Core performance counter extensions
|
||||
(0 << 24) | // NB performance counter extensions
|
||||
(0 << 25) | // Reserved
|
||||
(0 << 26) | // Data breakpoints extensions
|
||||
(0 << 27) | // Performance TSC
|
||||
(0 << 28) | // L2 perf counter extensions
|
||||
(0 << 29) | // MONITORX
|
||||
(0 << 30) | // Reserved
|
||||
(0 << 31); // Reserved
|
||||
Res.ecx = (1 << 0) | // LAHF/SAHF
|
||||
(1 << 1) | // 0 = Single core product, 1 = multi core product
|
||||
(0 << 2) | // SVM
|
||||
(1 << 3) | // Extended APIC register space
|
||||
(0 << 4) | // LOCK MOV CR0 means MOV CR8
|
||||
(1 << 5) | // ABM instructions
|
||||
(CTX->HostFeatures.SupportsSSE4a << 6) | // SSE4a
|
||||
(0 << 7) | // Misaligned SSE mode
|
||||
(1 << 8) | // PREFETCHW
|
||||
(0 << 9) | // OS visible workaround support
|
||||
(0 << 10) | // Instruction based sampling support
|
||||
(0 << 11) | // XOP
|
||||
(0 << 12) | // SKINIT
|
||||
(0 << 13) | // Watchdog timer support
|
||||
(0 << 14) | // Reserved
|
||||
(0 << 15) | // Lightweight profiling support
|
||||
(0 << 16) | // FMA4
|
||||
(1 << 17) | // Translation cache extension
|
||||
(0 << 18) | // Reserved
|
||||
(0 << 19) | // Reserved
|
||||
(0 << 20) | // Reserved
|
||||
(0 << 21) | // XOP-TBM
|
||||
(0 << 22) | // Topology extensions support
|
||||
(0 << 23) | // Core performance counter extensions
|
||||
(0 << 24) | // NB performance counter extensions
|
||||
(0 << 25) | // Reserved
|
||||
(0 << 26) | // Data breakpoints extensions
|
||||
(0 << 27) | // Performance TSC
|
||||
(0 << 28) | // L2 perf counter extensions
|
||||
(0 << 29) | // MONITORX
|
||||
(0 << 30) | // Reserved
|
||||
(0 << 31); // Reserved
|
||||
|
||||
Res.edx = (1 << 0) | // FPU
|
||||
(1 << 1) | // Virtual mode extensions
|
||||
@@ -1094,9 +1097,9 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_8000_0008h(uint32_t Leaf) con
|
||||
(CTX->HostFeatures.SupportsCLZERO << 0); // CLZERO support
|
||||
|
||||
uint32_t CoreCount = Cores - 1;
|
||||
Res.ecx = (0 << 16) | // PerfTscSize: Performance timestamp count size
|
||||
((uint32_t)std::log2(CoreCount + 1) << 12) | // ApicIdSize: Number of bits in ApicID
|
||||
(CoreCount << 0); // Count count subtract one
|
||||
Res.ecx = (0 << 16) | // PerfTscSize: Performance timestamp count size
|
||||
(std::bit_ceil(Cores) << 12) | // ApicIdSize: Number of bits in ApicID
|
||||
(CoreCount << 0); // Count count subtract one
|
||||
|
||||
return Res;
|
||||
}
|
||||
|
||||
@@ -277,7 +277,7 @@ private:
|
||||
// 0: Highest function parameter and ID
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 1: Processor info
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
{SupportsConstant::NONCONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 2: Cache and TLB info
|
||||
{SupportsConstant::CONSTANT, NeedsLeafConstant::NOLEAFCONSTANT},
|
||||
// 3: Serial Number(previously), now reserved
|
||||
|
||||
@@ -1,12 +1,213 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <Interface/Context/Context.h>
|
||||
#include "Utils/SpinWaitLock.h"
|
||||
|
||||
#include <Interface/Context/Context.h>
|
||||
#include <Interface/Core/ArchHelpers/Arm64Emitter.h>
|
||||
#include <Interface/Core/JIT/Relocations.h>
|
||||
#include <Interface/Core/LookupCache.h>
|
||||
|
||||
#include <FEXCore/Core/Thunks.h>
|
||||
#include <FEXCore/HLE/SourcecodeResolver.h>
|
||||
|
||||
#include <FEXHeaderUtils/Filesystem.h>
|
||||
|
||||
#include <git_version.h>
|
||||
|
||||
#include <xxhash.h>
|
||||
|
||||
#include <fstream>
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
#if __clang_major__ < 16
|
||||
ExecutableFileInfo::ExecutableFileInfo(fextl::unique_ptr<HLE::SourcecodeMap> Map, uint64_t FileId, fextl::string Filename)
|
||||
: SourcecodeMap(std::move(Map))
|
||||
, FileId(FileId)
|
||||
, Filename(Filename) {}
|
||||
#endif
|
||||
ExecutableFileInfo::~ExecutableFileInfo() = default;
|
||||
|
||||
fextl::string CodeMap::GetBaseFilename(const ExecutableFileInfo& MainExecutable, bool AddNombSuffix) {
|
||||
auto FileId = MainExecutable.FileId;
|
||||
|
||||
std::string_view base_filename = FHU::Filesystem::GetFilename(std::string_view {MainExecutable.Filename});
|
||||
if (FileId != 0xffff'ffff'ffff'ffff) {
|
||||
return fextl::fmt::format("{}-{:016x}{}", base_filename, MainExecutable.FileId, AddNombSuffix ? "-nomb" : "");
|
||||
}
|
||||
|
||||
return "";
|
||||
}
|
||||
|
||||
fextl::map<CodeMapFileId, CodeMap::ParsedContents> CodeMap::ParseCodeMap(std::ifstream& File) {
|
||||
fextl::map<CodeMapFileId, CodeMap::ParsedContents> Ret;
|
||||
while (true) {
|
||||
Entry Entry;
|
||||
File.read(reinterpret_cast<char*>(&Entry), sizeof(Entry));
|
||||
if (!File) {
|
||||
break;
|
||||
}
|
||||
|
||||
if (Entry.FileId == LoadExternalLibrary.FileId && Entry.BlockOffset == LoadExternalLibrary.BlockOffset) {
|
||||
ExternalLibraryInfo Info;
|
||||
File.read(reinterpret_cast<char*>(&Info), sizeof(Info));
|
||||
|
||||
fextl::string Filename;
|
||||
std::getline(File, Filename, '\0');
|
||||
|
||||
// Align to 4-byte boundary
|
||||
char Null[4];
|
||||
File.read(Null, AlignUp(Filename.size() + 1, 4) - Filename.size() - 1);
|
||||
if (!File) {
|
||||
break;
|
||||
}
|
||||
Ret[Info.ExternalFileId].Filename = std::move(Filename);
|
||||
} else if (Entry.FileId == SetExecutableFileId {}.Marker.FileId && Entry.BlockOffset == SetExecutableFileId {}.Marker.BlockOffset) {
|
||||
CodeMapFileId ExecutableFileId;
|
||||
File.read(reinterpret_cast<char*>(&ExecutableFileId), sizeof(ExecutableFileId));
|
||||
if (!File) {
|
||||
break;
|
||||
}
|
||||
Ret[ExecutableFileId].IsExecutable = true;
|
||||
} else {
|
||||
if (!Ret.contains(Entry.FileId)) {
|
||||
LogMan::Msg::EFmt("Code map referenced unknown file id {:016x}", Entry.FileId);
|
||||
} else {
|
||||
Ret[Entry.FileId].Blocks.insert(Entry.BlockOffset);
|
||||
}
|
||||
}
|
||||
|
||||
if (!File) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
return Ret;
|
||||
}
|
||||
|
||||
CodeMapWriter::CodeMapWriter(CodeMapOpener& Opener, bool OpenEagerly)
|
||||
: Buffer(4096)
|
||||
, FileOpener(Opener) {
|
||||
if (OpenEagerly) {
|
||||
CodeMapFD = FileOpener.OpenCodeMapFile();
|
||||
}
|
||||
}
|
||||
|
||||
CodeMapWriter::~CodeMapWriter() {
|
||||
if (CodeMapFD.value_or(-1) != -1) {
|
||||
Flush(BufferOffset);
|
||||
close(*CodeMapFD);
|
||||
}
|
||||
}
|
||||
|
||||
bool CodeMapWriter::IsWriteEnabled(const ExecutableFileSectionInfo& Section) {
|
||||
if (CodeMapFD == -1) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// PV libraries can't yet be read by FEXServer, so skip dumping them
|
||||
if (Section.FileInfo.Filename.starts_with("/run/pressure-vessel")) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if (CodeMapFD) {
|
||||
return true;
|
||||
}
|
||||
|
||||
// Acquire mutex and re-check CodeMapFD to avoid race conditions
|
||||
auto lk = std::unique_lock {Mutex};
|
||||
if (!CodeMapFD) {
|
||||
CodeMapFD = FileOpener.OpenCodeMapFile();
|
||||
}
|
||||
|
||||
return CodeMapFD != -1;
|
||||
}
|
||||
|
||||
void CodeMapWriter::Flush(size_t Offset) {
|
||||
// Acquire exclusive lock and flush circular buffer
|
||||
std::unique_lock Lock {Mutex};
|
||||
Flush(Offset, Lock);
|
||||
}
|
||||
|
||||
void CodeMapWriter::Flush(size_t Offset, std::unique_lock<std::shared_mutex>&) {
|
||||
write(*CodeMapFD, Buffer.data(), Offset);
|
||||
BufferOffset = 0;
|
||||
}
|
||||
|
||||
void CodeMapWriter::AppendBlock(const FEXCore::ExecutableFileSectionInfo& SectionInfo, uint64_t BlockEntry) {
|
||||
if (!IsWriteEnabled(SectionInfo)) {
|
||||
return;
|
||||
}
|
||||
|
||||
BlockEntry -= SectionInfo.FileStartVA;
|
||||
if (BlockEntry > std::numeric_limits<uint32_t>::max()) {
|
||||
ERROR_AND_DIE_FMT("Cannot write code map");
|
||||
}
|
||||
|
||||
// Register new library if not already known
|
||||
bool NewLibraryLoad = false;
|
||||
{
|
||||
// Check prior registration with shared lock
|
||||
std::shared_lock Lock {Mutex};
|
||||
NewLibraryLoad = !KnownFileIds.contains(SectionInfo.FileInfo.FileId);
|
||||
}
|
||||
if (NewLibraryLoad) {
|
||||
// Register to map with exclusive lock
|
||||
std::unique_lock Lock {Mutex};
|
||||
NewLibraryLoad &= KnownFileIds.insert(SectionInfo.FileInfo.FileId).second;
|
||||
}
|
||||
if (NewLibraryLoad) {
|
||||
// Add entry to code map
|
||||
AppendLibraryLoad(SectionInfo.FileInfo);
|
||||
}
|
||||
|
||||
// Register the actual code block
|
||||
CodeMap::Entry DataEntry {SectionInfo.FileInfo.FileId, static_cast<uint32_t>(BlockEntry)};
|
||||
AppendData(std::as_bytes(std::span {&DataEntry, 1}));
|
||||
}
|
||||
|
||||
void CodeMapWriter::AppendLibraryLoad(const FEXCore::ExecutableFileInfo& FileInfo) {
|
||||
// See CodeMap::ExternalLibraryInfo
|
||||
auto ExternalFileId = FileInfo.FileId;
|
||||
auto TotalSize = AlignUp(sizeof(CodeMap::LoadExternalLibrary) + sizeof(ExternalFileId) + FileInfo.Filename.size() + 1, 4);
|
||||
const auto Data = reinterpret_cast<char*>(alloca(TotalSize));
|
||||
auto WritePtr = std::copy_n(reinterpret_cast<const char*>(&CodeMap::LoadExternalLibrary), sizeof(CodeMap::LoadExternalLibrary), Data);
|
||||
WritePtr = std::copy_n(reinterpret_cast<const char*>(&ExternalFileId), sizeof(ExternalFileId), WritePtr);
|
||||
WritePtr = std::copy(FileInfo.Filename.begin(), FileInfo.Filename.end(), WritePtr);
|
||||
std::fill(WritePtr, Data + TotalSize, 0);
|
||||
AppendData(std::as_bytes(std::span {Data, TotalSize}));
|
||||
}
|
||||
|
||||
void CodeMapWriter::AppendSetMainExecutable(const FEXCore::ExecutableFileInfo& FileInfo) {
|
||||
CodeMap::SetExecutableFileId Data {.ExecutableFileId = FileInfo.FileId};
|
||||
AppendData(std::span {reinterpret_cast<const std::byte*>(&Data), sizeof(Data)});
|
||||
}
|
||||
|
||||
void CodeMapWriter::AppendData(std::span<const std::byte> Data) {
|
||||
std::shared_lock Lock {Mutex};
|
||||
auto Offset = BufferOffset.fetch_add(Data.size_bytes());
|
||||
if (Offset + Data.size_bytes() > Buffer.size()) {
|
||||
// Acquire exclusive lock and flush the buffer.
|
||||
// Under heavy pressure, multiple threads may observe an exhausted buffer simultaneously.
|
||||
// The thread with the last in-bounds Offset is responsible for flushing the buffer.
|
||||
Lock.unlock();
|
||||
bool IsResponsibleForFlush = false;
|
||||
{
|
||||
std::unique_lock ExclusiveLock {Mutex};
|
||||
IsResponsibleForFlush = (Offset <= Buffer.size());
|
||||
if (IsResponsibleForFlush) {
|
||||
Flush(Offset, ExclusiveLock);
|
||||
}
|
||||
}
|
||||
if (!IsResponsibleForFlush) {
|
||||
// Wait for the buffer to be flushed on the responsible thread
|
||||
Utils::SpinWaitLock::WaitPred<std::less_equal<>, size_t>(reinterpret_cast<size_t*>(&BufferOffset), Buffer.size());
|
||||
}
|
||||
AppendData(Data);
|
||||
return;
|
||||
}
|
||||
|
||||
memcpy(&Buffer.at(Offset), Data.data(), Data.size_bytes());
|
||||
}
|
||||
|
||||
} // namespace FEXCore
|
||||
|
||||
namespace FEXCore::Context {
|
||||
@@ -15,12 +216,156 @@ CodeCache::CodeCache(ContextImpl& CTX_)
|
||||
: CTX(CTX_) {}
|
||||
CodeCache::~CodeCache() = default;
|
||||
|
||||
uint64_t CodeCache::ComputeCodeMapId(std::string_view Filename, int FD) {
|
||||
if (Filename.empty()) {
|
||||
return 0xffff'ffff'ffff'ffff;
|
||||
}
|
||||
|
||||
// For now, we just use the file path as an identifier.
|
||||
// TODO: Ensure the hash is unique enough to distinguish executables while remaining independent of the installation location
|
||||
return XXH3_64bits(Filename.data(), Filename.size());
|
||||
}
|
||||
|
||||
struct CodeCacheHeader {
|
||||
char Magic[4] = {'F', 'X', 'C', 'C'};
|
||||
uint32_t FormatVersion = 1;
|
||||
char FEXVersion[8] = {};
|
||||
uint32_t NumBlocks;
|
||||
uint32_t NumCodePages;
|
||||
uint32_t CodeBufferSize;
|
||||
uint32_t NumRelocations;
|
||||
uint64_t SerializedBaseAddress;
|
||||
// TODO: Consider including information from LookupCache.BlockLinks
|
||||
};
|
||||
|
||||
void CodeCache::LoadData(Core::InternalThreadState& Thread, std::byte* MappedCacheFile, const ExecutableFileSectionInfo& GuestRIPLookup) {
|
||||
// TODO
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
static constexpr auto IsOrderedContainer(const T&) -> std::false_type;
|
||||
template<typename... T>
|
||||
static constexpr auto IsOrderedContainer(const std::map<T...>&) -> std::true_type;
|
||||
template<typename... T>
|
||||
static constexpr auto IsOrderedContainer(const std::set<T...>&) -> std::true_type;
|
||||
|
||||
bool CodeCache::SaveData(Core::InternalThreadState& Thread, int fd, const ExecutableFileSectionInfo& SourceBinary, uint64_t SerializedBaseAddress) {
|
||||
// TODO
|
||||
auto CodeBuffer = CTX.GetLatest();
|
||||
auto& LookupCache = *Thread.LookupCache->Shared;
|
||||
|
||||
auto Relocations = Thread.CPUBackend->TakeRelocations(SourceBinary.FileStartVA);
|
||||
|
||||
// Write file header
|
||||
CodeCacheHeader header;
|
||||
memcpy(&header.FEXVersion[0], GIT_SHORT_HASH, strlen(GIT_SHORT_HASH));
|
||||
header.NumBlocks = LookupCache.BlockList.size();
|
||||
header.NumCodePages = LookupCache.CodePages.size();
|
||||
header.CodeBufferSize = CTX.LatestOffset;
|
||||
header.NumRelocations = Relocations.size();
|
||||
header.SerializedBaseAddress = SerializedBaseAddress;
|
||||
::write(fd, &header, sizeof(header));
|
||||
|
||||
// Dump guest<->host block mappings
|
||||
{
|
||||
// Cache contents must be deterministic, so copy the unordered block list and then sort by key
|
||||
static_assert(!decltype(IsOrderedContainer(LookupCache.BlockList))::value, "Already deterministic; drop temporary container");
|
||||
fextl::vector<std::pair<uint64_t, const GuestToHostMap::BlockEntry*>> BlockList;
|
||||
BlockList.reserve(LookupCache.BlockList.size());
|
||||
for (auto& [Guest, BlockEntry] : LookupCache.BlockList) {
|
||||
static_assert(sizeof(Guest) == 8, "Breaking change in code cache data layout");
|
||||
BlockList.emplace_back(Guest, &BlockEntry);
|
||||
}
|
||||
std::ranges::sort(BlockList);
|
||||
|
||||
for (auto [Guest, Host] : BlockList) {
|
||||
static_assert(sizeof(Host->HostCode) == 8, "Breaking change in code cache data layout");
|
||||
static_assert(sizeof(Host->CodePages[0]) == 8, "Breaking change in code cache data layout");
|
||||
|
||||
Guest -= SourceBinary.FileStartVA;
|
||||
::write(fd, &Guest, sizeof(Guest));
|
||||
uint64_t HostCode = Host->HostCode - reinterpret_cast<uintptr_t>(CodeBuffer->Ptr);
|
||||
::write(fd, &HostCode, sizeof(HostCode));
|
||||
uint64_t NumCodePages = Host->CodePages.size();
|
||||
::write(fd, &NumCodePages, sizeof(NumCodePages));
|
||||
LOGMAN_THROW_A_FMT(std::ranges::is_sorted(Host->CodePages), "Code pages aren't sorted");
|
||||
for (auto CodePage : Host->CodePages) {
|
||||
CodePage -= SourceBinary.FileStartVA;
|
||||
::write(fd, &CodePage, sizeof(CodePage));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Dump relocations
|
||||
static_assert(sizeof(Relocations[0]) == 48, "Breaking change in code cache data layout");
|
||||
::write(fd, Relocations.data(), Relocations.size() * sizeof(Relocations[0]));
|
||||
|
||||
// Pad to next page in file so that the CodeBuffer can be mmap'ed into process on load
|
||||
char Zero[64] {};
|
||||
auto Off = lseek(fd, 0, SEEK_CUR);
|
||||
while (Off != AlignUp(Off, Utils::FEX_PAGE_SIZE)) {
|
||||
auto BytesToWrite = std::min(AlignUp(Off, Utils::FEX_PAGE_SIZE) - Off, sizeof(Zero));
|
||||
::write(fd, Zero, BytesToWrite);
|
||||
Off += BytesToWrite;
|
||||
}
|
||||
|
||||
// Dump the host code (relocated for position-independent serialization)
|
||||
std::vector CodeBufferData(reinterpret_cast<std::byte*>(CodeBuffer->Ptr), reinterpret_cast<std::byte*>(CodeBuffer->Ptr) + CTX.LatestOffset);
|
||||
if (!ApplyCodeRelocations(SerializedBaseAddress, CodeBufferData, Relocations, true)) {
|
||||
LOGMAN_THROW_A_FMT(false, "Failed to apply code relocations");
|
||||
return false;
|
||||
}
|
||||
::write(fd, CodeBufferData.data(), CodeBufferData.size());
|
||||
|
||||
// Dump code pages
|
||||
static_assert(decltype(IsOrderedContainer(LookupCache.CodePages))::value, "Non-deterministic data source");
|
||||
for (auto& [Page, Entrypoints] : LookupCache.CodePages) {
|
||||
static_assert(sizeof(Page) == 8, "Breaking change in code cache data layout");
|
||||
::write(fd, &Page, sizeof(Page));
|
||||
uint64_t NumEntrypoints = Entrypoints.size();
|
||||
::write(fd, &NumEntrypoints, sizeof(NumEntrypoints));
|
||||
::write(fd, Entrypoints.data(), Entrypoints.size() * sizeof(Entrypoints[0]));
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
bool CodeCache::ApplyCodeRelocations(uint64_t GuestEntry, std::span<std::byte> Code,
|
||||
std::span<const FEXCore::CPU::Relocation> EntryRelocations, bool ForStorage) {
|
||||
CPU::Arm64Emitter Emitter(&CTX, Code.data(), Code.size_bytes());
|
||||
for (size_t j = 0; j < EntryRelocations.size(); ++j) {
|
||||
const FEXCore::CPU::Relocation& Reloc = EntryRelocations[j];
|
||||
Emitter.SetCursorOffset(Reloc.Header.Offset);
|
||||
|
||||
switch (Reloc.Header.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
|
||||
// Generate a literal so we can place it
|
||||
uint64_t Pointer = ForStorage ? 0 : GetNamedSymbolLiteral(CTX, Reloc.NamedSymbolLiteral.Symbol);
|
||||
Emitter.dc64(Pointer);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
|
||||
uint64_t Pointer = ForStorage ? 0 : reinterpret_cast<uint64_t>(CTX.ThunkHandler->LookupThunk(Reloc.NamedThunkMove.Symbol));
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
|
||||
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.NamedThunkMove.RegisterIndex), Pointer, true);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_LITERAL: {
|
||||
Emitter.dc64(GuestEntry + Reloc.GuestRIP.GuestRIP);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
|
||||
uint64_t Pointer = Reloc.GuestRIP.GuestRIP + GuestEntry;
|
||||
Emitter.LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.GuestRIP.RegisterIndex), Pointer, true);
|
||||
break;
|
||||
}
|
||||
|
||||
default: ERROR_AND_DIE_FMT("Unknown relocation type {}", ToUnderlying(Reloc.Header.Type));
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
@@ -376,7 +376,9 @@ void ContextImpl::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread)
|
||||
Thread->FrontendDecoder = fextl::make_unique<FEXCore::Frontend::Decoder>(Thread);
|
||||
Thread->PassManager = fextl::make_unique<FEXCore::IR::PassManager>();
|
||||
|
||||
Thread->CurrentFrame->Pointers.Common.L1Pointer = Thread->LookupCache->GetL1Pointer();
|
||||
Thread->CurrentFrame->State.L1Pointer = Thread->LookupCache->GetL1Pointer();
|
||||
Thread->CurrentFrame->State.L1Mask = Thread->LookupCache->GetScaledL1PointerMask();
|
||||
|
||||
Thread->CurrentFrame->Pointers.Common.L2Pointer = Thread->LookupCache->GetPagePointer();
|
||||
|
||||
Dispatcher->InitThreadPointers(Thread);
|
||||
@@ -398,6 +400,7 @@ ContextImpl::CreateThread(uint64_t InitialRIP, uint64_t StackPointer, const FEXC
|
||||
FEXCore::Core::InternalThreadState* Thread = new FEXCore::Core::InternalThreadState {
|
||||
.CTX = this,
|
||||
};
|
||||
FEXCore::Allocator::VirtualName("FEXMem_ThreadState", Thread, sizeof(*Thread));
|
||||
|
||||
Thread->CurrentFrame->State.gregs[X86State::REG_RSP] = StackPointer;
|
||||
Thread->CurrentFrame->State.rip = InitialRIP;
|
||||
@@ -434,6 +437,10 @@ void ContextImpl::UnlockAfterFork(FEXCore::Core::InternalThreadState* LiveThread
|
||||
|
||||
Profiler::PostForkAction(Child);
|
||||
if (Child) {
|
||||
if (CodeMapWriter) {
|
||||
CodeMapWriter->ResetAfterFork();
|
||||
}
|
||||
|
||||
CodeInvalidationMutex.StealAndDropActiveLocks();
|
||||
if (Config.StrictInProcessSplitLocks) {
|
||||
StrictSplitLockMutex = 0;
|
||||
@@ -456,9 +463,14 @@ void ContextImpl::LockBeforeFork(FEXCore::Core::InternalThreadState* Thread) {
|
||||
}
|
||||
#endif
|
||||
|
||||
void ContextImpl::OnCodeBufferAllocated(CPU::CodeBuffer& Buffer) {
|
||||
void ContextImpl::OnCodeBufferAllocated(const fextl::shared_ptr<CPU::CodeBuffer>& Buffer) {
|
||||
if (Config.GlobalJITNaming()) {
|
||||
Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
|
||||
Symbols.RegisterJITSpace(Buffer->Ptr, Buffer->Size);
|
||||
}
|
||||
|
||||
{
|
||||
std::scoped_lock lk {CodeBufferListLock};
|
||||
CodeBufferList.emplace_back(Buffer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -470,7 +482,8 @@ void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, boo
|
||||
Thread->CPUBackend->ClearCache();
|
||||
} else {
|
||||
// Clear L1+L2 cache of this thread, and clear L3 cache across any threads using it
|
||||
Thread->LookupCache->ClearCache();
|
||||
auto lk = Thread->LookupCache->AcquireWriteLock();
|
||||
Thread->LookupCache->ClearCache(lk);
|
||||
}
|
||||
Allocator::VirtualDontNeed(Thread->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
|
||||
}
|
||||
@@ -642,10 +655,10 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
LogMan::Msg::EFmt("Invalid or Unknown instruction: {} 0x{:x}", TableInfo->Name ?: "UND", Block.Entry - GuestRIP);
|
||||
}
|
||||
|
||||
if (Block.BlockStatus == Frontend::Decoder::DecodedBlockStatus::NOEXEC_INST) {
|
||||
Thread->OpDispatcher->NoExecOp(DecodedInfo);
|
||||
} else {
|
||||
if (Block.BlockStatus == Frontend::Decoder::DecodedBlockStatus::INVALID_INST) {
|
||||
Thread->OpDispatcher->InvalidOp(DecodedInfo);
|
||||
} else {
|
||||
Thread->OpDispatcher->NoExecOp(DecodedInfo);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -711,7 +724,8 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
if (SourcecodeResolver && Config.GDBSymbols()) {
|
||||
auto MappedSection = SyscallHandler->LookupExecutableFileSection(*Thread, GuestRIP);
|
||||
if (MappedSection) {
|
||||
MappedSection->FileInfo.SourcecodeMap = SourcecodeResolver->GenerateMap(MappedSection->FileInfo.Filename, MappedSection->FileInfo.FileId);
|
||||
MappedSection->FileInfo.SourcecodeMap =
|
||||
SourcecodeResolver->GenerateMap(MappedSection->FileInfo.Filename, CodeMap::GetBaseFilename(MappedSection->FileInfo, false));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -728,7 +742,7 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
// but this would increase lock contention. Redundant frontend runs aren't
|
||||
// as expensive and are easily reverted.
|
||||
if (MaxInst != 1) {
|
||||
if (auto Block = Thread->LookupCache->FindBlock(GuestRIP)) {
|
||||
if (auto Block = Thread->LookupCache->FindBlock(Thread, GuestRIP)) {
|
||||
Thread->OpDispatcher->DelayedDisownBuffer();
|
||||
return {.CompiledCode = {.BlockBegin = reinterpret_cast<uint8_t*>(Block), .EntryPoints = {{GuestRIP, reinterpret_cast<uint8_t*>(Block)}}},
|
||||
.DebugData = nullptr,
|
||||
@@ -769,10 +783,13 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
|
||||
// Is the code in the cache?
|
||||
// The backends only check L1 and L2, not L3
|
||||
if (auto HostCode = Thread->LookupCache->FindBlock(GuestRIP)) {
|
||||
if (auto HostCode = Thread->LookupCache->FindBlock(Thread, GuestRIP)) {
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
// Accumulate a JIT count now, as even if another thread raced us, it should count as a compile.
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedJITCount, 1);
|
||||
|
||||
auto [CompiledCode, DebugData, StartAddr, Length, NeedsAddGuestCodeRanges] = CompileCode(Thread, GuestRIP, MaxInst);
|
||||
auto CodePtr = CompiledCode.EntryPoints[GuestRIP];
|
||||
if (CodePtr == nullptr) {
|
||||
@@ -826,20 +843,32 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
Thread->CPUBackend->ClearRelocations();
|
||||
}
|
||||
|
||||
fextl::vector<uint64_t> CodePages;
|
||||
|
||||
if (NeedsAddGuestCodeRanges) {
|
||||
// Track in the guest to host map all entrypoints for all pages the compiled block touches, if any page didn't previously
|
||||
// contain code, inform the frontend so it can setup SMC detection.
|
||||
auto BlockInfo = Thread->FrontendDecoder->GetDecodedBlockInfo();
|
||||
CodePages.reserve(BlockInfo->CodePages.size());
|
||||
CodePages.insert(CodePages.end(), BlockInfo->CodePages.begin(), BlockInfo->CodePages.end());
|
||||
for (auto CodePage : BlockInfo->CodePages) {
|
||||
if (Thread->LookupCache->AddBlockExecutableRange(BlockInfo->EntryPoints, CodePage, FEXCore::Utils::FEX_PAGE_SIZE)) {
|
||||
if (Thread->LookupCache->AddBlockExecutableRange(Thread, BlockInfo->EntryPoints, CodePage, FEXCore::Utils::FEX_PAGE_SIZE)) {
|
||||
SyscallHandler->MarkGuestExecutableRange(Thread, CodePage, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Insert to lookup cache
|
||||
|
||||
for (auto [GuestAddr, HostAddr] : CompiledCode.EntryPoints) {
|
||||
Thread->LookupCache->AddBlockMapping(GuestAddr, HostAddr);
|
||||
Thread->LookupCache->AddBlockMapping(Thread, GuestAddr, CodePages, HostAddr);
|
||||
}
|
||||
|
||||
if (CodeMapWriter) {
|
||||
auto Region = SyscallHandler->LookupExecutableFileSection(*Thread, GuestRIP);
|
||||
if (Region && Region->FileStartVA != 0) {
|
||||
CodeMapWriter->AppendBlock(*Region, GuestRIP);
|
||||
}
|
||||
}
|
||||
|
||||
return (uintptr_t)CodePtr;
|
||||
@@ -866,49 +895,37 @@ uintptr_t ContextImpl::CompileSingleStep(FEXCore::Core::CpuStateFrame* Frame, ui
|
||||
return (uintptr_t)CodePtr;
|
||||
}
|
||||
|
||||
static void InvalidateGuestThreadCodeRange(FEXCore::Core::InternalThreadState* Thread, InvalidatedEntryAccumulator& Accumulator,
|
||||
uint64_t Start, uint64_t Length) {
|
||||
void ContextImpl::InvalidateCodeBuffersCodeRange(uint64_t Start, uint64_t Length) {
|
||||
FEXCORE_PROFILE_SCOPED("InvalidateCodeBuffersCodeRange");
|
||||
|
||||
LOGMAN_THROW_A_FMT(CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to be unique_locked here");
|
||||
std::scoped_lock lk {CodeBufferListLock};
|
||||
auto it = CodeBufferList.begin();
|
||||
while (it != CodeBufferList.end()) {
|
||||
if (auto Strong = it->lock()) {
|
||||
Strong->LookupCache->InvalidateRange(Start, Length);
|
||||
it++;
|
||||
} else {
|
||||
it = CodeBufferList.erase(it);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::InvalidateThreadCachedCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) {
|
||||
LOGMAN_THROW_A_FMT(CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to be unique_locked here");
|
||||
|
||||
// Ensures now-modified mappings aren't cached as being in their previous non-executable state.
|
||||
// Accessing FrontendDecoder is safe as the thread's code invalidation mutex must be locked here.
|
||||
Thread->FrontendDecoder->ResetExecutableRangeCache();
|
||||
|
||||
auto lk = Thread->LookupCache->AcquireLock();
|
||||
auto& CodePages = Thread->LookupCache->Shared->CodePages;
|
||||
if (Thread->LookupCache->InvalidateCacheRange(Start, Length)) {
|
||||
FEXCORE_PROFILE_SCOPED("InvalidateCallRet");
|
||||
|
||||
auto lower = CodePages.lower_bound(Start >> 12);
|
||||
auto upper = CodePages.upper_bound((Start + Length - 1) >> 12);
|
||||
|
||||
for (auto it = lower; it != upper; it++) {
|
||||
Accumulator.emplace_back(std::move(it->second));
|
||||
}
|
||||
|
||||
bool InvalidatedAnyEntries = false;
|
||||
for (const auto& PageEntries : Accumulator) {
|
||||
for (const auto& Entry : PageEntries) {
|
||||
if (ContextImpl::ThreadRemoveCodeEntry(Thread, Entry)) {
|
||||
InvalidatedAnyEntries = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (InvalidatedAnyEntries) {
|
||||
// This may cause access violations in the thread on Windows as zeroing is not atomic, this is handled by the frontend
|
||||
Allocator::VirtualDontNeed(Thread->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, InvalidatedEntryAccumulator& Accumulator,
|
||||
uint64_t Start, uint64_t Length) {
|
||||
InvalidateGuestThreadCodeRange(Thread, Accumulator, Start, Length);
|
||||
}
|
||||
|
||||
bool ContextImpl::ThreadRemoveCodeEntry(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP) {
|
||||
LogMan::Throw::AFmt(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to "
|
||||
"be unique_locked here");
|
||||
|
||||
return Thread->LookupCache->Erase(Thread->CurrentFrame, GuestRIP);
|
||||
}
|
||||
|
||||
void ContextImpl::ThreadRemoveCodeEntryFromJit(FEXCore::Core::CpuStateFrame* Frame, uint64_t GuestRIP) {
|
||||
static_cast<ContextImpl*>(Frame->Thread->CTX)->SyscallHandler->InvalidateGuestCodeRange(Frame->Thread, GuestRIP, 1);
|
||||
}
|
||||
@@ -952,10 +969,10 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu
|
||||
|
||||
if (GPRSize == IR::OpSize::i64Bit) {
|
||||
IR::Ref R = emit->_StoreRegister(emit->Constant(Entrypoint), GPRSize);
|
||||
R->Reg = IR::PhysicalRegister(IR::GPRFixedClass, X86State::REG_R11).Raw;
|
||||
R->Reg = IR::PhysicalRegister(IR::RegClass::GPRFixed, X86State::REG_R11).Raw;
|
||||
} else {
|
||||
emit->_StoreContext(GPRSize, IR::FPRClass, emit->_VCastFromGPR(IR::OpSize::i64Bit, IR::OpSize::i64Bit, emit->Constant(Entrypoint)),
|
||||
offsetof(Core::CPUState, mm[0][0]));
|
||||
emit->_StoreContextFPR(GPRSize, emit->_VCastFromGPR(IR::OpSize::i64Bit, IR::OpSize::i64Bit, emit->Constant(Entrypoint)),
|
||||
offsetof(Core::CPUState, mm[0][0]));
|
||||
}
|
||||
emit->_ExitFunction(IR::OpSize::i64Bit, emit->Constant(GuestThunkEntrypoint), IR::BranchHint::None, emit->Invalid(), emit->Invalid());
|
||||
},
|
||||
@@ -976,7 +993,7 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu
|
||||
void ContextImpl::AddForceTSOInformation(const IntervalList<uint64_t>& ValidRanges, fextl::set<uint64_t>&& Instructions) {
|
||||
LogMan::Throw::AFmt(CodeInvalidationMutex.try_lock() == false, "CodeInvalidationMutex needs to be unique_locked here");
|
||||
ForceTSOValidRanges.Insert(ValidRanges);
|
||||
ForceTSOInstructions.merge(Instructions);
|
||||
ForceTSOInstructions.merge(std::move(Instructions));
|
||||
}
|
||||
|
||||
void ContextImpl::RemoveForceTSOInformation(uint64_t Address, uint64_t Size) {
|
||||
|
||||
@@ -5,7 +5,6 @@
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "Utils/MemberFunctionToPointer.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
@@ -36,12 +35,14 @@ static void SleepThread(FEXCore::Context::ContextImpl* CTX, FEXCore::Core::CpuSt
|
||||
CTX->SyscallHandler->SleepThread(CTX, Frame);
|
||||
}
|
||||
|
||||
constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096 * 4;
|
||||
constexpr size_t MAX_DISPATCHER_CODE_SIZE = FEXCore::Utils::FEX_PAGE_SIZE * 4;
|
||||
|
||||
Dispatcher::Dispatcher(FEXCore::Context::ContextImpl* ctx)
|
||||
: Arm64Emitter(ctx, FEXCore::Allocator::VirtualAlloc(MAX_DISPATCHER_CODE_SIZE, true), MAX_DISPATCHER_CODE_SIZE)
|
||||
, CTX {ctx} {
|
||||
EmitDispatcher();
|
||||
|
||||
FEXCore::Allocator::VirtualName("FEXMem_Misc", reinterpret_cast<void*>(GetBufferBase()), MAX_DISPATCHER_CODE_SIZE);
|
||||
}
|
||||
|
||||
Dispatcher::~Dispatcher() {
|
||||
@@ -96,7 +97,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
ARMEmitter::BiDirectionalLabel LoopTop {};
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
b(&LoopTop);
|
||||
(void)b(&LoopTop);
|
||||
|
||||
AbsoluteLoopTopAddressEnterECFillSRA = GetCursorAddress<uint64_t>();
|
||||
ldr(STATE, EC_ENTRY_CPUAREA_REG, CPU_AREA_EMULATOR_DATA_OFFSET);
|
||||
@@ -104,10 +105,10 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
ldr(RipReg, STATE_PTR(CpuStateFrame, State.rip));
|
||||
// Force a single instruction block if ENTRY_FILL_SRA_SINGLE_INST_REG is nonzero entering the JIT, used for inline SMC handling.
|
||||
cbnz(ARMEmitter::Size::i32Bit, ENTRY_FILL_SRA_SINGLE_INST_REG, &CompileSingleStep);
|
||||
(void)cbnz(ARMEmitter::Size::i32Bit, ENTRY_FILL_SRA_SINGLE_INST_REG, &CompileSingleStep);
|
||||
|
||||
// Enter JIT
|
||||
b(&LoopTop);
|
||||
(void)b(&LoopTop);
|
||||
|
||||
AbsoluteLoopTopAddressEnterEC = GetCursorAddress<uint64_t>();
|
||||
// Load ThreadState and write the target PC there
|
||||
@@ -128,7 +129,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP1, TMP2, REG_CALLRET_SP);
|
||||
// EC_CALL_CHECKER_PC_REG is REG_PF which isn't touched by any of the above
|
||||
sub(ARMEmitter::Size::i64Bit, TMP1, EC_CALL_CHECKER_PC_REG, TMP1);
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP1, &LoopTop);
|
||||
(void)cbnz(ARMEmitter::Size::i64Bit, TMP1, &LoopTop);
|
||||
|
||||
// If the entry at the TOS is for the target address, pop it and return to the JIT code
|
||||
add(ARMEmitter::Size::i64Bit, REG_CALLRET_SP, REG_CALLRET_SP, 0x10);
|
||||
@@ -167,67 +168,73 @@ void Dispatcher::EmitDispatcher() {
|
||||
ldr(TMP2, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionEC));
|
||||
br(TMP2);
|
||||
|
||||
(void)!Bind(&l_NotECCode);
|
||||
(void)Bind(&l_NotECCode);
|
||||
#endif
|
||||
|
||||
ldrb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
(void)cbnz(ARMEmitter::Size::i32Bit, TMP1, &CompileSingleStep);
|
||||
|
||||
// This is the block cache lookup routine
|
||||
// It matches what is going on it LookupCache.h::FindBlock
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
|
||||
|
||||
// Mask the address by the virtual address size so we can check for aliases
|
||||
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
|
||||
if (std::popcount(VirtualMemorySize) == 1) {
|
||||
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), VirtualMemorySize - 1);
|
||||
} else {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, VirtualMemorySize);
|
||||
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), TMP4);
|
||||
}
|
||||
|
||||
ARMEmitter::ForwardLabel NoBlock;
|
||||
|
||||
{
|
||||
// Offset the address and add to our page pointer
|
||||
lsr(ARMEmitter::Size::i64Bit, TMP2, TMP4, 12);
|
||||
if (DisableL2Cache()) {
|
||||
(void)b(&NoBlock);
|
||||
} else {
|
||||
// This is the block cache lookup routine
|
||||
// It matches what is going on it LookupCache.h::FindBlock
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
|
||||
|
||||
// Load the pointer from the offset
|
||||
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 3);
|
||||
// Mask the address by the virtual address size so we can check for aliases
|
||||
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
|
||||
if (std::popcount(VirtualMemorySize) == 1) {
|
||||
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), VirtualMemorySize - 1);
|
||||
} else {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, VirtualMemorySize);
|
||||
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), TMP4);
|
||||
}
|
||||
|
||||
// If page pointer is zero then we have no block
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &NoBlock);
|
||||
|
||||
// Steal the page offset
|
||||
and_(ARMEmitter::Size::i64Bit, TMP2, TMP4, 0x0FFF);
|
||||
|
||||
// Shift the offset by the size of the block cache entry
|
||||
add(TMP1, TMP1, TMP2, ARMEmitter::ShiftType::LSL, (int)log2(sizeof(FEXCore::LookupCache::LookupCacheEntry)));
|
||||
|
||||
// The the full LookupCacheEntry with a single LDP.
|
||||
// Check the guest address first to ensure it maps to the address we are currently at.
|
||||
// This fixes aliasing problems
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP4, TMP2, TMP1, 0);
|
||||
|
||||
// If the guest address doesn't match, Compile the block.
|
||||
sub(TMP2, TMP2, RipReg);
|
||||
(void)cbnz(ARMEmitter::Size::i64Bit, TMP2, &NoBlock);
|
||||
|
||||
|
||||
// Check the host address to see if it matches, else compile the block.
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP4, &NoBlock);
|
||||
|
||||
// If we've made it here then we have a real compiled block
|
||||
{
|
||||
// update L1 cache
|
||||
ldr(TMP1, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
// Offset the address and add to our page pointer
|
||||
lsr(ARMEmitter::Size::i64Bit, TMP2, TMP4, 12);
|
||||
|
||||
and_(ARMEmitter::Size::i64Bit, TMP2, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
|
||||
add(TMP1, TMP1, TMP2, ARMEmitter::ShiftType::LSL, 4);
|
||||
stp<ARMEmitter::IndexType::OFFSET>(TMP4, RipReg, TMP1);
|
||||
// Load the pointer from the offset
|
||||
ldr(TMP1, TMP1, TMP2, ARMEmitter::ExtendedType::LSL_64, 3);
|
||||
|
||||
// Jump to the block
|
||||
br(TMP4);
|
||||
// If page pointer is zero then we have no block
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP1, &NoBlock);
|
||||
|
||||
// Steal the page offset
|
||||
and_(ARMEmitter::Size::i64Bit, TMP2, TMP4, 0x0FFF);
|
||||
|
||||
// Shift the offset by the size of the block cache entry
|
||||
add(TMP1, TMP1, TMP2, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(sizeof(LookupCache::LookupCacheEntry)));
|
||||
|
||||
// The the full LookupCacheEntry with a single LDP.
|
||||
// Check the guest address first to ensure it maps to the address we are currently at.
|
||||
// This fixes aliasing problems
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP4, TMP2, TMP1, 0);
|
||||
|
||||
// If the guest address doesn't match, Compile the block.
|
||||
sub(TMP2, TMP2, RipReg);
|
||||
(void)cbnz(ARMEmitter::Size::i64Bit, TMP2, &NoBlock);
|
||||
|
||||
// Check the host address to see if it matches, else compile the block.
|
||||
(void)cbz(ARMEmitter::Size::i64Bit, TMP4, &NoBlock);
|
||||
|
||||
// If we've made it here then we have a real compiled block
|
||||
{
|
||||
// update L1 cache
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP1, TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, State.L1Pointer));
|
||||
|
||||
// Calculate (tmp1 + ((ripreg & L1_ENTRIES_MASK) << 4)) for the address
|
||||
// L1Mask is pre-shifted.
|
||||
and_(ARMEmitter::Size::i64Bit, TMP2, TMP2, RipReg.R(), ARMEmitter::ShiftType::LSL, FEXCore::ilog2(sizeof(LookupCache::LookupCacheEntry)));
|
||||
add(TMP1, TMP1, TMP2);
|
||||
|
||||
stp<ARMEmitter::IndexType::OFFSET>(TMP4, RipReg, TMP1);
|
||||
|
||||
// Jump to the block
|
||||
br(TMP4);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -481,7 +488,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
// Now push the callback return trampoline to the guest stack
|
||||
// Guest will be misaligned because calling a thunk won't correct the guest's stack once we call the callback from the host
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, CTX->X86CodeGen.CallbackReturn);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, CTX->SignalDelegation->GetThunkCallbackRET());
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r2, CTX->Config.Is64BitMode ? 16 : 12);
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/Interpreter/InterpreterOps.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
|
||||
#include <array>
|
||||
@@ -50,6 +51,10 @@ public:
|
||||
}
|
||||
#endif
|
||||
|
||||
uint64_t GetExitFunctionLinkerAddress() const {
|
||||
return ExitFunctionLinkerAddress;
|
||||
}
|
||||
|
||||
SignalDelegatorConfig MakeSignalDelegatorConfig() const;
|
||||
|
||||
protected:
|
||||
@@ -92,6 +97,8 @@ private:
|
||||
|
||||
void EmitDispatcher();
|
||||
uint64_t GenerateABICall(FallbackABI ABI);
|
||||
|
||||
FEX_CONFIG_OPT(DisableL2Cache, DISABLEL2CACHE);
|
||||
};
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -9,7 +9,6 @@ $end_info$
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
|
||||
#include <array>
|
||||
@@ -90,11 +89,6 @@ Decoder::Decoder(FEXCore::Core::InternalThreadState* Thread)
|
||||
}
|
||||
|
||||
bool Decoder::CheckRangeExecutable(uint64_t Address, uint64_t Size) {
|
||||
// Treat FEX-internal X86 callbacks as always executable
|
||||
if (EntryPoint == CTX->X86CodeGen.CallbackReturn) {
|
||||
return true;
|
||||
}
|
||||
|
||||
while (Address < ExecutableRangeBase || Address + Size > ExecutableRangeEnd) {
|
||||
auto RangeInfo = CTX->SyscallHandler->QueryGuestExecutableRange(Thread, Address);
|
||||
ExecutableRangeBase = RangeInfo.Base;
|
||||
@@ -1047,8 +1041,11 @@ Decoder::DecodedBlockStatus Decoder::DecodeInstruction(uint64_t PC) {
|
||||
// Put an invalid instruction in the stream so the core can raise SIGILL if hit
|
||||
// Error while decoding instruction. We don't know the table or instruction size
|
||||
DecodeInst->TableInfo = nullptr;
|
||||
auto Result = ErrorDuringDecoding ? DecodedBlockStatus::INVALID_INST :
|
||||
DecodeInst->InstSize ? DecodedBlockStatus::PARTIAL_DECODE_INST :
|
||||
DecodedBlockStatus::NOEXEC_INST;
|
||||
DecodeInst->InstSize = 0;
|
||||
return ErrorDuringDecoding ? DecodedBlockStatus::INVALID_INST : DecodedBlockStatus::NOEXEC_INST;
|
||||
return Result;
|
||||
} else if (!DecodeInst->TableInfo || (DecodeInst->TableInfo->Type == TYPE_INST && !DecodeInst->TableInfo->OpcodeDispatcher.OpDispatch)) {
|
||||
// If there wasn't an error during decoding but we have no dispatcher for the instruction then claim invalid instruction.
|
||||
return DecodedBlockStatus::INVALID_INST;
|
||||
@@ -1450,7 +1447,10 @@ void Decoder::DecodeInstructionsAtEntry(FEXCore::Core::InternalThreadState* Thre
|
||||
EraseBlock = true;
|
||||
} else {
|
||||
LogMan::Msg::EFmt("{} instruction in entry block: {:X}",
|
||||
BlockIt->BlockStatus == DecodedBlockStatus::INVALID_INST ? "Invalid" : "NoExec", OpAddress);
|
||||
BlockIt->BlockStatus == DecodedBlockStatus::INVALID_INST ? "Invalid" :
|
||||
BlockIt->BlockStatus == DecodedBlockStatus::NOEXEC_INST ? "NoExec" :
|
||||
"PartialDecode",
|
||||
OpAddress);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -27,6 +27,7 @@ public:
|
||||
SUCCESS,
|
||||
INVALID_INST,
|
||||
NOEXEC_INST,
|
||||
PARTIAL_DECODE_INST,
|
||||
};
|
||||
|
||||
// New Frontend decoding
|
||||
|
||||
@@ -50,14 +50,13 @@ DEF_OP(EntrypointOffset) {
|
||||
auto Op = IROp->C<IR::IROp_EntrypointOffset>();
|
||||
|
||||
auto Constant = Entry + Op->Offset;
|
||||
auto Dst = GetReg(Node);
|
||||
uint64_t Mask = ~0ULL;
|
||||
const auto OpSize = IROp->Size;
|
||||
if (OpSize == IR::OpSize::i32Bit) {
|
||||
Mask = 0xFFFF'FFFFULL;
|
||||
}
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Dst, Constant & Mask);
|
||||
InsertGuestRIPMove(GetReg(Node), Constant & Mask);
|
||||
}
|
||||
|
||||
DEF_OP(InlineConstant) {
|
||||
@@ -372,7 +371,7 @@ DEF_OP(CondSubNZCV) {
|
||||
DEF_OP(Neg) {
|
||||
auto Op = IROp->C<IR::IROp_Neg>();
|
||||
|
||||
if (Op->Cond == FEXCore::IR::COND_AL) {
|
||||
if (Op->Cond == IR::CondClass::AL) {
|
||||
neg(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src));
|
||||
} else {
|
||||
cneg(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src), MapCC(Op->Cond));
|
||||
@@ -1190,6 +1189,19 @@ DEF_OP(Rev) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Rbit) {
|
||||
auto Op = IROp->C<IR::IROp_Rbit>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_A_FMT(OpSize == IR::OpSize::i32Bit || OpSize == IR::OpSize::i64Bit, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = ConvertSize48(IROp);
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src);
|
||||
|
||||
rbit(EmitSize, Dst, Src);
|
||||
}
|
||||
|
||||
DEF_OP(Bfi) {
|
||||
auto Op = IROp->C<IR::IROp_Bfi>();
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
@@ -1273,6 +1285,16 @@ DEF_OP(Sbfe) {
|
||||
sbfx(ConvertSize(IROp), Dst, Src, Op->lsb, Op->Width);
|
||||
}
|
||||
|
||||
DEF_OP(MaskGenerateFromBitWidth) {
|
||||
auto Op = IROp->C<IR::IROp_MaskGenerateFromBitWidth>();
|
||||
auto BitWidth = GetReg(Op->BitWidth);
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, -1);
|
||||
cmp(ARMEmitter::Size::i64Bit, BitWidth, 0);
|
||||
lslv(ARMEmitter::Size::i64Bit, TMP2, TMP1, BitWidth);
|
||||
csinv(ARMEmitter::Size::i64Bit, GetReg(Node), TMP1, TMP2, ARMEmitter::Condition::CC_EQ);
|
||||
}
|
||||
|
||||
DEF_OP(Select) {
|
||||
auto Op = IROp->C<IR::IROp_Select>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -11,23 +11,18 @@ $end_info$
|
||||
#include <FEXCore/Core/Thunks.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
uint64_t Arm64JITCore::GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
uint64_t GetNamedSymbolLiteral(FEXCore::Context::ContextImpl& CTX, FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
switch (Op) {
|
||||
case FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER:
|
||||
return ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker;
|
||||
break;
|
||||
default: ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op)); break;
|
||||
return CTX.Dispatcher->GetExitFunctionLinkerAddress();
|
||||
|
||||
default: ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op));
|
||||
}
|
||||
return ~0ULL;
|
||||
}
|
||||
|
||||
void Arm64JITCore::InsertNamedThunkRelocation(ARMEmitter::Register Reg, const IR::SHA256Sum& Sum) {
|
||||
Relocation MoveABI {};
|
||||
MoveABI.NamedThunkMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE;
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t*>();
|
||||
MoveABI.NamedThunkMove.Offset = CurrentCursor - CodeData.BlockBegin;
|
||||
MoveABI.NamedThunkMove.Header = {.Offset = GetCursorOffset(), .Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE};
|
||||
MoveABI.NamedThunkMove.Symbol = Sum;
|
||||
MoveABI.NamedThunkMove.RegisterIndex = Reg.Idx();
|
||||
|
||||
@@ -38,9 +33,9 @@ void Arm64JITCore::InsertNamedThunkRelocation(ARMEmitter::Register Reg, const IR
|
||||
}
|
||||
|
||||
Arm64JITCore::NamedSymbolLiteralPair Arm64JITCore::InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Op);
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(*CTX, Op);
|
||||
|
||||
Arm64JITCore::NamedSymbolLiteralPair Lit {
|
||||
NamedSymbolLiteralPair Lit {
|
||||
.Lit = Pointer,
|
||||
.MoveABI =
|
||||
{
|
||||
@@ -48,92 +43,72 @@ Arm64JITCore::NamedSymbolLiteralPair Arm64JITCore::InsertNamedSymbolLiteral(FEXC
|
||||
{
|
||||
.Header =
|
||||
{
|
||||
.Offset = 0, // Set by PlaceNamedSymbolLiteral
|
||||
.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL,
|
||||
},
|
||||
.Symbol = Op,
|
||||
.Offset = 0,
|
||||
},
|
||||
},
|
||||
};
|
||||
return Lit;
|
||||
}
|
||||
|
||||
void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair& Lit) {
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t*>();
|
||||
Lit.MoveABI.NamedSymbolLiteral.Offset = CurrentCursor - CodeData.BlockBegin;
|
||||
void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair Lit) {
|
||||
switch (Lit.MoveABI.Header.Type) {
|
||||
case RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL:
|
||||
case RelocationTypes::RELOC_GUEST_RIP_LITERAL: {
|
||||
Lit.MoveABI.Header.Offset = GetCursorOffset();
|
||||
break;
|
||||
}
|
||||
|
||||
default: ERROR_AND_DIE_FMT("Unknown relocation type for {}", __FUNCTION__);
|
||||
}
|
||||
|
||||
BindOrRestart(&Lit.Loc);
|
||||
dc64(Lit.Lit);
|
||||
Relocations.emplace_back(Lit.MoveABI);
|
||||
}
|
||||
|
||||
auto Arm64JITCore::InsertGuestRIPLiteral(uint64_t GuestRIP) -> NamedSymbolLiteralPair {
|
||||
return {
|
||||
.Lit = GuestRIP,
|
||||
.MoveABI =
|
||||
{
|
||||
.GuestRIP = {.Header =
|
||||
{
|
||||
.Offset = 0, // Set by PlaceNamedSymbolLiteral
|
||||
.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_LITERAL,
|
||||
},
|
||||
// NOTE: Cache serialization will subtract the guest binary base address later to produce consistency results
|
||||
.GuestRIP = GuestRIP},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void Arm64JITCore::InsertGuestRIPMove(ARMEmitter::Register Reg, uint64_t Constant) {
|
||||
Relocation MoveABI {};
|
||||
MoveABI.GuestRIPMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE;
|
||||
// Offset is the offset from the entrypoint of the block
|
||||
auto CurrentCursor = GetCursorAddress<uint8_t*>();
|
||||
MoveABI.GuestRIPMove.Offset = CurrentCursor - CodeData.BlockBegin;
|
||||
MoveABI.GuestRIPMove.GuestRIP = Constant;
|
||||
MoveABI.GuestRIPMove.RegisterIndex = Reg.Idx();
|
||||
MoveABI.GuestRIP.Header = {.Offset = GetCursorOffset(), .Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE};
|
||||
// NOTE: Cache serialization will subtract the guest binary base address later to produce consistency results
|
||||
MoveABI.GuestRIP.GuestRIP = Constant;
|
||||
MoveABI.GuestRIP.RegisterIndex = Reg.Idx();
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Reg, Constant, false);
|
||||
Relocations.emplace_back(MoveABI);
|
||||
}
|
||||
|
||||
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, std::span<std::byte> Code, std::span<const FEXCore::CPU::Relocation> Relocations) {
|
||||
const auto OrigBase = GetBufferBase();
|
||||
const auto OrigSize = GetBufferSize();
|
||||
const auto OrigOffset = GetCursorOffset();
|
||||
|
||||
SetBuffer(reinterpret_cast<std::uint8_t*>(Code.data()), Code.size_bytes());
|
||||
for (auto& Reloc : Relocations) {
|
||||
switch (Reloc.Header.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
|
||||
uint64_t Pointer = GetNamedSymbolLiteral(Reloc.NamedSymbolLiteral.Symbol);
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor
|
||||
SetCursorOffset(Reloc.NamedSymbolLiteral.Offset);
|
||||
|
||||
// Generate a literal so we can place it
|
||||
dc64(Pointer);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
|
||||
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc.NamedThunkMove.Symbol));
|
||||
if (Pointer == ~0ULL) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
SetCursorOffset(Reloc.NamedThunkMove.Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.NamedThunkMove.RegisterIndex), Pointer, true);
|
||||
break;
|
||||
}
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
|
||||
// XXX: Reenable once the JIT Object Cache is upstream
|
||||
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
|
||||
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
|
||||
if (Pointer == ~0ULL) {
|
||||
SetBuffer(OrigBase, OrigSize);
|
||||
SetCursorOffset(OrigOffset);
|
||||
return false;
|
||||
}
|
||||
|
||||
// Relocation occurs at the cursorEntry + offset relative to that cursor.
|
||||
SetCursorOffset(Reloc.GuestRIPMove.Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Register(Reloc.GuestRIPMove.RegisterIndex), Pointer, true);
|
||||
fextl::vector<FEXCore::CPU::Relocation> Arm64JITCore::TakeRelocations(uint64_t GuestBaseAddress) {
|
||||
// Rebase relocations to library base address
|
||||
for (auto& Relocation : Relocations) {
|
||||
switch (Relocation.Header.Type) {
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE:
|
||||
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_LITERAL: {
|
||||
Relocation.GuestRIP.GuestRIP -= GuestBaseAddress;
|
||||
break;
|
||||
}
|
||||
default:;
|
||||
}
|
||||
}
|
||||
|
||||
SetBuffer(OrigBase, OrigSize);
|
||||
SetCursorOffset(OrigOffset);
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
fextl::vector<FEXCore::CPU::Relocation> Arm64JITCore::TakeRelocations() {
|
||||
return std::move(Relocations);
|
||||
}
|
||||
|
||||
|
||||
@@ -138,26 +138,6 @@ DEF_OP(CAS) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicXor) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicXor>();
|
||||
const auto EmitSize = ConvertSize(IROp);
|
||||
const auto SubEmitSize = ConvertSubRegSize8(IROp->Size);
|
||||
|
||||
auto MemSrc = GetReg(Op->Addr);
|
||||
auto Src = GetReg(Op->Value);
|
||||
|
||||
if (CTX->HostFeatures.SupportsAtomics) {
|
||||
steorl(SubEmitSize, Src, MemSrc);
|
||||
} else {
|
||||
ARMEmitter::BackwardLabel LoopTop;
|
||||
(void)Bind(&LoopTop);
|
||||
ldaxr(SubEmitSize, TMP2, MemSrc);
|
||||
eor(EmitSize, TMP2, TMP2, Src);
|
||||
stlxr(SubEmitSize, TMP2, TMP2, MemSrc);
|
||||
(void)cbnz(EmitSize, TMP2, &LoopTop);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(AtomicSwap) {
|
||||
auto Op = IROp->C<IR::IROp_AtomicSwap>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -174,14 +174,12 @@ DEF_OP(ExitFunction) {
|
||||
}
|
||||
|
||||
// L1 Cache
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer));
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP1, TMP2, STATE, offsetof(FEXCore::Core::CpuStateFrame, State.L1Pointer));
|
||||
|
||||
// Calculate (tmp1 + ((ripreg & L1_ENTRIES_MASK) << 4)) for the address
|
||||
// arithmetic. ubfiz+add is marginally faster on Firestorm than
|
||||
// and+add(shift). Same performance on Cortex.
|
||||
static_assert(LookupCache::L1_ENTRIES_MASK == ((1u << 20) - 1));
|
||||
ubfiz(ARMEmitter::Size::i64Bit, TMP4, RipReg, 4, 20);
|
||||
add(TMP1, TMP1, TMP4);
|
||||
// L1Mask is pre-shifted.
|
||||
and_(ARMEmitter::Size::i64Bit, TMP2, TMP2, RipReg, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(sizeof(LookupCache::LookupCacheEntry)));
|
||||
add(TMP1, TMP1, TMP2);
|
||||
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(TMP2, TMP1, TMP1, 0);
|
||||
|
||||
@@ -235,16 +233,16 @@ DEF_OP(CondJump) {
|
||||
LOGMAN_THROW_A_FMT(IsGPR(Op->Cmp1), "CondJump: Expected GPR");
|
||||
LOGMAN_THROW_A_FMT(isConst, "CondJump: Expected constant source");
|
||||
|
||||
if (Op->Cond.Val == FEXCore::IR::COND_EQ) {
|
||||
if (Op->Cond == IR::CondClass::EQ) {
|
||||
LOGMAN_THROW_A_FMT(Const == 0, "CondJump: Expected 0 source");
|
||||
cbz_OrRestart(Size, Reg, TrueTargetLabel);
|
||||
} else if (Op->Cond.Val == FEXCore::IR::COND_NEQ) {
|
||||
} else if (Op->Cond == IR::CondClass::NEQ) {
|
||||
LOGMAN_THROW_A_FMT(Const == 0, "CondJump: Expected 0 source");
|
||||
cbnz_OrRestart(Size, Reg, TrueTargetLabel);
|
||||
} else if (Op->Cond.Val == FEXCore::IR::COND_TSTZ) {
|
||||
} else if (Op->Cond == IR::CondClass::TSTZ) {
|
||||
LOGMAN_THROW_A_FMT(Const < 64, "CondJump: Expected valid bit source");
|
||||
tbz_OrRestart(Reg, Const, TrueTargetLabel);
|
||||
} else if (Op->Cond.Val == FEXCore::IR::COND_TSTNZ) {
|
||||
} else if (Op->Cond == IR::CondClass::TSTNZ) {
|
||||
LOGMAN_THROW_A_FMT(Const < 64, "CondJump: Expected valid bit source");
|
||||
tbnz_OrRestart(Reg, Const, TrueTargetLabel);
|
||||
} else {
|
||||
@@ -262,16 +260,10 @@ DEF_OP(Syscall) {
|
||||
// X1: ThreadState
|
||||
// X2: Pointer to SyscallArguments
|
||||
|
||||
FEXCore::IR::SyscallFlags Flags = Op->Flags;
|
||||
PushDynamicRegs(TMP1);
|
||||
|
||||
uint32_t GPRSpillMask = ~0U;
|
||||
uint32_t FPRSpillMask = ~0U;
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) == FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY) {
|
||||
// Need to spill all caller saved registers still
|
||||
GPRSpillMask = CALLER_GPR_MASK;
|
||||
FPRSpillMask = CALLER_FPR_MASK;
|
||||
}
|
||||
|
||||
SpillStaticRegs(TMP1, true, GPRSpillMask, FPRSpillMask);
|
||||
|
||||
@@ -305,117 +297,22 @@ DEF_OP(Syscall) {
|
||||
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
FillStaticRegs(true, GPRSpillMask, FPRSpillMask, ARMEmitter::Reg::r1, ARMEmitter::Reg::r2);
|
||||
// Result is now in x0
|
||||
// Fix the stack and any values that were stepped on
|
||||
FillStaticRegs(true, GPRSpillMask, FPRSpillMask, ARMEmitter::Reg::r1, ARMEmitter::Reg::r2);
|
||||
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
PopDynamicRegs();
|
||||
PopDynamicRegs();
|
||||
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NORETURNEDRESULT) != FEXCore::IR::SyscallFlags::NORETURNEDRESULT) {
|
||||
// Move result to its destination register.
|
||||
// Only if `NORETURNEDRESULT` wasn't set, otherwise we might overwrite the CPUState refilled with `FillStaticRegs`
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), ARMEmitter::Reg::r0);
|
||||
}
|
||||
}
|
||||
}
|
||||
const auto OSABI = CTX->SyscallHandler->GetOSABI();
|
||||
|
||||
DEF_OP(InlineSyscall) {
|
||||
auto Op = IROp->C<IR::IROp_InlineSyscall>();
|
||||
// Arguments are passed as follows:
|
||||
// X8: SyscallNumber - RA INTERSECT
|
||||
// X0: Arg0 & Return
|
||||
// X1: Arg1
|
||||
// X2: Arg2
|
||||
// X3: Arg3
|
||||
// X4: Arg4 - RA INTERSECT
|
||||
// X5: Arg5 - RA INTERSECT
|
||||
// X6: Arg6 - Doesn't exist in x86-64 land. RA INTERSECT
|
||||
|
||||
// One argument is removed from the SyscallArguments::MAX_ARGS since the first argument was syscall number
|
||||
const static std::array<ARMEmitter::XRegister, FEXCore::HLE::SyscallArguments::MAX_ARGS - 1> RegArgs = {
|
||||
{ARMEmitter::XReg::x0, ARMEmitter::XReg::x1, ARMEmitter::XReg::x2, ARMEmitter::XReg::x3, ARMEmitter::XReg::x4, ARMEmitter::XReg::x5}};
|
||||
|
||||
bool Intersects {};
|
||||
// We always need to spill x8 since we can't know if it is live at this SSA location
|
||||
uint32_t SpillMask = 1U << 8;
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS - 1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) {
|
||||
break;
|
||||
}
|
||||
|
||||
auto Reg = GetReg(Op->Header.Args[i]);
|
||||
if (Reg == ARMEmitter::Reg::r8 || Reg == ARMEmitter::Reg::r4 || Reg == ARMEmitter::Reg::r5) {
|
||||
|
||||
SpillMask |= (1U << Reg.Idx());
|
||||
Intersects = true;
|
||||
}
|
||||
}
|
||||
|
||||
// Ordering is incredibly important here
|
||||
// We must spill any overlapping registers first THEN claim we are in a syscall without invalidating state at all
|
||||
// Only spill the registers that intersect with our usage
|
||||
SpillStaticRegs(TMP1, false, SpillMask);
|
||||
|
||||
// Now that we are spilled, store in the state that we are in a syscall
|
||||
// Still without overwriting registers that matter
|
||||
// 16bit LoadConstant to be a single instruction
|
||||
// We must always spill at least one register (x8) so this value always has a bit set
|
||||
// This gives the signal handler a value to check to see if we are in a syscall at all
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SpillMask & 0xFFFF);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
// Now that we have claimed to be a syscall we can set up the arguments
|
||||
const auto EmitSize = CTX->Config.Is64BitMode() ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
const auto EmitSubSize = CTX->Config.Is64BitMode() ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i32Bit;
|
||||
if (Intersects) {
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS - 1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) {
|
||||
break;
|
||||
}
|
||||
|
||||
auto Reg = GetReg(Op->Header.Args[i]);
|
||||
if (SpillMask & (1U << Reg.Idx())) {
|
||||
// In the case of intersection with x4, x5, or x8 then these are currently SRA
|
||||
// for registers RAX, RDX, and RSP. Which have just been spilled
|
||||
// Just load back from the context.
|
||||
auto Correlation = GetX86RegRelationToARMReg(Reg);
|
||||
LOGMAN_THROW_A_FMT(Correlation != X86State::REG_INVALID, "Invalid register mapping");
|
||||
ldr(EmitSubSize, RegArgs[i].R(), STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[Correlation]));
|
||||
} else {
|
||||
mov(EmitSize, RegArgs[i].R(), Reg);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
for (uint32_t i = 0; i < FEXCore::HLE::SyscallArguments::MAX_ARGS - 1; ++i) {
|
||||
if (Op->Header.Args[i].IsInvalid()) {
|
||||
break;
|
||||
}
|
||||
|
||||
mov(EmitSize, RegArgs[i].R(), GetReg(Op->Header.Args[i]));
|
||||
}
|
||||
}
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, Op->HostSyscallNumber);
|
||||
svc(0);
|
||||
// On updated signal mask we can receive a signal RIGHT HERE
|
||||
|
||||
if ((Op->Flags & FEXCore::IR::SyscallFlags::NORETURN) != FEXCore::IR::SyscallFlags::NORETURN) {
|
||||
// Now that we are done in the syscall we need to carefully peel back the state
|
||||
// First unspill the registers from before
|
||||
FillStaticRegs(false, SpillMask, ~0U, ARMEmitter::Reg::r8, ARMEmitter::Reg::r1);
|
||||
|
||||
// Now the registers we've spilled are back in their original host registers
|
||||
// We can safely claim we are no longer in a syscall
|
||||
str(ARMEmitter::XReg::zr, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
// Result is now in x0
|
||||
// Move result to its destination register
|
||||
mov(EmitSize, GetReg(Node), ARMEmitter::Reg::r0);
|
||||
if (OSABI != FEXCore::HLE::SyscallOSABI::OS_GENERIC) {
|
||||
// Move result to its destination register.
|
||||
// Only if `NORETURNEDRESULT` wasn't set, otherwise we might overwrite the CPUState refilled with `FillStaticRegs`
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), ARMEmitter::Reg::r0);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -431,8 +328,7 @@ DEF_OP(Thunk) {
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetReg(Op->ArgPtr));
|
||||
|
||||
auto thunkFn = static_cast<Context::ContextImpl*>(ThreadState->CTX)->ThunkHandler->LookupThunk(Op->ThunkNameHash);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, (uintptr_t)thunkFn);
|
||||
InsertNamedThunkRelocation(ARMEmitter::Reg::r2, Op->ThunkNameHash);
|
||||
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
|
||||
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
|
||||
} else {
|
||||
|
||||
@@ -423,11 +423,11 @@ DEF_OP(Vector_FToI) {
|
||||
const auto Mask = PRED_TMP_32B.Merging();
|
||||
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val: frintn(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val: frintm(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val: frintp(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val: frintz(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case FEXCore::IR::Round_Host.Val: frinti(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::Nearest: frintn(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::NegInfinity: frintm(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::PosInfinity: frintp(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::TowardsZero: frintz(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::Host: frinti(SubEmitSize, Dst.Z(), Mask, Vector.Z()); break;
|
||||
}
|
||||
} else {
|
||||
const auto IsScalar = ElementSize == OpSize;
|
||||
@@ -449,21 +449,21 @@ DEF_OP(Vector_FToI) {
|
||||
}
|
||||
|
||||
switch (Op->Round) {
|
||||
case IR::Round_Nearest.Val: ROUNDING_FN(frintn); break;
|
||||
case IR::Round_Negative_Infinity.Val: ROUNDING_FN(frintm); break;
|
||||
case IR::Round_Positive_Infinity.Val: ROUNDING_FN(frintp); break;
|
||||
case IR::Round_Towards_Zero.Val: ROUNDING_FN(frintz); break;
|
||||
case IR::Round_Host.Val: ROUNDING_FN(frinti); break;
|
||||
case IR::RoundMode::Nearest: ROUNDING_FN(frintn); break;
|
||||
case IR::RoundMode::NegInfinity: ROUNDING_FN(frintm); break;
|
||||
case IR::RoundMode::PosInfinity: ROUNDING_FN(frintp); break;
|
||||
case IR::RoundMode::TowardsZero: ROUNDING_FN(frintz); break;
|
||||
case IR::RoundMode::Host: ROUNDING_FN(frinti); break;
|
||||
}
|
||||
|
||||
#undef ROUNDING_FN
|
||||
} else {
|
||||
switch (Op->Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val: frintn(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val: frintm(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val: frintp(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val: frintz(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Host.Val: frinti(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::Nearest: frintn(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::NegInfinity: frintm(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::PosInfinity: frintp(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::TowardsZero: frintz(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::Host: frinti(SubEmitSize, Dst.Q(), Vector.Q()); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -539,11 +539,11 @@ DEF_OP(Vector_F64ToI32) {
|
||||
// Then convert to integers using fcvtzs.
|
||||
auto CVTReg = Dst.Z();
|
||||
switch (Round) {
|
||||
case IR::Round_Nearest.Val: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::Round_Negative_Infinity.Val: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::Round_Positive_Infinity.Val: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::Round_Towards_Zero.Val: CVTReg = Vector.Z(); break;
|
||||
case IR::Round_Host.Val: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::Nearest: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::NegInfinity: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::PosInfinity: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
case IR::RoundMode::TowardsZero: CVTReg = Vector.Z(); break;
|
||||
case IR::RoundMode::Host: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), Mask, Vector.Z()); break;
|
||||
}
|
||||
|
||||
fcvtzs(Dst.Z(), ARMEmitter::SubRegSize::i32Bit, Mask, CVTReg, ARMEmitter::SubRegSize::i64Bit);
|
||||
@@ -567,11 +567,11 @@ DEF_OP(Vector_F64ToI32) {
|
||||
|
||||
///< Round float to integral depending on rounding mode.
|
||||
switch (Round) {
|
||||
case FEXCore::IR::Round_Nearest.Val: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Negative_Infinity.Val: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Positive_Infinity.Val: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Towards_Zero.Val: frintz(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case FEXCore::IR::Round_Host.Val: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::Nearest: frintn(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::NegInfinity: frintm(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::PosInfinity: frintp(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::TowardsZero: frintz(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
case IR::RoundMode::Host: frinti(ARMEmitter::SubRegSize::i64Bit, Dst.Q(), Vector.Q()); break;
|
||||
}
|
||||
|
||||
// Now narrow from f64 to f32.
|
||||
|
||||
@@ -493,7 +493,7 @@ void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) {
|
||||
}
|
||||
}
|
||||
|
||||
static void DirectBlockDelinker(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record, bool Call) {
|
||||
static void DirectBlockDelinker(FEXCore::Context::ExitFunctionLinkData* Record, bool Call) {
|
||||
uintptr_t JumpThunkStartAddress = reinterpret_cast<uintptr_t>(Record) - 0x10;
|
||||
uintptr_t CallerAddress = JumpThunkStartAddress + Record->CallerOffset;
|
||||
auto BranchOffset = JumpThunkStartAddress / 4 - CallerAddress / 4;
|
||||
@@ -511,11 +511,12 @@ static void DirectBlockDelinker(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Co
|
||||
ARMEmitter::Emitter::ClearICache(reinterpret_cast<void*>(CallerAddress), 4);
|
||||
}
|
||||
|
||||
static void IndirectBlockDelinker(FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
|
||||
static void IndirectBlockDelinker(FEXCore::Context::ExitFunctionLinkData* Record) {
|
||||
uintptr_t JumpThunkStartAddress = reinterpret_cast<uintptr_t>(Record) - 0x10;
|
||||
uint32_t BranchInst = 0;
|
||||
ARMEmitter::Emitter BranchEmit(reinterpret_cast<uint8_t*>(&BranchInst), 4);
|
||||
BranchEmit.b(0x8);
|
||||
// Restore branch +2 instructions to jump to the linker block
|
||||
BranchEmit.b(0x2);
|
||||
|
||||
std::atomic_ref<uint32_t>(*reinterpret_cast<uint32_t*>(JumpThunkStartAddress)).store(BranchInst, std::memory_order::relaxed);
|
||||
ARMEmitter::Emitter::ClearICache(reinterpret_cast<void*>(JumpThunkStartAddress), 4);
|
||||
@@ -538,7 +539,7 @@ uint64_t Arm64JITCore::ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEX
|
||||
// Guard the LookupCache lock with the code invalidation mutex, to avoid issues with forking
|
||||
auto lk_inval =
|
||||
GuardSignalDeferringSection<std::shared_lock>(static_cast<Context::ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
HostCode = Thread->LookupCache->FindBlock(GuestRip);
|
||||
HostCode = Thread->LookupCache->FindBlock(Thread, GuestRip);
|
||||
}
|
||||
if (!HostCode) {
|
||||
// Hold a reference to the code buffer, to avoid linking unmapped code if compilation triggers a recreation.
|
||||
@@ -563,7 +564,7 @@ uint64_t Arm64JITCore::ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEX
|
||||
auto lk_inval = GuardSignalDeferringSection<std::shared_lock>(static_cast<Context::ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
// Lock here is necessary to prevent simultaneous linking and delinking
|
||||
auto lk = Thread->LookupCache->AcquireLock();
|
||||
auto lk = Thread->LookupCache->AcquireWriteLock();
|
||||
|
||||
// For non-calls, this would extend into the block's code, however that's fine as an out-of-range adr would never
|
||||
// be generated avoiding any false positives.
|
||||
@@ -576,14 +577,12 @@ uint64_t Arm64JITCore::ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEX
|
||||
|
||||
if (KnownCallMarkerInst == ExpectedKnownCallMarkerInst) {
|
||||
BranchEmit.bl(BranchOffset);
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, Record, [](FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
|
||||
DirectBlockDelinker(Frame, Record, true);
|
||||
});
|
||||
Thread->LookupCache->AddBlockLink(
|
||||
GuestRip, Record, [](FEXCore::Context::ExitFunctionLinkData* Record) { DirectBlockDelinker(Record, true); }, lk);
|
||||
} else {
|
||||
BranchEmit.b(BranchOffset);
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, Record, [](FEXCore::Core::CpuStateFrame* Frame, FEXCore::Context::ExitFunctionLinkData* Record) {
|
||||
DirectBlockDelinker(Frame, Record, false);
|
||||
});
|
||||
Thread->LookupCache->AddBlockLink(
|
||||
GuestRip, Record, [](FEXCore::Context::ExitFunctionLinkData* Record) { DirectBlockDelinker(Record, false); }, lk);
|
||||
}
|
||||
|
||||
std::atomic_ref<uint32_t>(*reinterpret_cast<uint32_t*>(CallerAddress)).store(BranchInst, std::memory_order::relaxed);
|
||||
@@ -602,7 +601,7 @@ uint64_t Arm64JITCore::ExitFunctionLink(FEXCore::Core::CpuStateFrame* Frame, FEX
|
||||
std::atomic_ref<uint32_t>(*reinterpret_cast<uint32_t*>(JumpThunkStartAddress)).store(LdrInst, std::memory_order::relaxed);
|
||||
ARMEmitter::Emitter::ClearICache(reinterpret_cast<void*>(JumpThunkStartAddress), 4);
|
||||
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, Record, IndirectBlockDelinker);
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, Record, IndirectBlockDelinker, lk);
|
||||
}
|
||||
|
||||
return HostCode;
|
||||
@@ -623,10 +622,10 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
|
||||
|
||||
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRClass, GeneralRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRFixedClass, StaticRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRClass, GeneralFPRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRFixedClass, StaticFPRegisters.size());
|
||||
RAPass->AddRegisters(IR::RegClass::GPR, GeneralRegisters.size());
|
||||
RAPass->AddRegisters(IR::RegClass::GPRFixed, StaticRegisters.size());
|
||||
RAPass->AddRegisters(IR::RegClass::FPR, GeneralFPRegisters.size());
|
||||
RAPass->AddRegisters(IR::RegClass::FPRFixed, StaticFPRegisters.size());
|
||||
RAPass->PairRegs = PairRegisters;
|
||||
|
||||
{
|
||||
@@ -678,13 +677,13 @@ void Arm64JITCore::EmitDetectionString() {
|
||||
void Arm64JITCore::ClearCache() {
|
||||
// NOTE: Holding on to the reference here is required to ensure validity of the WriteLock mutex
|
||||
auto PrevCodeBuffer = CurrentCodeBuffer;
|
||||
std::lock_guard lk(PrevCodeBuffer->LookupCache->WriteLock);
|
||||
auto lk = PrevCodeBuffer->LookupCache->AcquireWriteLock();
|
||||
|
||||
auto CodeBuffer = GetEmptyCodeBuffer();
|
||||
SetBuffer(CodeBuffer->Ptr, CodeBuffer->Size);
|
||||
EmitDetectionString();
|
||||
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*PrevCodeBuffer, *CurrentCodeBuffer->LookupCache);
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*PrevCodeBuffer, *CurrentCodeBuffer->LookupCache, lk);
|
||||
}
|
||||
|
||||
Arm64JITCore::~Arm64JITCore() {}
|
||||
@@ -784,7 +783,7 @@ void Arm64JITCore::EmitSuspendInterruptCheck() {
|
||||
|
||||
ldr(TMP2.W(), STATE_PTR(CpuStateFrame, SuspendDoorbell));
|
||||
ARMEmitter::ForwardLabel l_NoSuspend;
|
||||
cbz(ARMEmitter::Size::i32Bit, TMP2, &l_NoSuspend);
|
||||
(void)cbz(ARMEmitter::Size::i32Bit, TMP2, &l_NoSuspend);
|
||||
brk(SuspendMagic);
|
||||
(void)Bind(&l_NoSuspend);
|
||||
#endif
|
||||
@@ -810,22 +809,35 @@ void Arm64JITCore::EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool C
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::rsp, ARMEmitter::XReg::rsp, TMP1, ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
}
|
||||
}
|
||||
|
||||
EmitSuspendInterruptCheck();
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
|
||||
FEXCore::Core::DebugData* DebugData, bool CheckTF) {
|
||||
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
|
||||
|
||||
const auto PrevNumAllocations = Relocations.size();
|
||||
|
||||
this->Entry = Entry;
|
||||
this->DebugData = DebugData;
|
||||
this->IR = IR;
|
||||
RequiresFarARM64Jumps = false;
|
||||
SSANodeMultiplier = 24;
|
||||
|
||||
switch (static_cast<RestartOptions::Control>(FEXCore::LongJump::SetJump(RestartControl.RestartJump))) {
|
||||
// Prepare restart via long jump in case branch encoding fails.
|
||||
// This uses UncheckedLongJump since we don't implement std::longjmp in WoA setups
|
||||
switch (static_cast<RestartOptions::Control>(FEXCore::UncheckedLongJump::SetJump(ThreadState->RestartJump))) {
|
||||
case RestartOptions::Control::Incoming:
|
||||
// Nothing
|
||||
break;
|
||||
case RestartOptions::Control::EnableFarARM64Jumps: RequiresFarARM64Jumps = true; break;
|
||||
default: ERROR_AND_DIE_FMT("Unhandled Arm64 restart condition!");
|
||||
case RestartOptions::Control::NeedsLargerJITSpace:
|
||||
// Get rid of the claimed buffer immediately, we can't fit in it at all.
|
||||
TempAllocator.UnclaimBuffer();
|
||||
SSANodeMultiplier *= 2;
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Arm64 restart condition!");
|
||||
}
|
||||
|
||||
uint32_t SSACount = IR->GetSSACount();
|
||||
@@ -837,12 +849,19 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
CodeData.EntryPoints.clear();
|
||||
|
||||
// Fairly excessive buffer range to make sure we don't overflow
|
||||
uint32_t BufferRange = 0x1000 + SSACount * 24;
|
||||
// One page baseline, plus SSANodeMultipler bytes, plus another page for guard page.
|
||||
const uint32_t DesiredBufferRange = AlignUp(FEXCore::Utils::FEX_PAGE_SIZE * 2 + SSACount * SSANodeMultiplier, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
|
||||
// JIT output is first written to a temporary buffer and later relocated to the CodeBuffer.
|
||||
// This minimizes lock contention of CodeBufferWriteMutex.
|
||||
auto TempCodeBuffer = TempAllocator.ReownOrClaimBuffer(BufferRange);
|
||||
SetBuffer(TempCodeBuffer, BufferRange);
|
||||
auto TempCodeBufferInfo = TempAllocator.ReownOrClaimBufferWithSize(DesiredBufferRange);
|
||||
auto TempCodeBuffer = TempCodeBufferInfo.Ptr;
|
||||
const uint32_t UsableBufferRange = TempCodeBufferInfo.Size - FEXCore::Utils::FEX_PAGE_SIZE;
|
||||
|
||||
SetBuffer(TempCodeBuffer, UsableBufferRange);
|
||||
|
||||
ThreadState->JITGuardPage = reinterpret_cast<uintptr_t>(TempCodeBuffer) + UsableBufferRange;
|
||||
ThreadState->JITGuardOverflowArgument = FEXCore::ToUnderlying(RestartOptions::Control::NeedsLargerJITSpace);
|
||||
|
||||
CodeData.BlockBegin = GetCursorAddress<uint8_t*>();
|
||||
|
||||
@@ -932,8 +951,6 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
|
||||
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
|
||||
switch (IROp->Op) {
|
||||
#define REGISTER_OP_RT(op, x) \
|
||||
case FEXCore::IR::IROps::OP_##op: std::invoke(RT_##x, this, IROp, CodeNode); break
|
||||
#define REGISTER_OP(op, x) \
|
||||
case FEXCore::IR::IROps::OP_##op: Op_##x(IROp, CodeNode); break
|
||||
|
||||
@@ -974,22 +991,28 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
|
||||
// This is a ExitFunctionLinkData struct
|
||||
BindOrRestart(&l_ExitLink);
|
||||
dc64(0); // HostCode
|
||||
dc64(PendingJumpThunk.GuestRIP); // GuestRIP
|
||||
dc64(PendingJumpThunk.CallerAddress - ThunkAddress); // CallerOffset
|
||||
dc64(0); // HostCode
|
||||
PlaceNamedSymbolLiteral(InsertGuestRIPLiteral(PendingJumpThunk.GuestRIP)); // GuestRIP
|
||||
dc64(PendingJumpThunk.CallerAddress - ThunkAddress); // CallerOffset
|
||||
}
|
||||
|
||||
BindOrRestart(&l_ExitLink);
|
||||
dc64(ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker);
|
||||
PlaceNamedSymbolLiteral(InsertNamedSymbolLiteral(RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER));
|
||||
|
||||
// CodeSize not including the header or tail data.
|
||||
const uint64_t CodeOnlySize = GetCursorAddress<uint8_t*>() - CodeBegin;
|
||||
|
||||
// Add the JitCodeTail
|
||||
// Add the JitCodeTail (written later)
|
||||
Align(alignof(JITCodeTail));
|
||||
auto JITBlockTailLocation = GetCursorAddress<uint8_t*>();
|
||||
auto JITBlockTail = GetCursorAddress<JITCodeTail*>();
|
||||
CursorIncrement(sizeof(JITCodeTail));
|
||||
const auto JITBlockTailLocation = GetCursorAddress<uint8_t*>();
|
||||
CodeHeader->OffsetToBlockTail = JITBlockTailLocation - CodeData.BlockBegin;
|
||||
|
||||
JITCodeTail JITBlockTail {
|
||||
.RIP = Entry,
|
||||
.GuestSize = Size,
|
||||
.SpinLockFutex = 0,
|
||||
.SingleInst = SingleInst,
|
||||
};
|
||||
|
||||
// Entries that live after the JITCodeTail.
|
||||
// These entries correlate JIT code regions with guest RIP regions.
|
||||
@@ -1007,23 +1030,13 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
// FEXCore::Utils::vl64 GuestRIPOffset;
|
||||
// };
|
||||
|
||||
auto JITRIPEntriesBegin = GetCursorAddress<uint8_t*>();
|
||||
|
||||
// Put the block's RIP entry in the tail.
|
||||
// This will be used for RIP reconstruction in the future.
|
||||
// TODO: This needs to be a data RIP relocation once code caching works.
|
||||
// Current relocation code doesn't support this feature yet.
|
||||
JITBlockTail->RIP = Entry;
|
||||
JITBlockTail->GuestSize = Size;
|
||||
JITBlockTail->SingleInst = SingleInst;
|
||||
JITBlockTail->SpinLockFutex = 0;
|
||||
|
||||
const auto JITRIPEntriesBegin = JITBlockTailLocation + sizeof(JITBlockTail);
|
||||
auto JITRIPEntriesLocation = JITRIPEntriesBegin;
|
||||
|
||||
{
|
||||
// Store the RIP entries.
|
||||
JITBlockTail->NumberOfRIPEntries = DebugData->GuestOpcodes.size();
|
||||
JITBlockTail->OffsetToRIPEntries = JITRIPEntriesBegin - JITBlockTailLocation;
|
||||
JITBlockTail.NumberOfRIPEntries = DebugData->GuestOpcodes.size();
|
||||
JITBlockTail.OffsetToRIPEntries = JITRIPEntriesBegin - JITBlockTailLocation;
|
||||
uintptr_t CurrentRIPOffset = 0;
|
||||
uint64_t CurrentPCOffset = 0;
|
||||
|
||||
@@ -1039,14 +1052,20 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
}
|
||||
}
|
||||
|
||||
CursorIncrement(JITRIPEntriesLocation - JITRIPEntriesBegin);
|
||||
SetCursorOffset(JITRIPEntriesLocation - CodeData.BlockBegin);
|
||||
Align();
|
||||
|
||||
CodeHeader->OffsetToBlockTail = JITBlockTailLocation - CodeData.BlockBegin;
|
||||
|
||||
CodeData.Size = GetCursorAddress<uint8_t*>() - CodeData.BlockBegin;
|
||||
|
||||
JITBlockTail->Size = CodeData.Size;
|
||||
// Finalize and write block tail data
|
||||
JITBlockTail.Size = CodeData.Size;
|
||||
{
|
||||
auto PrevCur = GetCursorOffset();
|
||||
memcpy(JITBlockTailLocation, &JITBlockTail, sizeof(JITBlockTail));
|
||||
SetCursorOffset(JITBlockTailLocation - CodeData.BlockBegin + offsetof(JITCodeTail, RIP));
|
||||
PlaceNamedSymbolLiteral(InsertGuestRIPLiteral(JITBlockTail.RIP));
|
||||
SetCursorOffset(PrevCur);
|
||||
}
|
||||
|
||||
// Migrate the compile output from temporary storage to the actual CodeBuffer.
|
||||
// This can block progress in other compiling threads, so the duration of the lock should be as small as possible.
|
||||
@@ -1055,7 +1074,6 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
|
||||
// Query size of generated code
|
||||
const auto TempSize = GetCursorOffset();
|
||||
LOGMAN_THROW_A_FMT(TempSize <= BufferRange, "Exceeded bounds of temporary buffer ({:#x} vs {:#x})", TempSize, BufferRange);
|
||||
|
||||
// Bring CodeBuffer up to date
|
||||
{
|
||||
@@ -1063,7 +1081,8 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
"doesn't match up!\n");
|
||||
if (auto Prev = CheckCodeBufferUpdate()) {
|
||||
Allocator::VirtualDontNeed(ThreadState->CallRetStackBase, FEXCore::Core::InternalThreadState::CALLRET_STACK_SIZE);
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*Prev, *CurrentCodeBuffer->LookupCache);
|
||||
auto lk = ThreadState->LookupCache->AcquireWriteLock();
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*Prev, *CurrentCodeBuffer->LookupCache, lk);
|
||||
}
|
||||
|
||||
// NOTE: 16-byte alignment of the new cursor offset must be preserved for block linking records
|
||||
@@ -1086,6 +1105,10 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
}
|
||||
CodeBegin += Delta;
|
||||
|
||||
for (std::size_t Idx = PrevNumAllocations; Idx != Relocations.size(); ++Idx) {
|
||||
Relocations[Idx].Header.Offset += CodeBuffers.LatestOffset;
|
||||
}
|
||||
|
||||
// Copy over CodeBuffer contents
|
||||
memcpy(GetCursorAddress<uint8_t*>(), TempCodeBuffer, TempSize);
|
||||
SetCursorOffset(CodeBuffers.LatestOffset + TempSize);
|
||||
|
||||
@@ -10,13 +10,17 @@ $end_info$
|
||||
#include "Interface/Core/ArchHelpers/Arm64Emitter.h"
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include "Interface/Core/JIT/Relocations.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IntrusiveIRList.h"
|
||||
#include "Interface/IR/RegisterAllocationData.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXCore/Utils/LongJump.h>
|
||||
@@ -26,16 +30,19 @@ $end_info$
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <functional>
|
||||
#include <optional>
|
||||
#include <utility>
|
||||
#include <variant>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::Context {
|
||||
struct ExitFunctionLinkData;
|
||||
}
|
||||
namespace FEXCore::IR {
|
||||
class RegisterAllocationPass;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
class Arm64JITCore final : public CPUBackend, public Arm64Emitter {
|
||||
@@ -54,9 +61,6 @@ public:
|
||||
}
|
||||
|
||||
private:
|
||||
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
|
||||
FEX_CONFIG_OPT(HalfBarrierTSOEnabled, HALFBARRIERTSOENABLED);
|
||||
|
||||
const bool HostSupportsSVE128 {};
|
||||
const bool HostSupportsSVE256 {};
|
||||
const bool HostSupportsAVX256 {};
|
||||
@@ -64,10 +68,10 @@ private:
|
||||
const bool HostSupportsAFP {};
|
||||
|
||||
struct RestartOptions {
|
||||
FEXCore::LongJump::JumpBuf RestartJump;
|
||||
enum class Control : uint64_t {
|
||||
Incoming = 0,
|
||||
EnableFarARM64Jumps = 1,
|
||||
NeedsLargerJITSpace = 2,
|
||||
};
|
||||
};
|
||||
|
||||
@@ -75,6 +79,8 @@ private:
|
||||
// In the rare case when those assumptions are broken, FEX needs to safely restart the JIT.
|
||||
RestartOptions RestartControl {};
|
||||
bool RequiresFarARM64Jumps {};
|
||||
// Default to 6 instructions per SSA node.
|
||||
uint32_t SSANodeMultiplier {24};
|
||||
|
||||
ARMEmitter::BiDirectionalLabel* PendingTargetLabel {};
|
||||
ARMEmitter::BiDirectionalLabel* PendingCallReturnTargetLabel {};
|
||||
@@ -105,11 +111,13 @@ private:
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Register GetReg(IR::PhysicalRegister Reg) const {
|
||||
LOGMAN_THROW_A_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
const auto RegClass = Reg.AsRegClass();
|
||||
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
LOGMAN_THROW_A_FMT(RegClass == IR::RegClass::GPRFixed || RegClass == IR::RegClass::GPR, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (RegClass == IR::RegClass::GPRFixed) {
|
||||
return StaticRegisters[Reg.Reg];
|
||||
} else if (Reg.Class == IR::GPRClass.Val) {
|
||||
} else if (RegClass == IR::RegClass::GPR) {
|
||||
return GeneralRegisters[Reg.Reg];
|
||||
}
|
||||
|
||||
@@ -128,11 +136,13 @@ private:
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::VRegister GetVReg(IR::PhysicalRegister Reg) const {
|
||||
LOGMAN_THROW_A_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
const auto RegClass = Reg.AsRegClass();
|
||||
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
LOGMAN_THROW_A_FMT(RegClass == IR::RegClass::FPRFixed || RegClass == IR::RegClass::FPR, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (RegClass == IR::RegClass::FPRFixed) {
|
||||
return StaticFPRegisters[Reg.Reg];
|
||||
} else if (Reg.Class == IR::FPRClass.Val) {
|
||||
} else if (RegClass == IR::RegClass::FPR) {
|
||||
return GeneralFPRegisters[Reg.Reg];
|
||||
}
|
||||
|
||||
@@ -150,8 +160,8 @@ private:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
FEXCore::IR::RegisterClassType GetRegClass(IR::Ref Node) const {
|
||||
return FEXCore::IR::RegisterClassType {IR::PhysicalRegister(Node).Class};
|
||||
static IR::RegClass GetRegClass(IR::Ref Node) {
|
||||
return IR::PhysicalRegister(Node).AsRegClass();
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
@@ -168,7 +178,7 @@ private:
|
||||
// Converts IR-base shift type to ARMEmitter shift type.
|
||||
// Will be a no-op, only a type conversion since the two definitions match.
|
||||
[[nodiscard]]
|
||||
ARMEmitter::ShiftType ConvertIRShiftType(IR::ShiftType Shift) const {
|
||||
static ARMEmitter::ShiftType ConvertIRShiftType(IR::ShiftType Shift) {
|
||||
return Shift == IR::ShiftType::LSL ? ARMEmitter::ShiftType::LSL :
|
||||
Shift == IR::ShiftType::LSR ? ARMEmitter::ShiftType::LSR :
|
||||
Shift == IR::ShiftType::ASR ? ARMEmitter::ShiftType::ASR :
|
||||
@@ -176,18 +186,23 @@ private:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Size ConvertSize(const IR::IROp_Header* Op) {
|
||||
static ARMEmitter::Size ConvertSize(const IR::IROp_Header* Op) {
|
||||
return Op->Size == IR::OpSize::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Size ConvertSize48(const IR::IROp_Header* Op) {
|
||||
static ARMEmitter::Size ConvertSize48(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_A_FMT(Op->Size == IR::OpSize::i32Bit || Op->Size == IR::OpSize::i64Bit, "Invalid size");
|
||||
return ConvertSize(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize16(IR::OpSize ElementSize) {
|
||||
static ARMEmitter::Size ConvertSize(IR::OpSize Size) {
|
||||
return Size == IR::OpSize::i64Bit ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static ARMEmitter::SubRegSize ConvertSubRegSize16(IR::OpSize ElementSize) {
|
||||
LOGMAN_THROW_A_FMT(ElementSize == IR::OpSize::i8Bit || ElementSize == IR::OpSize::i16Bit || ElementSize == IR::OpSize::i32Bit ||
|
||||
ElementSize == IR::OpSize::i64Bit || ElementSize == IR::OpSize::i128Bit,
|
||||
"Invalid size");
|
||||
@@ -199,105 +214,105 @@ private:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize16(const IR::IROp_Header* Op) {
|
||||
static ARMEmitter::SubRegSize ConvertSubRegSize16(const IR::IROp_Header* Op) {
|
||||
return ConvertSubRegSize16(Op->ElementSize);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize8(IR::OpSize ElementSize) {
|
||||
static ARMEmitter::SubRegSize ConvertSubRegSize8(IR::OpSize ElementSize) {
|
||||
LOGMAN_THROW_A_FMT(ElementSize != IR::OpSize::i128Bit, "Invalid size");
|
||||
return ConvertSubRegSize16(ElementSize);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize8(const IR::IROp_Header* Op) {
|
||||
static ARMEmitter::SubRegSize ConvertSubRegSize8(const IR::IROp_Header* Op) {
|
||||
return ConvertSubRegSize8(Op->ElementSize);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize4(const IR::IROp_Header* Op) {
|
||||
static ARMEmitter::SubRegSize ConvertSubRegSize4(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i64Bit, "Invalid size");
|
||||
return ConvertSubRegSize8(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::SubRegSize ConvertSubRegSize248(const IR::IROp_Header* Op) {
|
||||
static ARMEmitter::SubRegSize ConvertSubRegSize248(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i8Bit, "Invalid size");
|
||||
return ConvertSubRegSize8(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair16(const IR::IROp_Header* Op) {
|
||||
static ARMEmitter::VectorRegSizePair ConvertSubRegSizePair16(const IR::IROp_Header* Op) {
|
||||
return ARMEmitter::ToVectorSizePair(ConvertSubRegSize16(Op));
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair8(const IR::IROp_Header* Op) {
|
||||
static ARMEmitter::VectorRegSizePair ConvertSubRegSizePair8(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i128Bit, "Invalid size");
|
||||
return ConvertSubRegSizePair16(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::VectorRegSizePair ConvertSubRegSizePair248(const IR::IROp_Header* Op) {
|
||||
static ARMEmitter::VectorRegSizePair ConvertSubRegSizePair248(const IR::IROp_Header* Op) {
|
||||
LOGMAN_THROW_A_FMT(Op->ElementSize != IR::OpSize::i8Bit, "Invalid size");
|
||||
return ConvertSubRegSizePair8(Op);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
ARMEmitter::Condition MapCC(IR::CondClassType Cond) {
|
||||
switch (Cond.Val) {
|
||||
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
|
||||
case FEXCore::IR::COND_NEQ: return ARMEmitter::Condition::CC_NE;
|
||||
case FEXCore::IR::COND_SGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_SLT: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_SGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_SLE: return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_UGE: return ARMEmitter::Condition::CC_CS;
|
||||
case FEXCore::IR::COND_ULT: return ARMEmitter::Condition::CC_CC;
|
||||
case FEXCore::IR::COND_UGT: return ARMEmitter::Condition::CC_HI;
|
||||
case FEXCore::IR::COND_ULE: return ARMEmitter::Condition::CC_LS;
|
||||
case FEXCore::IR::COND_FLU: return ARMEmitter::Condition::CC_LT;
|
||||
case FEXCore::IR::COND_FGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_FLEU: return ARMEmitter::Condition::CC_LE;
|
||||
case FEXCore::IR::COND_FGT: return ARMEmitter::Condition::CC_GT;
|
||||
case FEXCore::IR::COND_FU:
|
||||
case FEXCore::IR::COND_VS: return ARMEmitter::Condition::CC_VS;
|
||||
case FEXCore::IR::COND_FNU:
|
||||
case FEXCore::IR::COND_VC: return ARMEmitter::Condition::CC_VC;
|
||||
case FEXCore::IR::COND_MI: return ARMEmitter::Condition::CC_MI;
|
||||
case FEXCore::IR::COND_PL: return ARMEmitter::Condition::CC_PL;
|
||||
static ARMEmitter::Condition MapCC(IR::CondClass Cond) {
|
||||
switch (Cond) {
|
||||
case IR::CondClass::EQ: return ARMEmitter::Condition::CC_EQ;
|
||||
case IR::CondClass::NEQ: return ARMEmitter::Condition::CC_NE;
|
||||
case IR::CondClass::SGE: return ARMEmitter::Condition::CC_GE;
|
||||
case IR::CondClass::SLT: return ARMEmitter::Condition::CC_LT;
|
||||
case IR::CondClass::SGT: return ARMEmitter::Condition::CC_GT;
|
||||
case IR::CondClass::SLE: return ARMEmitter::Condition::CC_LE;
|
||||
case IR::CondClass::UGE: return ARMEmitter::Condition::CC_CS;
|
||||
case IR::CondClass::ULT: return ARMEmitter::Condition::CC_CC;
|
||||
case IR::CondClass::UGT: return ARMEmitter::Condition::CC_HI;
|
||||
case IR::CondClass::ULE: return ARMEmitter::Condition::CC_LS;
|
||||
case IR::CondClass::FLU: return ARMEmitter::Condition::CC_LT;
|
||||
case IR::CondClass::FGE: return ARMEmitter::Condition::CC_GE;
|
||||
case IR::CondClass::FLEU: return ARMEmitter::Condition::CC_LE;
|
||||
case IR::CondClass::FGT: return ARMEmitter::Condition::CC_GT;
|
||||
case IR::CondClass::FU:
|
||||
case IR::CondClass::VS: return ARMEmitter::Condition::CC_VS;
|
||||
case IR::CondClass::FNU:
|
||||
case IR::CondClass::VC: return ARMEmitter::Condition::CC_VC;
|
||||
case IR::CondClass::MI: return ARMEmitter::Condition::CC_MI;
|
||||
case IR::CondClass::PL: return ARMEmitter::Condition::CC_PL;
|
||||
default: LOGMAN_MSG_A_FMT("Unsupported compare type"); return ARMEmitter::Condition::CC_NV;
|
||||
}
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsFPR(IR::RegisterClassType Class) const {
|
||||
return Class == IR::FPRClass || Class == IR::FPRFixedClass;
|
||||
static bool IsFPR(IR::RegClass Class) {
|
||||
return Class == IR::RegClass::FPR || Class == IR::RegClass::FPRFixed;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsGPR(IR::RegisterClassType Class) const {
|
||||
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
|
||||
static bool IsGPR(IR::RegClass Class) {
|
||||
return Class == IR::RegClass::GPR || Class == IR::RegClass::GPRFixed;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsGPR(IR::Ref Node) {
|
||||
static bool IsGPR(IR::Ref Node) {
|
||||
return IsGPR(GetRegClass(Node));
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsFPR(IR::Ref Node) {
|
||||
static bool IsFPR(IR::Ref Node) {
|
||||
return IsFPR(GetRegClass(Node));
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsGPR(IR::OrderedNodeWrapper Wrap) {
|
||||
return IsGPR(IR::RegisterClassType {IR::PhysicalRegister(Wrap).Class});
|
||||
static bool IsGPR(IR::OrderedNodeWrapper Wrap) {
|
||||
return IsGPR(IR::PhysicalRegister(Wrap).AsRegClass());
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsFPR(IR::OrderedNodeWrapper Wrap) {
|
||||
return IsFPR(IR::RegisterClassType {IR::PhysicalRegister(Wrap).Class});
|
||||
static bool IsFPR(IR::OrderedNodeWrapper Wrap) {
|
||||
return IsFPR(IR::PhysicalRegister(Wrap).AsRegClass());
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
@@ -339,9 +354,7 @@ private:
|
||||
}
|
||||
|
||||
// Restart helpers
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void bl_OrRestart(T* Label) {
|
||||
if (bl(Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
@@ -349,12 +362,10 @@ private:
|
||||
|
||||
// We can support this but currently unnecessary.
|
||||
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void b_OrRestart(T* Label) {
|
||||
if (b(Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
@@ -362,12 +373,10 @@ private:
|
||||
|
||||
// We can support this but currently unnecessary.
|
||||
ERROR_AND_DIE_FMT("Tried to branch larger than 128MB away!");
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void b_OrRestart(ARMEmitter::Condition Cond, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
@@ -385,12 +394,10 @@ private:
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void cbz_OrRestart(ARMEmitter::Size s, ARMEmitter::Register rt, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
@@ -408,12 +415,10 @@ private:
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void cbnz_OrRestart(ARMEmitter::Size s, ARMEmitter::Register rt, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
@@ -431,12 +436,10 @@ private:
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void tbz_OrRestart(ARMEmitter::Register rt, uint32_t Bit, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
@@ -454,12 +457,10 @@ private:
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void tbnz_OrRestart(ARMEmitter::Register rt, uint32_t Bit, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
ARMEmitter::ForwardLabel Skip {};
|
||||
@@ -477,38 +478,40 @@ private:
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void adr_OrRestart(ARMEmitter::Register rd, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
if (LongAddressGen(rd, Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
|
||||
ERROR_AND_DIE_FMT("Unable to encode long ADR.");
|
||||
}
|
||||
return;
|
||||
}
|
||||
if (adr(rd, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
// We can support this but currently unnecessary.
|
||||
ERROR_AND_DIE_FMT("Long ADR currently unsupported!");
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void adrp_OrRestart(ARMEmitter::Register rd, T* Label) {
|
||||
if (RequiresFarARM64Jumps) {
|
||||
if (LongAddressGen(rd, Label) == ARMEmitter::BranchEncodeSucceeded::Failure) {
|
||||
ERROR_AND_DIE_FMT("Unable to encode long ADRP.");
|
||||
}
|
||||
return;
|
||||
}
|
||||
if (adrp(rd, Label) == ARMEmitter::BranchEncodeSucceeded::Success) {
|
||||
return;
|
||||
}
|
||||
|
||||
// We can support this but currently unnecessary.
|
||||
ERROR_AND_DIE_FMT("Long ADRP currently unsupported!");
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
requires (std::is_same_v<T, ARMEmitter::ForwardLabel> || std::is_same_v<T, ARMEmitter::BackwardLabel> ||
|
||||
std::is_same_v<T, ARMEmitter::BiDirectionalLabel> || std::is_same_v<T, ARMEmitter::ForwardLabel::Reference>)
|
||||
template<ARMEmitter::IsLabel T>
|
||||
void BindOrRestart(T* Label) {
|
||||
if (Bind(Label)) {
|
||||
return;
|
||||
@@ -516,11 +519,11 @@ private:
|
||||
|
||||
if (RequiresFarARM64Jumps) {
|
||||
// This should have been caught before this point.
|
||||
ERROR_AND_DIE_FMT("Oops. Unhandled long bind.");
|
||||
ERROR_AND_DIE_FMT("Unhandled long bind");
|
||||
return;
|
||||
}
|
||||
|
||||
FEXCore::LongJump::LongJump(RestartControl.RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
FEXCore::UncheckedLongJump::LongJump(ThreadState->RestartJump, FEXCore::ToUnderlying(RestartOptions::Control::EnableFarARM64Jumps));
|
||||
}
|
||||
|
||||
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
|
||||
@@ -533,8 +536,6 @@ private:
|
||||
* @name Relocations
|
||||
* @{ */
|
||||
|
||||
uint64_t GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
|
||||
|
||||
/**
|
||||
* @brief A literal pair relocation object for named symbol literals
|
||||
*/
|
||||
@@ -571,19 +572,30 @@ private:
|
||||
*/
|
||||
NamedSymbolLiteralPair InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
|
||||
|
||||
/**
|
||||
* @brief Inserts a relocation for a constant value relative to the guest entrypoint
|
||||
*
|
||||
* @param Reg - The GPR to move the guest RIP in to
|
||||
* @param Constant - The guest RIP that will be relocated
|
||||
*/
|
||||
NamedSymbolLiteralPair InsertGuestRIPLiteral(uint64_t GuestRIP);
|
||||
|
||||
/**
|
||||
* @brief Place the named symbol literal relocation in memory
|
||||
*
|
||||
* @param Lit - Which literal to place
|
||||
*/
|
||||
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair& Lit);
|
||||
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair Lit);
|
||||
|
||||
fextl::vector<FEXCore::CPU::Relocation> Relocations;
|
||||
|
||||
///< Relocation code loading
|
||||
bool ApplyRelocations(uint64_t GuestEntry, std::span<std::byte> Code, std::span<const FEXCore::CPU::Relocation>);
|
||||
|
||||
fextl::vector<FEXCore::CPU::Relocation> TakeRelocations() override;
|
||||
/**
|
||||
* Returns any relocations generated since the last call to TakeRelocations.
|
||||
*
|
||||
* GuestBaseAddress must match the base virtual address to which the
|
||||
* input x86 binary is mapped.
|
||||
*/
|
||||
fextl::vector<FEXCore::CPU::Relocation> TakeRelocations(uint64_t GuestBaseAddress) override;
|
||||
|
||||
/** @} */
|
||||
|
||||
@@ -606,7 +618,7 @@ private:
|
||||
void Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize, ARMEmitter::VRegister Dst, ARMEmitter::VRegister IncomingDst,
|
||||
std::optional<ARMEmitter::Register> BaseAddr, ARMEmitter::VRegister VectorIndexLow,
|
||||
std::optional<ARMEmitter::VRegister> VectorIndexHigh, ARMEmitter::VRegister MaskReg, IR::OpSize VectorIndexSize,
|
||||
size_t DataElementOffsetStart, size_t IndexElementOffsetStart, uint8_t OffsetScale);
|
||||
size_t DataElementOffsetStart, size_t IndexElementOffsetStart, uint8_t OffsetScale, IR::OpSize AddrSize);
|
||||
|
||||
void EmitTFCheck();
|
||||
|
||||
|
||||
@@ -21,7 +21,7 @@ DEF_OP(LoadContext) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
auto Dst = GetReg(Node);
|
||||
|
||||
switch (OpSize) {
|
||||
@@ -52,7 +52,7 @@ DEF_OP(LoadContext) {
|
||||
DEF_OP(LoadContextPair) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadContextPair>();
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Dst1 = GetReg(Op->OutValue1);
|
||||
const auto Dst2 = GetReg(Op->OutValue2);
|
||||
|
||||
@@ -78,7 +78,7 @@ DEF_OP(StoreContext) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
auto Src = GetZeroableReg(Op->Value);
|
||||
|
||||
switch (OpSize) {
|
||||
@@ -110,7 +110,7 @@ DEF_OP(StoreContextPair) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreContextPair>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
auto Src1 = GetZeroableReg(Op->Value1);
|
||||
auto Src2 = GetZeroableReg(Op->Value2);
|
||||
|
||||
@@ -135,11 +135,11 @@ DEF_OP(StoreContextPair) {
|
||||
DEF_OP(LoadRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
LOGMAN_THROW_A_FMT(Op->Reg < StaticRegisters.size(), "out of range reg");
|
||||
|
||||
mov(GetReg(Node).X(), StaticRegisters[Op->Reg].X());
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
} else if (Op->Class == IR::RegClass::FPR) {
|
||||
const auto regSize = HostSupportsAVX256 ? IR::OpSize::i256Bit : IR::OpSize::i128Bit;
|
||||
LOGMAN_THROW_A_FMT(Op->Reg < StaticFPRegisters.size(), "out of range reg");
|
||||
LOGMAN_THROW_A_FMT(IROp->Size == regSize, "expected sized");
|
||||
@@ -175,12 +175,13 @@ DEF_OP(LoadAF) {
|
||||
|
||||
DEF_OP(StoreRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
auto Reg = IR::PhysicalRegister(Node);
|
||||
const auto Reg = IR::PhysicalRegister(Node);
|
||||
const auto RegClass = Reg.AsRegClass();
|
||||
|
||||
if (Reg.Class == IR::GPRFixedClass) {
|
||||
if (RegClass == IR::RegClass::GPRFixed) {
|
||||
// Always use 64-bit, it's faster. Upper bits ignored for 32-bit mode.
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Reg), GetReg(Op->Value));
|
||||
} else if (Reg.Class == IR::FPRFixedClass) {
|
||||
} else if (RegClass == IR::RegClass::FPRFixed) {
|
||||
const auto regSize = HostSupportsAVX256 ? IR::OpSize::i256Bit : IR::OpSize::i128Bit;
|
||||
LOGMAN_THROW_A_FMT(IROp->Size == regSize, "expected sized");
|
||||
|
||||
@@ -193,7 +194,7 @@ DEF_OP(StoreRegister) {
|
||||
mov(guest.Q(), host.Q());
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", Reg.Class);
|
||||
LOGMAN_THROW_A_FMT(false, "Unhandled Op->Class {}", RegClass);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -225,7 +226,7 @@ DEF_OP(LoadContextIndexed) {
|
||||
|
||||
const auto Index = GetReg(Op->Index);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
switch (Op->Stride) {
|
||||
case 1:
|
||||
case 2:
|
||||
@@ -288,7 +289,7 @@ DEF_OP(StoreContextIndexed) {
|
||||
|
||||
const auto Index = GetReg(Op->Index);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Value = GetReg(Op->Value);
|
||||
|
||||
switch (Op->Stride) {
|
||||
@@ -372,7 +373,7 @@ DEF_OP(SpillRegister) {
|
||||
const auto OpSize = IROp->Size;
|
||||
const uint32_t SlotOffset = Op->Slot * MaxSpillSlotSize;
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Src = GetReg(Op->Value);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: {
|
||||
@@ -413,7 +414,7 @@ DEF_OP(SpillRegister) {
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize); break;
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::FPRClass) {
|
||||
} else if (Op->Class == FEXCore::IR::RegClass::FPR) {
|
||||
const auto Src = GetVReg(Op->Value);
|
||||
|
||||
switch (OpSize) {
|
||||
@@ -452,7 +453,7 @@ DEF_OP(SpillRegister) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled SpillRegister size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unhandled SpillRegister class: {}", Op->Class.Val);
|
||||
LOGMAN_MSG_A_FMT("Unhandled SpillRegister class: {}", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -461,7 +462,7 @@ DEF_OP(FillRegister) {
|
||||
const auto OpSize = IROp->Size;
|
||||
const uint32_t SlotOffset = Op->Slot * MaxSpillSlotSize;
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Dst = GetReg(Node);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: {
|
||||
@@ -502,7 +503,7 @@ DEF_OP(FillRegister) {
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize); break;
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::FPRClass) {
|
||||
} else if (Op->Class == FEXCore::IR::RegClass::FPR) {
|
||||
const auto Dst = GetVReg(Node);
|
||||
|
||||
switch (OpSize) {
|
||||
@@ -541,7 +542,7 @@ DEF_OP(FillRegister) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled FillRegister size: {}", OpSize); break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_MSG_A_FMT("Unhandled FillRegister class: {}", Op->Class.Val);
|
||||
LOGMAN_MSG_A_FMT("Unhandled FillRegister class: {}", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -578,14 +579,14 @@ ARMEmitter::ExtendedMemOperand Arm64JITCore::GenerateMemOperand(
|
||||
return ARMEmitter::ExtendedMemOperand(Base.X(), ARMEmitter::IndexType::OFFSET, Const);
|
||||
} else {
|
||||
auto RegOffset = GetReg(Offset);
|
||||
switch (OffsetType.Val) {
|
||||
case IR::MEM_OFFSET_SXTX.Val:
|
||||
switch (OffsetType) {
|
||||
case IR::MemOffsetType::SXTX:
|
||||
return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::SXTX, FEXCore::ilog2(OffsetScale));
|
||||
case IR::MEM_OFFSET_UXTW.Val:
|
||||
case IR::MemOffsetType::UXTW:
|
||||
return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::UXTW, FEXCore::ilog2(OffsetScale));
|
||||
case IR::MEM_OFFSET_SXTW.Val:
|
||||
case IR::MemOffsetType::SXTW:
|
||||
return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale));
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled GenerateMemOperand OffsetType: {}", OffsetType.Val); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled GenerateMemOperand OffsetType: {}", OffsetType); break;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -612,20 +613,20 @@ ARMEmitter::Register Arm64JITCore::ApplyMemOperand(IR::OpSize AccessSize, ARMEmi
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, Tmp, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(OffsetScale));
|
||||
} else {
|
||||
auto RegOffset = GetReg(Offset);
|
||||
switch (OffsetType.Val) {
|
||||
case IR::MEM_OFFSET_SXTX.Val:
|
||||
switch (OffsetType) {
|
||||
case IR::MemOffsetType::SXTX:
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, RegOffset, ARMEmitter::ExtendedType::SXTX, FEXCore::ilog2(OffsetScale));
|
||||
break;
|
||||
|
||||
case IR::MEM_OFFSET_UXTW.Val:
|
||||
case IR::MemOffsetType::UXTW:
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, RegOffset, ARMEmitter::ExtendedType::UXTW, FEXCore::ilog2(OffsetScale));
|
||||
break;
|
||||
|
||||
case IR::MEM_OFFSET_SXTW.Val:
|
||||
case IR::MemOffsetType::SXTW:
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, RegOffset, ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale));
|
||||
break;
|
||||
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled OffsetType: {}", OffsetType.Val); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled OffsetType: {}", OffsetType); break;
|
||||
}
|
||||
}
|
||||
return Tmp;
|
||||
@@ -676,7 +677,7 @@ ARMEmitter::SVEMemOperand Arm64JITCore::GenerateSVEMemOperand(IR::OpSize AccessS
|
||||
// Note that we do nothing with the offset type and offset scale,
|
||||
// since SVE loads and stores don't have the ability to perform an
|
||||
// optional extension or shift as part of their behavior.
|
||||
LOGMAN_THROW_A_FMT(OffsetType.Val == IR::MEM_OFFSET_SXTX.Val, "Currently only the default offset type (SXTX) is supported.");
|
||||
LOGMAN_THROW_A_FMT(OffsetType == IR::MemOffsetType::SXTX, "Currently only the default offset type (SXTX) is supported.");
|
||||
|
||||
const auto RegOffset = GetReg(Offset);
|
||||
return ARMEmitter::SVEMemOperand(Base.X(), RegOffset.X());
|
||||
@@ -689,7 +690,7 @@ DEF_OP(LoadMem) {
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
switch (OpSize) {
|
||||
@@ -723,7 +724,7 @@ DEF_OP(LoadMemPair) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemPair>();
|
||||
const auto Addr = GetReg(Op->Addr);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Dst1 = GetReg(Op->OutValue1);
|
||||
const auto Dst2 = GetReg(Op->OutValue2);
|
||||
|
||||
@@ -751,13 +752,13 @@ DEF_OP(LoadMemTSO) {
|
||||
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid() || CTX->HostFeatures.SupportsTSOImm9, "unexpected offset");
|
||||
LOGMAN_THROW_A_FMT(Op->OffsetScale == 1, "unexpected offset scale");
|
||||
LOGMAN_THROW_A_FMT(Op->OffsetType == IR::MEM_OFFSET_SXTX, "unexpected offset type");
|
||||
LOGMAN_THROW_A_FMT(Op->OffsetType == IR::MemOffsetType::SXTX, "unexpected offset type");
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == IR::RegClass::GPR) {
|
||||
const auto Dst = GetReg(Node);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
@@ -776,12 +777,10 @@ DEF_OP(LoadMemTSO) {
|
||||
case IR::OpSize::i64Bit: ldapur(Dst.X(), MemReg, Offset); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
} else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
} else if (CTX->HostFeatures.SupportsRCPC && Op->Class == IR::RegClass::GPR) {
|
||||
const auto Dst = GetReg(Node);
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
@@ -793,12 +792,10 @@ DEF_OP(LoadMemTSO) {
|
||||
case IR::OpSize::i64Bit: ldapr(Dst.X(), MemReg); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
} else if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Dst = GetReg(Node);
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
@@ -810,10 +807,8 @@ DEF_OP(LoadMemTSO) {
|
||||
case IR::OpSize::i64Bit: ldar(Dst.X(), MemReg); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
} else {
|
||||
const auto Dst = GetVReg(Node);
|
||||
@@ -1045,7 +1040,7 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
|
||||
ARMEmitter::VRegister IncomingDst, std::optional<ARMEmitter::Register> BaseAddr,
|
||||
ARMEmitter::VRegister VectorIndexLow, std::optional<ARMEmitter::VRegister> VectorIndexHigh,
|
||||
ARMEmitter::VRegister MaskReg, IR::OpSize VectorIndexSize, size_t DataElementOffsetStart,
|
||||
size_t IndexElementOffsetStart, uint8_t OffsetScale) {
|
||||
size_t IndexElementOffsetStart, uint8_t OffsetScale, IR::OpSize AddrSize) {
|
||||
LOGMAN_THROW_A_FMT(ElementSize >= IR::OpSize::i8Bit && ElementSize <= IR::OpSize::i64Bit, "Invalid element size");
|
||||
|
||||
const auto PerformSMove = [this](IR::OpSize ElementSize, const ARMEmitter::Register Dst, const ARMEmitter::VRegister Vector, int index) {
|
||||
@@ -1121,17 +1116,17 @@ void Arm64JITCore::Emulate128BitGather(IR::OpSize Size, IR::OpSize ElementSize,
|
||||
// Calculate memory position for this gather load
|
||||
if (BaseAddr.has_value()) {
|
||||
if (VectorIndexSize == IR::OpSize::i32Bit) {
|
||||
add(ARMEmitter::Size::i64Bit, TempMemReg, *BaseAddr, WorkingReg, ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale));
|
||||
add(ConvertSize(AddrSize), TempMemReg, *BaseAddr, WorkingReg, ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale));
|
||||
} else {
|
||||
add(ARMEmitter::Size::i64Bit, TempMemReg, *BaseAddr, WorkingReg, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(OffsetScale));
|
||||
add(ConvertSize(AddrSize), TempMemReg, *BaseAddr, WorkingReg, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(OffsetScale));
|
||||
}
|
||||
} else {
|
||||
///< In this case we have no base address, All addresses come from the vector register itself
|
||||
if (VectorIndexSize == IR::OpSize::i32Bit) {
|
||||
// Sign extend and shift in to the 64-bit register
|
||||
sbfiz(ARMEmitter::Size::i64Bit, TempMemReg, WorkingReg, FEXCore::ilog2(OffsetScale), 32);
|
||||
sbfiz(ConvertSize(AddrSize), TempMemReg, WorkingReg, FEXCore::ilog2(OffsetScale), 32);
|
||||
} else {
|
||||
lsl(ARMEmitter::Size::i64Bit, TempMemReg, WorkingReg, FEXCore::ilog2(OffsetScale));
|
||||
lsl(ConvertSize(AddrSize), TempMemReg, WorkingReg, FEXCore::ilog2(OffsetScale));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1189,7 +1184,8 @@ DEF_OP(VLoadVectorGatherMasked) {
|
||||
|
||||
///< If the host supports SVE and the offset scale matches SVE limitations then it can do an SVE style load.
|
||||
const bool SupportsSVELoad = (HostSupportsSVE128 || HostSupportsSVE256) &&
|
||||
(OffsetScale == 1 || OffsetScale == IR::OpSizeToSize(VectorIndexSize)) && VectorIndexSize == IROp->ElementSize;
|
||||
(OffsetScale == 1 || OffsetScale == IR::OpSizeToSize(VectorIndexSize)) &&
|
||||
VectorIndexSize == IROp->ElementSize && Op->AddrSize == IR::OpSize::i64Bit;
|
||||
|
||||
if (SupportsSVELoad) {
|
||||
uint8_t SVEScale = FEXCore::ilog2(OffsetScale);
|
||||
@@ -1247,7 +1243,7 @@ DEF_OP(VLoadVectorGatherMasked) {
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit, "Can't emulate this gather load in the backend! Programming error!");
|
||||
Emulate128BitGather(IROp->Size, IROp->ElementSize, Dst, IncomingDst, BaseAddr, VectorIndexLow, VectorIndexHigh, MaskReg,
|
||||
VectorIndexSize, DataElementOffsetStart, IndexElementOffsetStart, OffsetScale);
|
||||
VectorIndexSize, DataElementOffsetStart, IndexElementOffsetStart, OffsetScale, Op->AddrSize);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1272,7 +1268,9 @@ DEF_OP(VLoadVectorGatherMaskedQPS) {
|
||||
!Op->VectorIndexHigh.IsInvalid() ? std::make_optional(GetVReg(Op->VectorIndexHigh)) : std::nullopt;
|
||||
|
||||
///< If the host supports SVE and the offset scale matches SVE limitations then it can do an SVE style load.
|
||||
if (HostSupportsSVE128 && (OffsetScale == 1 || OffsetScale == 4)) {
|
||||
const bool SupportsSVELoad = HostSupportsSVE128 && (OffsetScale == 1 || OffsetScale == 4) && Op->AddrSize == IR::OpSize::i64Bit;
|
||||
|
||||
if (SupportsSVELoad) {
|
||||
ARMEmitter::SVEModType ModType = ARMEmitter::SVEModType::MOD_NONE;
|
||||
if (OffsetScale != 1) {
|
||||
ModType = ARMEmitter::SVEModType::MOD_LSL;
|
||||
@@ -1326,7 +1324,7 @@ DEF_OP(VLoadVectorGatherMaskedQPS) {
|
||||
}
|
||||
} else {
|
||||
Emulate128BitGather(IR::OpSize::i128Bit, IR::OpSize::i32Bit, Dst, IncomingDst, BaseAddr, VectorIndexLow, VectorIndexHigh, MaskReg,
|
||||
IR::OpSize::i64Bit, 0, 0, OffsetScale);
|
||||
IR::OpSize::i64Bit, 0, 0, OffsetScale, Op->AddrSize);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1627,7 +1625,7 @@ DEF_OP(StoreMem) {
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: strb(Src, MemSrc); break;
|
||||
@@ -1738,7 +1736,7 @@ DEF_OP(StoreMemPair) {
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Addr = GetReg(Op->Addr);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Src1 = GetZeroableReg(Op->Value1);
|
||||
const auto Src2 = GetZeroableReg(Op->Value2);
|
||||
switch (OpSize) {
|
||||
@@ -1765,13 +1763,13 @@ DEF_OP(StoreMemTSO) {
|
||||
|
||||
const auto MemReg = GetReg(Op->Addr);
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (Op->Class == IR::RegClass::GPR) {
|
||||
LOGMAN_THROW_A_FMT(Op->Offset.IsInvalid() || CTX->HostFeatures.SupportsTSOImm9, "unexpected offset");
|
||||
LOGMAN_THROW_A_FMT(Op->OffsetScale == 1, "unexpected offset scale");
|
||||
LOGMAN_THROW_A_FMT(Op->OffsetType == IR::MEM_OFFSET_SXTX, "unexpected offset type");
|
||||
LOGMAN_THROW_A_FMT(Op->OffsetType == IR::MemOffsetType::SXTX, "unexpected offset type");
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == IR::RegClass::GPR) {
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
@@ -1783,10 +1781,8 @@ DEF_OP(StoreMemTSO) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
stlurb(Src, MemReg, Offset);
|
||||
} else {
|
||||
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i16Bit: stlurh(Src, MemReg, Offset); break;
|
||||
case IR::OpSize::i32Bit: stlur(Src.W(), MemReg, Offset); break;
|
||||
@@ -1794,17 +1790,15 @@ DEF_OP(StoreMemTSO) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", OpSize); break;
|
||||
}
|
||||
}
|
||||
} else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
} else if (Op->Class == IR::RegClass::GPR) {
|
||||
const auto Src = GetZeroableReg(Op->Value);
|
||||
|
||||
if (OpSize == IR::OpSize::i8Bit) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
stlrb(Src, MemReg);
|
||||
} else {
|
||||
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i16Bit: stlrh(Src, MemReg); break;
|
||||
case IR::OpSize::i32Bit: stlr(Src.W(), MemReg); break;
|
||||
@@ -1898,9 +1892,7 @@ DEF_OP(MemSet) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
stlrb(Value.W(), TMP2);
|
||||
} else {
|
||||
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
|
||||
nop();
|
||||
}
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case 2: stlrh(Value.W(), TMP2); break;
|
||||
case 4: stlr(Value.W(), TMP2); break;
|
||||
@@ -2118,11 +2110,9 @@ DEF_OP(MemCpy) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size); break;
|
||||
}
|
||||
|
||||
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
|
||||
// Placeholders for backpatching barriers (one per load/store)
|
||||
nop();
|
||||
nop();
|
||||
}
|
||||
// Placeholders for backpatching barriers (one per load/store)
|
||||
nop();
|
||||
nop();
|
||||
|
||||
switch (OpSize) {
|
||||
case 2: stlrh(TMP4.W(), TMP2); break;
|
||||
@@ -2144,11 +2134,9 @@ DEF_OP(MemCpy) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled {} size: {}", __func__, Size); break;
|
||||
}
|
||||
|
||||
if (HalfBarrierTSOEnabled() && !ParanoidTSO()) {
|
||||
// Placeholders for backpatching barriers (one per load/store)
|
||||
nop();
|
||||
nop();
|
||||
}
|
||||
// Placeholders for backpatching barriers (one per load/store)
|
||||
nop();
|
||||
nop();
|
||||
|
||||
switch (OpSize) {
|
||||
case 2: stlrh(TMP4.W(), TMP2); break;
|
||||
|
||||
@@ -15,6 +15,7 @@ $end_info$
|
||||
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
@@ -47,10 +48,10 @@ DEF_OP(GuestOpcode) {
|
||||
DEF_OP(Fence) {
|
||||
auto Op = IROp->C<IR::IROp_Fence>();
|
||||
switch (Op->Fence) {
|
||||
case IR::Fence_Load.Val: dmb(ARMEmitter::BarrierScope::LD); break;
|
||||
case IR::Fence_LoadStore.Val: dmb(ARMEmitter::BarrierScope::SY); break;
|
||||
case IR::Fence_Store.Val: dmb(ARMEmitter::BarrierScope::ST); break;
|
||||
case IR::Fence_Inst.Val: isb(); break;
|
||||
case IR::FenceType::Load: dmb(ARMEmitter::BarrierScope::LD); break;
|
||||
case IR::FenceType::LoadStore: dmb(ARMEmitter::BarrierScope::SY); break;
|
||||
case IR::FenceType::Store: dmb(ARMEmitter::BarrierScope::ST); break;
|
||||
case IR::FenceType::Inst: isb(); break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Fence: {}", Op->Fence); break;
|
||||
}
|
||||
}
|
||||
@@ -107,10 +108,10 @@ DEF_OP(GetRoundingMode) {
|
||||
// zero. Just swapping 01 and 10. That's a bitfield reverse. Round mode is in
|
||||
// bottom two bits. After reversing as a 32-bit operation, it'll be in [31:30]
|
||||
// and ripe for reinsertion back at 0.
|
||||
static_assert(IR::ROUND_MODE_NEAREST == 0);
|
||||
static_assert(IR::ROUND_MODE_NEGATIVE_INFINITY == 1);
|
||||
static_assert(IR::ROUND_MODE_POSITIVE_INFINITY == 2);
|
||||
static_assert(IR::ROUND_MODE_TOWARDS_ZERO == 3);
|
||||
static_assert(FEXCore::ToUnderlying(IR::RoundMode::Nearest) == 0);
|
||||
static_assert(FEXCore::ToUnderlying(IR::RoundMode::NegInfinity) == 1);
|
||||
static_assert(FEXCore::ToUnderlying(IR::RoundMode::PosInfinity) == 2);
|
||||
static_assert(FEXCore::ToUnderlying(IR::RoundMode::TowardsZero) == 3);
|
||||
|
||||
rbit(ARMEmitter::Size::i32Bit, TMP1, Dst);
|
||||
bfi(ARMEmitter::Size::i64Bit, Dst, TMP1, 30, 2);
|
||||
|
||||
@@ -18,18 +18,4 @@ DEF_OP(RMWHandle) {
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), GetReg(IROp->Args[0]));
|
||||
}
|
||||
|
||||
DEF_OP(Swap1) {
|
||||
auto Op = IROp->C<IR::IROp_Swap1>();
|
||||
auto A = GetReg(Op->A), B = GetReg(Op->B);
|
||||
LOGMAN_THROW_A_FMT(B == GetReg(Node), "Invariant");
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, TMP1, A);
|
||||
mov(ARMEmitter::Size::i64Bit, A, B);
|
||||
mov(ARMEmitter::Size::i64Bit, B, TMP1);
|
||||
}
|
||||
|
||||
DEF_OP(Swap2) {
|
||||
// Implemented above
|
||||
}
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -1,79 +1,89 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
enum class RelocationTypes : uint8_t {
|
||||
enum class RelocationTypes : uint32_t {
|
||||
// 8 byte literal in memory for symbol
|
||||
// Aligned to struct RelocNamedSymbolLiteral
|
||||
RELOC_NAMED_SYMBOL_LITERAL,
|
||||
|
||||
// Fixed size named thunk move
|
||||
// 4 instruction constant generation on AArch64
|
||||
// 64-bit mov on x86-64
|
||||
// 4 instruction constant generation
|
||||
// Aligned to struct RelocNamedThunkMove
|
||||
RELOC_NAMED_THUNK_MOVE,
|
||||
|
||||
// 8 byte literal (relative to binary base address)
|
||||
RELOC_GUEST_RIP_LITERAL,
|
||||
|
||||
// Fixed size guest RIP move
|
||||
// 4 instruction constant generation on AArch64
|
||||
// 64-bit mov on x86-64
|
||||
// Aligned to struct RelocGuestRIPMove
|
||||
// 4 instruction constant generation
|
||||
// Aligned to struct RelocGuestRIP
|
||||
RELOC_GUEST_RIP_MOVE,
|
||||
};
|
||||
|
||||
struct RelocationTypeHeader final {
|
||||
struct FEX_PACKED RelocationHeader final {
|
||||
// Offset to the relocated host code data
|
||||
uint64_t Offset {};
|
||||
|
||||
RelocationTypes Type;
|
||||
};
|
||||
|
||||
struct RelocNamedSymbolLiteral final {
|
||||
enum class NamedSymbol : uint8_t {
|
||||
enum class NamedSymbol : uint32_t {
|
||||
///< Thread specific relocations
|
||||
// JIT Literal pointers
|
||||
SYMBOL_LITERAL_EXITFUNCTION_LINKER,
|
||||
};
|
||||
|
||||
RelocationTypeHeader Header {};
|
||||
RelocationHeader Header {};
|
||||
|
||||
NamedSymbol Symbol;
|
||||
|
||||
// Offset in to the code section to begin the relocation
|
||||
uint64_t Offset {};
|
||||
uint32_t Pad[8];
|
||||
};
|
||||
|
||||
struct RelocNamedThunkMove final {
|
||||
RelocationTypeHeader Header {};
|
||||
RelocationHeader Header {};
|
||||
|
||||
// GPR index the constant is being moved to
|
||||
uint8_t RegisterIndex;
|
||||
uint32_t RegisterIndex;
|
||||
|
||||
// The thunk SHA256 hash
|
||||
IR::SHA256Sum Symbol;
|
||||
|
||||
// Offset in to the code section to begin the relocation
|
||||
uint64_t Offset {};
|
||||
};
|
||||
|
||||
struct RelocGuestRIPMove final {
|
||||
RelocationTypeHeader Header {};
|
||||
struct RelocGuestRIP final {
|
||||
RelocationHeader Header {};
|
||||
|
||||
// GPR index the constant is being moved to
|
||||
// GPR index the constant is being moved to (for non-literal relocations)
|
||||
uint8_t RegisterIndex;
|
||||
|
||||
// Offset in to the code section to begin the relocation
|
||||
uint64_t Offset {};
|
||||
char Pad[3];
|
||||
|
||||
// The unrelocated RIP that is being moved
|
||||
// The base RIP (to be moved by the register for non-literal relocations).
|
||||
// In a serialized code cache, this is relative to the binary base address.
|
||||
uint64_t GuestRIP;
|
||||
|
||||
uint32_t pad2[6] {};
|
||||
};
|
||||
|
||||
union Relocation {
|
||||
RelocationTypeHeader Header {};
|
||||
RelocationHeader Header {};
|
||||
|
||||
RelocNamedSymbolLiteral NamedSymbolLiteral;
|
||||
// This makes our union of relocations at least 48 bytes
|
||||
// It might be more efficient to not use a union
|
||||
RelocNamedThunkMove NamedThunkMove;
|
||||
|
||||
RelocGuestRIPMove GuestRIPMove;
|
||||
RelocGuestRIP GuestRIP;
|
||||
};
|
||||
|
||||
uint64_t GetNamedSymbolLiteral(FEXCore::Context::ContextImpl&, RelocNamedSymbolLiteral::NamedSymbol);
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -41,6 +41,7 @@ namespace FEXCore::CPU {
|
||||
const auto Op = IROp->C<IR::IROp_##FEXOp>(); \
|
||||
const auto OpSize = IROp->Size; \
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit; \
|
||||
const auto Is128Bit = OpSize == IR::OpSize::i128Bit; \
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__); \
|
||||
\
|
||||
const auto Dst = GetVReg(Node); \
|
||||
@@ -49,8 +50,10 @@ namespace FEXCore::CPU {
|
||||
\
|
||||
if (HostSupportsSVE256 && Is256Bit) { \
|
||||
ARMOp(Dst.Z(), Vector1.Z(), Vector2.Z()); \
|
||||
} else { \
|
||||
} else if (Is128Bit) { \
|
||||
ARMOp(Dst.Q(), Vector1.Q(), Vector2.Q()); \
|
||||
} else { \
|
||||
ARMOp(Dst.D(), Vector1.D(), Vector2.D()); \
|
||||
} \
|
||||
}
|
||||
|
||||
@@ -744,11 +747,11 @@ DEF_OP(VFToIScalarInsert) {
|
||||
auto Src = *std::get_if<ARMEmitter::VRegister>(&SrcVar);
|
||||
|
||||
switch (RoundMode) {
|
||||
case IR::Round_Nearest: frintn(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::Round_Negative_Infinity: frintm(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::Round_Positive_Infinity: frintp(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::Round_Towards_Zero: frintz(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::Round_Host: frinti(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::RoundMode::Nearest: frintn(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::RoundMode::NegInfinity: frintm(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::RoundMode::PosInfinity: frintp(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::RoundMode::TowardsZero: frintz(SubRegSize.Scalar, Dst, Src); break;
|
||||
case IR::RoundMode::Host: frinti(SubRegSize.Scalar, Dst, Src); break;
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
@@ -15,7 +15,7 @@ $end_info$
|
||||
|
||||
namespace FEXCore {
|
||||
GuestToHostMap::GuestToHostMap()
|
||||
: BlockLinks_mbr {fextl::pmr::get_default_resource()} {
|
||||
: BlockLinks_mbr {"FEXMem_BlockLinks"} {
|
||||
BlockLinks_pma = fextl::make_unique<std::pmr::polymorphic_allocator<std::byte>>(&BlockLinks_mbr);
|
||||
// Setup our PMR map.
|
||||
BlockLinks = BlockLinks_pma->new_object<BlockLinksMapType>();
|
||||
@@ -24,7 +24,7 @@ GuestToHostMap::GuestToHostMap()
|
||||
LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
|
||||
: ctx {CTX} {
|
||||
|
||||
TotalCacheSize = ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE + L1_SIZE;
|
||||
TotalCacheSize = ctx->Config.VirtualMemSize / FEXCore::Utils::FEX_PAGE_SIZE * 8 + CODE_SIZE + MAX_L1_SIZE;
|
||||
|
||||
// Block cache ends up looking like this
|
||||
// PageMemoryMap[VirtualMemoryRegion >> 12]
|
||||
@@ -39,6 +39,10 @@ LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
|
||||
// We need one pointer per page of virtual memory
|
||||
// At 64GB of virtual memory this will allocate 128MB of virtual memory space
|
||||
PagePointer = reinterpret_cast<uintptr_t>(FEXCore::Allocator::VirtualAlloc(TotalCacheSize, false, false));
|
||||
LOGMAN_THROW_A_FMT(PagePointer != -1ULL, "Failed to allocate PagePointer");
|
||||
|
||||
FEXCore::Allocator::VirtualName("FEXMem_Lookup", reinterpret_cast<void*>(PagePointer),
|
||||
ctx->Config.VirtualMemSize / FEXCore::Utils::FEX_PAGE_SIZE * 8 + CODE_SIZE);
|
||||
CTX->SyscallHandler->MarkOvercommitRange(PagePointer, TotalCacheSize);
|
||||
|
||||
// Allocate our memory backing our pages
|
||||
@@ -46,14 +50,21 @@ LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
|
||||
// XXX: We can drop down to 16KB if we store 4byte offsets from the code base
|
||||
// We currently limit to 128MB of real memory for caching for the total cache size.
|
||||
// Can end up being inefficient if we compile a small number of blocks per page
|
||||
PageMemory = PagePointer + ctx->Config.VirtualMemSize / 4096 * 8;
|
||||
LOGMAN_THROW_A_FMT(PageMemory != -1ULL, "Failed to allocate page memory");
|
||||
PageMemory = PagePointer + ctx->Config.VirtualMemSize / FEXCore::Utils::FEX_PAGE_SIZE * 8;
|
||||
|
||||
// L1 Cache
|
||||
L1Pointer = PageMemory + CODE_SIZE;
|
||||
LOGMAN_THROW_A_FMT(L1Pointer != -1ULL, "Failed to allocate L1Pointer");
|
||||
FEXCore::Allocator::VirtualName("FEXMem_Lookup_L1", reinterpret_cast<void*>(L1Pointer), MAX_L1_SIZE);
|
||||
|
||||
VirtualMemSize = ctx->Config.VirtualMemSize;
|
||||
|
||||
if (DynamicL1Cache()) {
|
||||
// Start at minimum size when dynamic.
|
||||
L1PointerMask = MIN_L1_ENTRIES - 1;
|
||||
} else {
|
||||
// Start at maximum instead.
|
||||
L1PointerMask = MAX_L1_ENTRIES - 1;
|
||||
}
|
||||
}
|
||||
|
||||
LookupCache::~LookupCache() {
|
||||
@@ -64,31 +75,27 @@ LookupCache::~LookupCache() {
|
||||
// These will get freed when their memory allocators are deallocated.
|
||||
}
|
||||
|
||||
void LookupCache::ClearL2Cache() {
|
||||
auto lk = Shared->AcquireLock();
|
||||
void LookupCache::ClearL2Cache(const FEXCore::LookupCacheBaseLockToken& lk) {
|
||||
// Clear out the page memory
|
||||
// PagePointer and PageMemory are sequential with each other. Clear both at once.
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE, false);
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer),
|
||||
ctx->Config.VirtualMemSize / FEXCore::Utils::FEX_PAGE_SIZE * 8 + CODE_SIZE, false);
|
||||
AllocateOffset = 0;
|
||||
}
|
||||
|
||||
void LookupCache::ClearThreadLocalCaches() {
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
void LookupCache::ClearThreadLocalCaches(const LookupCacheWriteLockToken&) {
|
||||
// Clear L1 and L2 by clearing the full cache.
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), TotalCacheSize, false);
|
||||
CachedCodePages.clear();
|
||||
}
|
||||
|
||||
void LookupCache::ClearCache() {
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
void LookupCache::ClearCache(const LookupCacheWriteLockToken& lk) {
|
||||
// Clear L1 and L2 by clearing the full cache.
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), TotalCacheSize, false);
|
||||
|
||||
ClearThreadLocalCaches(lk);
|
||||
Shared->ClearCache(lk);
|
||||
}
|
||||
|
||||
void GuestToHostMap::ClearCache(const LockToken&) {
|
||||
void GuestToHostMap::ClearCache(const LookupCacheWriteLockToken&) {
|
||||
// Allocate a new pointer from the BlockLinks pma again.
|
||||
BlockLinks = BlockLinks_pma->new_object<BlockLinksMapType>();
|
||||
// All code is gone, clear the block list
|
||||
|
||||
@@ -2,30 +2,57 @@
|
||||
#pragma once
|
||||
#include "Interface/Context/Context.h"
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/SHMStats.h>
|
||||
#include "Utils/WritePriorityMutex.h"
|
||||
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/memory_resource.h>
|
||||
#include <FEXCore/fextl/robin_map.h>
|
||||
#include <FEXCore/fextl/robin_set.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
#include <FEXCore/fextl/memory_resource.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <functional>
|
||||
#include <stddef.h>
|
||||
#include <utility>
|
||||
#include <mutex>
|
||||
|
||||
namespace FEXCore {
|
||||
struct LookupCacheBaseLockToken {
|
||||
protected:
|
||||
// Protected constructor - only derived classes can construct
|
||||
LookupCacheBaseLockToken() = default;
|
||||
};
|
||||
|
||||
struct LookupCacheWriteLockToken : public LookupCacheBaseLockToken {
|
||||
private:
|
||||
// Only constructible by GuestToHostMap
|
||||
friend struct GuestToHostMap;
|
||||
LookupCacheWriteLockToken(FEXCore::Utils::WritePriorityMutex::Mutex& Mutex)
|
||||
: Lock {Mutex} {}
|
||||
std::lock_guard<FEXCore::Utils::WritePriorityMutex::Mutex> Lock;
|
||||
};
|
||||
|
||||
struct LookupCacheReadLockToken : public LookupCacheBaseLockToken {
|
||||
private:
|
||||
// Only constructible by GuestToHostMap
|
||||
friend struct GuestToHostMap;
|
||||
LookupCacheReadLockToken(FEXCore::Utils::WritePriorityMutex::Mutex& Mutex)
|
||||
: Lock {Mutex} {}
|
||||
std::shared_lock<FEXCore::Utils::WritePriorityMutex::Mutex> Lock;
|
||||
};
|
||||
|
||||
struct GuestToHostMap {
|
||||
std::recursive_mutex WriteLock;
|
||||
|
||||
struct LockToken {
|
||||
std::lock_guard<std::recursive_mutex> Lock;
|
||||
};
|
||||
FEXCore::Utils::WritePriorityMutex::Mutex Lock {};
|
||||
|
||||
[[nodiscard]]
|
||||
LockToken AcquireLock() {
|
||||
return LockToken {std::lock_guard {WriteLock}};
|
||||
LookupCacheWriteLockToken AcquireWriteLock() {
|
||||
return LookupCacheWriteLockToken {Lock};
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
LookupCacheReadLockToken AcquireReadLock() {
|
||||
return LookupCacheReadLockToken {Lock};
|
||||
}
|
||||
|
||||
struct BlockLinkTag {
|
||||
@@ -49,53 +76,72 @@ struct GuestToHostMap {
|
||||
// walking each block member and destructing objects.
|
||||
//
|
||||
// This makes `BlockLinks` look like a raw pointer that could memory leak, but since it is backed by the MBR, it won't.
|
||||
std::pmr::monotonic_buffer_resource BlockLinks_mbr;
|
||||
fextl::pmr::named_monotonic_page_buffer_resource BlockLinks_mbr;
|
||||
using BlockLinksMapType = std::pmr::map<BlockLinkTag, FEXCore::Context::BlockDelinkerFunc>;
|
||||
fextl::unique_ptr<std::pmr::polymorphic_allocator<std::byte>> BlockLinks_pma;
|
||||
BlockLinksMapType* BlockLinks;
|
||||
|
||||
fextl::robin_map<uint64_t, uint64_t> BlockList;
|
||||
struct BlockEntry {
|
||||
uint64_t HostCode;
|
||||
fextl::vector<uint64_t> CodePages;
|
||||
};
|
||||
|
||||
fextl::robin_map<uint64_t, BlockEntry> BlockList;
|
||||
|
||||
fextl::map<uint64_t, fextl::vector<uint64_t>> CodePages;
|
||||
|
||||
GuestToHostMap();
|
||||
|
||||
// Adds to Guest -> Host code mapping
|
||||
void AddBlockMapping(uint64_t Address, void* HostCode, const LockToken&) {
|
||||
const BlockEntry& AddBlockMapping(uint64_t Address, const fextl::vector<uint64_t>& CodePages, void* HostCode, const LookupCacheWriteLockToken&) {
|
||||
// This may replace an existing mapping
|
||||
// NOTE: Generally no previous entry should exist, however there is one exception:
|
||||
// If the backend updates the active thread's CodeBuffer, the new associated LookupCache
|
||||
// may already contain the block address. Since is comparatively rare, we'll just leak
|
||||
// one of the two blocks in this case.
|
||||
BlockList[Address] = (uintptr_t)HostCode;
|
||||
return BlockList.insert_or_assign(Address, BlockEntry {(uintptr_t)HostCode, CodePages}).first->second;
|
||||
}
|
||||
|
||||
std::optional<uintptr_t> FindBlock(uint64_t Address, const LockToken&) {
|
||||
const BlockEntry* FindBlock(uint64_t Address, const LookupCacheReadLockToken&) {
|
||||
auto HostCode = BlockList.find(Address);
|
||||
if (HostCode == BlockList.end()) {
|
||||
return std::nullopt;
|
||||
return nullptr;
|
||||
}
|
||||
return HostCode->second;
|
||||
return &HostCode->second;
|
||||
}
|
||||
|
||||
bool Erase(FEXCore::Core::CpuStateFrame* Frame, uint64_t Address, const LockToken&) {
|
||||
bool Erase(uint64_t Address, const LookupCacheWriteLockToken&) {
|
||||
// Sever any links to this block
|
||||
auto lower = BlockLinks->lower_bound({Address, nullptr});
|
||||
auto upper = BlockLinks->upper_bound({Address, reinterpret_cast<FEXCore::Context::ExitFunctionLinkData*>(UINTPTR_MAX)});
|
||||
for (auto it = lower; it != upper; it = BlockLinks->erase(it)) {
|
||||
it->second(Frame, it->first.HostLink);
|
||||
it->second(it->first.HostLink);
|
||||
}
|
||||
|
||||
// Remove from BlockList
|
||||
return BlockList.erase(Address) != 0;
|
||||
}
|
||||
|
||||
void InvalidateRange(uint64_t Start, uint64_t Length) {
|
||||
auto lk = AcquireWriteLock();
|
||||
|
||||
auto lower = CodePages.lower_bound(Start >> 12);
|
||||
auto upper = CodePages.upper_bound((Start + Length - 1) >> 12);
|
||||
|
||||
for (auto it = lower; it != upper; it++) {
|
||||
for (const auto& Entry : it->second) {
|
||||
Erase(Entry, lk);
|
||||
}
|
||||
}
|
||||
CodePages.erase(lower, upper);
|
||||
}
|
||||
|
||||
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink,
|
||||
const FEXCore::Context::BlockDelinkerFunc& delinker, const LockToken&) {
|
||||
const FEXCore::Context::BlockDelinkerFunc& delinker, const LookupCacheWriteLockToken&) {
|
||||
BlockLinks->insert({{GuestDestination, HostLink}, delinker});
|
||||
}
|
||||
|
||||
bool AddBlockExecutableRange(const fextl::set<uint64_t>& Addresses, uint64_t Start, uint64_t Length, const LockToken&) {
|
||||
bool AddBlockExecutableRange(const fextl::set<uint64_t>& Addresses, uint64_t Start, uint64_t Length, const LookupCacheWriteLockToken&) {
|
||||
bool rv = false;
|
||||
|
||||
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length - 1) >> 12; CurrentPage <= EndPage; CurrentPage++) {
|
||||
@@ -107,7 +153,7 @@ struct GuestToHostMap {
|
||||
return rv;
|
||||
}
|
||||
|
||||
void ClearCache(const LockToken&);
|
||||
void ClearCache(const LookupCacheWriteLockToken&);
|
||||
};
|
||||
|
||||
class LookupCache {
|
||||
@@ -122,122 +168,199 @@ public:
|
||||
|
||||
// Swaps out the underlying GuestToHostMap and clears all associated caches.
|
||||
// This interface requires the previous CodeBuffer to be provided despite not using it. This ensures the shared write lock is still valid.
|
||||
void ChangeGuestToHostMapping([[maybe_unused]] CPU::CodeBuffer& Prev, GuestToHostMap& NewMap) {
|
||||
ClearThreadLocalCaches();
|
||||
void ChangeGuestToHostMapping([[maybe_unused]] CPU::CodeBuffer& Prev, GuestToHostMap& NewMap, const LookupCacheWriteLockToken& lk) {
|
||||
ClearThreadLocalCaches(lk);
|
||||
Shared = &NewMap;
|
||||
}
|
||||
|
||||
uintptr_t FindBlock(uint64_t Address) {
|
||||
uintptr_t FindBlock(FEXCore::Core::InternalThreadState* Thread, uint64_t Address) {
|
||||
// Try L1, no lock needed
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1PointerMask];
|
||||
if (L1Entry.GuestCode == Address) {
|
||||
return L1Entry.HostCode;
|
||||
}
|
||||
|
||||
// L2 and L3 need to be locked
|
||||
auto lk = Shared->AcquireLock();
|
||||
uintptr_t HostPtr {};
|
||||
{
|
||||
std::optional<FEXCore::SHMStats::AccumulationBlock<uint64_t>> LockTime(
|
||||
Thread->ThreadStats ? &Thread->ThreadStats->AccumulatedCacheReadLockTime : nullptr);
|
||||
auto lk = Shared->AcquireReadLock();
|
||||
LockTime.reset();
|
||||
|
||||
// Try L2
|
||||
const auto PageIndex = (Address & (VirtualMemSize - 1)) >> 12;
|
||||
const auto PageOffset = Address & (0x0FFF);
|
||||
if (!DisableL2Cache()) {
|
||||
// Try L2
|
||||
const auto PageIndex = (Address & (VirtualMemSize - 1)) >> 12;
|
||||
const auto PageOffset = Address & (0x0FFF);
|
||||
|
||||
const auto Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
auto LocalPagePointer = Pointers[PageIndex];
|
||||
const auto Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
auto LocalPagePointer = Pointers[PageIndex];
|
||||
|
||||
// Do we a page pointer for this address?
|
||||
if (LocalPagePointer) {
|
||||
// Find there pointer for the address in the blocks
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
// Do we a page pointer for this address?
|
||||
if (LocalPagePointer) {
|
||||
// Find there pointer for the address in the blocks
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
|
||||
if (BlockPointers[PageOffset].GuestCode == Address) {
|
||||
L1Entry.GuestCode = Address;
|
||||
L1Entry.HostCode = BlockPointers[PageOffset].HostCode;
|
||||
return L1Entry.HostCode;
|
||||
if (BlockPointers[PageOffset].GuestCode == Address) {
|
||||
L1Entry.GuestCode = Address;
|
||||
L1Entry.HostCode = BlockPointers[PageOffset].HostCode;
|
||||
HostPtr = L1Entry.HostCode;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (!HostPtr) {
|
||||
// Try L3
|
||||
auto Entry = Shared->FindBlock(Address, lk);
|
||||
if (Entry) {
|
||||
CacheBlockMapping(Address, *Entry, false, lk);
|
||||
HostPtr = Entry->HostCode;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Try L3
|
||||
auto HostCode = Shared->FindBlock(Address, lk);
|
||||
if (HostCode) {
|
||||
CacheBlockMapping(Address, HostCode.value());
|
||||
return HostCode.value();
|
||||
if (HostPtr && DynamicL1Cache()) {
|
||||
UpdateDynamicL1Stats(Thread);
|
||||
}
|
||||
|
||||
// Failed to find
|
||||
return 0;
|
||||
FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedCacheMissCount, 1);
|
||||
|
||||
return HostPtr;
|
||||
}
|
||||
|
||||
void UpdateDynamicL1Stats(FEXCore::Core::InternalThreadState* Thread) {
|
||||
// If host pointer was found in L2 or L3, then add it to the counter.
|
||||
// Keeping track not L1 misses, but specifically L2/L3 hits.
|
||||
++L2L3CacheHits;
|
||||
|
||||
const auto CurrentTime = std::chrono::system_clock::now();
|
||||
const auto Period = CurrentTime - LastPeriod;
|
||||
if (Period >= SamplePeriod) {
|
||||
// If larger than the sample period then check if we need to increase L1 cache size.
|
||||
const double AveragePerSecond = static_cast<double>(L2L3CacheHits) /
|
||||
static_cast<double>(std::chrono::duration_cast<std::chrono::milliseconds>(Period).count()) * 1000.0;
|
||||
|
||||
if (AveragePerSecond >= DynamicL1CacheIncreaseCountHeuristic()) {
|
||||
if (CurrentL1Entries < MAX_L1_ENTRIES) {
|
||||
CurrentL1Entries <<= 1;
|
||||
L1PointerMask = CurrentL1Entries - 1;
|
||||
|
||||
// Update the thread's L1 pointer mask to increase how much cache it uses.
|
||||
// Since we're in C-code, this is safe to update here.
|
||||
Thread->CurrentFrame->State.L1Mask = GetScaledL1PointerMask();
|
||||
}
|
||||
} else if (AveragePerSecond < DynamicL1CacheDecreaseCountHeuristic()) {
|
||||
if (CurrentL1Entries > MIN_L1_ENTRIES) {
|
||||
CurrentL1Entries >>= 1;
|
||||
L1PointerMask = CurrentL1Entries - 1;
|
||||
|
||||
// Madvise the entries that we are dropping. Gives the memory back to the OS.
|
||||
LookupCacheEntry* FirstZeroL1Entry = &reinterpret_cast<LookupCacheEntry*>(L1Pointer)[CurrentL1Entries];
|
||||
size_t ZeroMemorySize = (MAX_L1_ENTRIES - CurrentL1Entries) * sizeof(LookupCacheEntry);
|
||||
FEXCore::Allocator::VirtualDontNeed(FirstZeroL1Entry, ZeroMemorySize, false);
|
||||
|
||||
// Update the thread's L1 pointer mask to increase how much cache it uses.
|
||||
// Since we're in C-code, this is safe to update here.
|
||||
Thread->CurrentFrame->State.L1Mask = GetScaledL1PointerMask();
|
||||
}
|
||||
}
|
||||
|
||||
// Update Last period to start again.
|
||||
LastPeriod = CurrentTime;
|
||||
L2L3CacheHits = 0;
|
||||
}
|
||||
}
|
||||
|
||||
GuestToHostMap* Shared = nullptr;
|
||||
|
||||
// Appends a list of Block {Address} to CodePages [Start, Start + Length)
|
||||
// Returns true if new pages are marked as containing code
|
||||
bool AddBlockExecutableRange(const fextl::set<uint64_t>& Addresses, uint64_t Start, uint64_t Length) {
|
||||
auto lk = Shared->AcquireLock();
|
||||
bool AddBlockExecutableRange(FEXCore::Core::InternalThreadState* Thread, const fextl::set<uint64_t>& Addresses, uint64_t Start, uint64_t Length) {
|
||||
std::optional<FEXCore::SHMStats::AccumulationBlock<uint64_t>> LockTime(
|
||||
Thread->ThreadStats ? &Thread->ThreadStats->AccumulatedCacheWriteLockTime : nullptr);
|
||||
auto lk = Shared->AcquireWriteLock();
|
||||
LockTime.reset();
|
||||
|
||||
return Shared->AddBlockExecutableRange(Addresses, Start, Length, lk);
|
||||
}
|
||||
|
||||
// Adds to Guest -> Host code mapping
|
||||
void AddBlockMapping(uint64_t Address, void* HostCode) {
|
||||
auto lk = Shared->AcquireLock();
|
||||
void AddBlockMapping(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, const fextl::vector<uint64_t>& CodePages, void* HostCode) {
|
||||
std::optional<FEXCore::SHMStats::AccumulationBlock<uint64_t>> LockTime(
|
||||
Thread->ThreadStats ? &Thread->ThreadStats->AccumulatedCacheWriteLockTime : nullptr);
|
||||
auto lk = Shared->AcquireWriteLock();
|
||||
LockTime.reset();
|
||||
|
||||
Shared->AddBlockMapping(Address, HostCode, lk);
|
||||
const auto& Entry = Shared->AddBlockMapping(Address, CodePages, HostCode, lk);
|
||||
|
||||
// There is no need to update L1 or L2, they will get updated on first lookup
|
||||
// However, adding to L1 here increases performance
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
L1Entry.GuestCode = Address;
|
||||
L1Entry.HostCode = (uintptr_t)HostCode;
|
||||
CacheBlockMapping(Address, Entry, true, lk);
|
||||
}
|
||||
|
||||
// NOTE: It's the caller's responsibility to call Erase() for all other
|
||||
// GuestToHostMaps that share the same LookupCache. Otherwise, the
|
||||
// L1/L2 caches will contain stale references to deallocated memory.
|
||||
bool Erase(FEXCore::Core::CpuStateFrame* Frame, uint64_t Address) {
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
bool ErasedAny = Shared->Erase(Frame, Address, lk);
|
||||
|
||||
// Invalidates L1/L2 for a given guest block
|
||||
void InvalidateCache(uint64_t Address, const LookupCacheWriteLockToken& lk) {
|
||||
// Do L1
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1PointerMask];
|
||||
if (L1Entry.GuestCode == Address) {
|
||||
L1Entry.GuestCode = 0;
|
||||
ErasedAny = true;
|
||||
// Leave L1Entry.HostCode as is, so that concurrent lookups won't read a null pointer
|
||||
// This is a soft guarantee for cross thread invalidation, as atomics are not used
|
||||
// and it hasn't been thoroughly tested
|
||||
}
|
||||
|
||||
// Do full map
|
||||
Address = Address & (VirtualMemSize - 1);
|
||||
uint64_t PageOffset = Address & (0x0FFF);
|
||||
Address >>= 12;
|
||||
if (!DisableL2Cache()) {
|
||||
// Do full map
|
||||
Address = Address & (VirtualMemSize - 1);
|
||||
uint64_t PageOffset = Address & (0x0FFF);
|
||||
Address >>= 12;
|
||||
|
||||
uintptr_t* Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
uint64_t LocalPagePointer = Pointers[Address];
|
||||
if (!LocalPagePointer) {
|
||||
// Page for this code didn't even exist, nothing to do
|
||||
return ErasedAny;
|
||||
uintptr_t* Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
uint64_t LocalPagePointer = Pointers[Address];
|
||||
if (!LocalPagePointer) {
|
||||
// Page for this code didn't even exist, nothing to do
|
||||
return;
|
||||
}
|
||||
|
||||
// Page exists, just set the offset to zero
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
BlockPointers[PageOffset].GuestCode = 0;
|
||||
BlockPointers[PageOffset].HostCode = 0;
|
||||
}
|
||||
|
||||
// Page exists, just set the offset to zero
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
BlockPointers[PageOffset].GuestCode = 0;
|
||||
BlockPointers[PageOffset].HostCode = 0;
|
||||
return true;
|
||||
}
|
||||
|
||||
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink, const FEXCore::Context::BlockDelinkerFunc& delinker) {
|
||||
auto lk = Shared->AcquireLock();
|
||||
// Invalidates all L1/L2 entries for all guest block that intersect the given range
|
||||
bool InvalidateCacheRange(uint64_t Start, uint64_t Length) {
|
||||
auto lk = Shared->AcquireWriteLock();
|
||||
|
||||
auto lower = CachedCodePages.lower_bound(Start >> 12);
|
||||
auto upper = CachedCodePages.upper_bound((Start + Length - 1) >> 12);
|
||||
|
||||
for (auto it = lower; it != upper; it++) {
|
||||
for (const auto& Entry : it->second) {
|
||||
InvalidateCache(Entry, lk);
|
||||
}
|
||||
}
|
||||
bool ret = upper != lower;
|
||||
CachedCodePages.erase(lower, upper);
|
||||
return ret;
|
||||
}
|
||||
|
||||
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink,
|
||||
const FEXCore::Context::BlockDelinkerFunc& delinker, const LookupCacheWriteLockToken& lk) {
|
||||
Shared->AddBlockLink(GuestDestination, HostLink, delinker, lk);
|
||||
}
|
||||
|
||||
void ClearCache();
|
||||
void ClearL2Cache();
|
||||
void ClearThreadLocalCaches();
|
||||
void ClearCache(const LookupCacheWriteLockToken&);
|
||||
void ClearL2Cache(const LookupCacheBaseLockToken&);
|
||||
void ClearThreadLocalCaches(const LookupCacheWriteLockToken&);
|
||||
|
||||
uintptr_t GetL1Pointer() const {
|
||||
return L1Pointer;
|
||||
}
|
||||
uintptr_t GetScaledL1PointerMask() const {
|
||||
return L1PointerMask << FEXCore::ilog2(sizeof(LookupCache::LookupCacheEntry));
|
||||
}
|
||||
uintptr_t GetPagePointer() const {
|
||||
return PagePointer;
|
||||
}
|
||||
@@ -245,9 +368,6 @@ public:
|
||||
return VirtualMemSize;
|
||||
}
|
||||
|
||||
constexpr static size_t L1_ENTRIES = 1 * 1024 * 1024; // Must be a power of 2
|
||||
constexpr static size_t L1_ENTRIES_MASK = L1_ENTRIES - 1;
|
||||
|
||||
// This needs to be taken before reads or writes to L2, L3, CodePages,
|
||||
// and before writes to L1. Concurrent access from a thread that this LookupCache doesn't belong to
|
||||
// may only happen during cross thread invalidation (::Erase).
|
||||
@@ -255,45 +375,52 @@ public:
|
||||
// Some care is taken so that L1 lookups can be done without locks, and even tearing is unlikely to lead to a crash.
|
||||
// This approach has not been fully vetted yet.
|
||||
// Also note that L1 lookups might be inlined in the JIT Dispatcher and/or block ends.
|
||||
auto AcquireLock() {
|
||||
return Shared->AcquireLock();
|
||||
auto AcquireWriteLock() {
|
||||
return Shared->AcquireWriteLock();
|
||||
}
|
||||
|
||||
private:
|
||||
void CacheBlockMapping(uint64_t Address, uintptr_t HostCode) {
|
||||
// Do L1
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
L1Entry.GuestCode = Address;
|
||||
L1Entry.HostCode = HostCode;
|
||||
|
||||
// Do ful map
|
||||
auto FullAddress = Address;
|
||||
Address = Address & (VirtualMemSize - 1);
|
||||
|
||||
uint64_t PageOffset = Address & (0x0FFF);
|
||||
Address >>= 12;
|
||||
uintptr_t* Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
uint64_t LocalPagePointer = Pointers[Address];
|
||||
if (!LocalPagePointer) {
|
||||
// We don't have a page pointer for this address
|
||||
// Allocate one now if we can
|
||||
uintptr_t NewPageBacking = AllocateBackingForPage();
|
||||
if (!NewPageBacking) {
|
||||
// Couldn't allocate, clear L2 and retry
|
||||
ClearL2Cache();
|
||||
CacheBlockMapping(Address, HostCode);
|
||||
return;
|
||||
}
|
||||
Pointers[Address] = NewPageBacking;
|
||||
LocalPagePointer = NewPageBacking;
|
||||
void CacheBlockMapping(uint64_t Address, const GuestToHostMap::BlockEntry& Entry, bool L1Only, const LookupCacheBaseLockToken& lk) {
|
||||
for (const auto& CodePage : Entry.CodePages) {
|
||||
CachedCodePages[CodePage >> 12].insert(Address);
|
||||
}
|
||||
|
||||
// Add the new pointer to the page block
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
// Do L1
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1PointerMask];
|
||||
L1Entry.GuestCode = Address;
|
||||
L1Entry.HostCode = Entry.HostCode;
|
||||
|
||||
// This silently replaces existing mappings
|
||||
BlockPointers[PageOffset].GuestCode = FullAddress;
|
||||
BlockPointers[PageOffset].HostCode = HostCode;
|
||||
if (!DisableL2Cache() && !L1Only) {
|
||||
// Do ful map
|
||||
auto FullAddress = Address;
|
||||
Address = Address & (VirtualMemSize - 1);
|
||||
|
||||
uint64_t PageOffset = Address & (0x0FFF);
|
||||
Address >>= 12;
|
||||
|
||||
uintptr_t* Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
||||
uint64_t LocalPagePointer = Pointers[Address];
|
||||
if (!LocalPagePointer) {
|
||||
// We don't have a page pointer for this address
|
||||
// Allocate one now if we can
|
||||
uintptr_t NewPageBacking = AllocateBackingForPage();
|
||||
if (!NewPageBacking) {
|
||||
// Couldn't allocate, clear L2 and retry
|
||||
ClearL2Cache(lk);
|
||||
CacheBlockMapping(Address, Entry, false, lk);
|
||||
return;
|
||||
}
|
||||
Pointers[Address] = NewPageBacking;
|
||||
LocalPagePointer = NewPageBacking;
|
||||
}
|
||||
|
||||
// Add the new pointer to the page block
|
||||
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
||||
|
||||
// This silently replaces existing mappings
|
||||
BlockPointers[PageOffset].GuestCode = FullAddress;
|
||||
BlockPointers[PageOffset].HostCode = Entry.HostCode;
|
||||
}
|
||||
}
|
||||
|
||||
uintptr_t AllocateBackingForPage() {
|
||||
@@ -310,19 +437,38 @@ private:
|
||||
return PageMemory + NewBase;
|
||||
}
|
||||
|
||||
// Maps from a page index to all blocks in the page that have at some point been fetched into L1/L2
|
||||
fextl::map<uint64_t, fextl::robin_set<uint64_t>> CachedCodePages;
|
||||
|
||||
uintptr_t PagePointer;
|
||||
uintptr_t PageMemory;
|
||||
uintptr_t L1Pointer;
|
||||
uintptr_t L1PointerMask;
|
||||
|
||||
size_t TotalCacheSize;
|
||||
|
||||
// Start with 8k entries in L1 to give 128KB of L1 cache to each thread.
|
||||
// Max out at 1 million entries to give each thread 16MB of L1 cache maximum.
|
||||
constexpr static size_t MIN_L1_ENTRIES = 8 * 1024; // Must be a power of 2
|
||||
constexpr static size_t MAX_L1_ENTRIES = 1 * 1024 * 1024; // Must be a power of 2
|
||||
|
||||
constexpr static size_t CODE_SIZE = 128 * 1024 * 1024;
|
||||
constexpr static size_t SIZE_PER_PAGE = 4096 * sizeof(LookupCacheEntry);
|
||||
constexpr static size_t L1_SIZE = L1_ENTRIES * sizeof(LookupCacheEntry);
|
||||
constexpr static size_t SIZE_PER_PAGE = FEXCore::Utils::FEX_PAGE_SIZE * sizeof(LookupCacheEntry);
|
||||
constexpr static size_t MAX_L1_SIZE = MAX_L1_ENTRIES * sizeof(LookupCacheEntry);
|
||||
|
||||
size_t AllocateOffset {};
|
||||
|
||||
FEXCore::Context::ContextImpl* ctx;
|
||||
uint64_t VirtualMemSize {};
|
||||
|
||||
size_t CurrentL1Entries = MIN_L1_ENTRIES;
|
||||
uint64_t L2L3CacheHits {};
|
||||
std::chrono::time_point<std::chrono::system_clock> LastPeriod {};
|
||||
constexpr static std::chrono::seconds SamplePeriod {1};
|
||||
FEX_CONFIG_OPT(DynamicL1CacheIncreaseCountHeuristic, DYNAMICL1CACHEINCREASECOUNTHEURISTIC);
|
||||
FEX_CONFIG_OPT(DynamicL1CacheDecreaseCountHeuristic, DYNAMICL1CACHEDECREASECOUNTHEURISTIC);
|
||||
|
||||
FEX_CONFIG_OPT(DynamicL1Cache, DYNAMICL1CACHE);
|
||||
FEX_CONFIG_OPT(DisableL2Cache, DISABLEL2CACHE);
|
||||
};
|
||||
} // namespace FEXCore
|
||||
File diff suppressed because it is too large.
Load diff
@@ -139,27 +139,28 @@ public:
|
||||
FlushRegisterCache();
|
||||
return _Jump(_TargetBlock);
|
||||
}
|
||||
IRPair<IROp_CondJump> CondJump(Ref _Cmp1, Ref _Cmp2, Ref _TrueBlock, Ref _FalseBlock, CondClassType _Cond = {COND_NEQ},
|
||||
IRPair<IROp_CondJump> CondJump(Ref _Cmp1, Ref _Cmp2, Ref _TrueBlock, Ref _FalseBlock, CondClass _Cond = CondClass::NEQ,
|
||||
IR::OpSize _CompareSize = OpSize::iInvalid) {
|
||||
FlushRegisterCache();
|
||||
return _CondJump(_Cmp1, _Cmp2, _TrueBlock, _FalseBlock, _Cond, _CompareSize);
|
||||
}
|
||||
IRPair<IROp_CondJump> CondJump(Ref ssa0, CondClassType cond = {COND_NEQ}) {
|
||||
IRPair<IROp_CondJump> CondJump(Ref ssa0, CondClass cond = CondClass::NEQ) {
|
||||
FlushRegisterCache();
|
||||
return _CondJump(ssa0, cond);
|
||||
}
|
||||
IRPair<IROp_CondJump> CondJump(Ref ssa0, Ref ssa1, Ref ssa2, CondClassType cond = {COND_NEQ}) {
|
||||
IRPair<IROp_CondJump> CondJump(Ref ssa0, Ref ssa1, Ref ssa2, CondClass cond = CondClass::NEQ) {
|
||||
FlushRegisterCache();
|
||||
return _CondJump(ssa0, ssa1, ssa2, cond);
|
||||
}
|
||||
IRPair<IROp_CondJump> CondJumpNZCV(CondClassType Cond) {
|
||||
IRPair<IROp_CondJump> CondJumpNZCV(CondClass Cond) {
|
||||
FlushRegisterCache();
|
||||
return _CondJump(InvalidNode, InvalidNode, InvalidNode, InvalidNode, Cond, OpSize::iInvalid, true);
|
||||
}
|
||||
IRPair<IROp_CondJump> CondJumpBit(Ref Src, unsigned Bit, bool Set) {
|
||||
FlushRegisterCache();
|
||||
auto InlineConst = _InlineConstant(Bit);
|
||||
return _CondJump(Src, InlineConst, InvalidNode, InvalidNode, {Set ? COND_TSTNZ : COND_TSTZ}, OpSize::iInvalid, false);
|
||||
auto Cond = Set ? CondClass::TSTNZ : CondClass::TSTZ;
|
||||
return _CondJump(Src, InlineConst, InvalidNode, InvalidNode, Cond, OpSize::iInvalid, false);
|
||||
}
|
||||
IRPair<IROp_ExitFunction> ExitFunction(Ref NewRIP, BranchHint Hint = BranchHint::None) {
|
||||
FlushRegisterCache();
|
||||
@@ -251,7 +252,7 @@ public:
|
||||
auto ExitBlock = CreateNewCodeBlockAfter(BackwardBlock);
|
||||
|
||||
auto DF = GetRFLAG(X86State::RFLAG_DF_RAW_LOC);
|
||||
CondJump(DF, Zero, ForwardBlock, BackwardBlock, {COND_EQ});
|
||||
CondJump(DF, Zero, ForwardBlock, BackwardBlock, CondClass::EQ);
|
||||
|
||||
for (auto D = 0; D < 2; ++D) {
|
||||
SetCurrentCodeBlock(D ? BackwardBlock : ForwardBlock);
|
||||
@@ -564,7 +565,7 @@ public:
|
||||
template<IR::OpSize DstElementSize, IR::OpSize SrcElementSize>
|
||||
void AVXInsertScalar_CVT_Float_To_Float(OpcodeArgs);
|
||||
|
||||
RoundType TranslateRoundType(uint8_t Mode);
|
||||
RoundMode TranslateRoundType(uint8_t Mode);
|
||||
|
||||
template<IR::OpSize ElementSize>
|
||||
void InsertScalarRound(OpcodeArgs);
|
||||
@@ -758,7 +759,6 @@ public:
|
||||
void X87FXTRACT(OpcodeArgs);
|
||||
void X87FYL2X(OpcodeArgs, bool IsFYL2XP1);
|
||||
void X87LDENV(OpcodeArgs);
|
||||
void X87LDSW(OpcodeArgs);
|
||||
void X87ModifySTP(OpcodeArgs, bool Inc);
|
||||
void X87OpHelper(OpcodeArgs, FEXCore::IR::IROps IROp, bool ZeroC2);
|
||||
|
||||
@@ -907,6 +907,10 @@ public:
|
||||
void VPCLMULQDQOp(OpcodeArgs);
|
||||
|
||||
void CRC32(OpcodeArgs);
|
||||
void Extrq_imm(OpcodeArgs);
|
||||
void Insertq_imm(OpcodeArgs);
|
||||
void Extrq(OpcodeArgs);
|
||||
void Insertq(OpcodeArgs);
|
||||
|
||||
void BreakOp(OpcodeArgs, FEXCore::IR::BreakDefinition BreakDefinition);
|
||||
void UnimplementedOp(OpcodeArgs);
|
||||
@@ -985,7 +989,6 @@ public:
|
||||
void AVX128_VPSIGN(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void AVX128_UCOMISx(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void AVX128_VectorScalarInsertALU(OpcodeArgs, FEXCore::IR::IROps IROp, IR::OpSize ElementSize);
|
||||
Ref AVX128_VFCMPImpl(IR::OpSize ElementSize, Ref Src1, Ref Src2, uint8_t CompType);
|
||||
void AVX128_VFCMP(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void AVX128_InsertScalarFCMP(OpcodeArgs, IR::OpSize ElementSize);
|
||||
void AVX128_MOVBetweenGPR_FPR(OpcodeArgs);
|
||||
@@ -1004,9 +1007,7 @@ public:
|
||||
void AVX128_VINSERT(OpcodeArgs);
|
||||
void AVX128_VINSERTPS(OpcodeArgs);
|
||||
|
||||
Ref AVX128_PHSUBImpl(Ref Src1, Ref Src2, size_t ElementSize);
|
||||
void AVX128_VPHSUB(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
void AVX128_VPHSUBSW(OpcodeArgs);
|
||||
|
||||
void AVX128_VADDSUBP(OpcodeArgs, IR::OpSize ElementSize);
|
||||
@@ -1098,8 +1099,8 @@ public:
|
||||
void AVX128_VFMAScalarImpl(OpcodeArgs, IROps IROp, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx);
|
||||
void AVX128_VFMAddSubImpl(OpcodeArgs, bool AddSub, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx);
|
||||
|
||||
RefPair AVX128_VPGatherQPSImpl(Ref Dest, Ref Mask, RefVSIB VSIB);
|
||||
RefPair AVX128_VPGatherImpl(OpSize Size, OpSize ElementLoadSize, OpSize AddrElementSize, RefPair Dest, RefPair Mask, RefVSIB VSIB);
|
||||
RefPair AVX128_VPGatherQPSImpl(OpcodeArgs, Ref Dest, Ref Mask, RefVSIB VSIB);
|
||||
RefPair AVX128_VPGatherImpl(OpcodeArgs, OpSize Size, OpSize ElementLoadSize, OpSize AddrElementSize, RefPair Dest, RefPair Mask, RefVSIB VSIB);
|
||||
|
||||
void AVX128_VPGATHER(OpcodeArgs, OpSize AddrElementSize);
|
||||
|
||||
@@ -1109,8 +1110,8 @@ public:
|
||||
// End of AVX 128-bit implementation
|
||||
|
||||
// AVX 256-bit operations
|
||||
void StoreResult_WithAVXInsert(VectorOpType Type, FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, Ref Value,
|
||||
IR::OpSize Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
|
||||
void StoreResult_WithAVXInsert(VectorOpType Type, RegClass Class, FEXCore::X86Tables::DecodedOp Op, Ref Value,
|
||||
IR::OpSize Align = IR::OpSize::iInvalid, MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
|
||||
if (Op->Dest.IsGPR() && Op->Dest.Data.GPR.GPR >= X86State::REG_XMM_0 && Op->Dest.Data.GPR.GPR <= X86State::REG_XMM_15 &&
|
||||
GetGuestVectorLength() == OpSize::i256Bit && Type == VectorOpType::SSE) {
|
||||
const auto gpr = Op->Dest.Data.GPR.GPR;
|
||||
@@ -1161,7 +1162,7 @@ public:
|
||||
}
|
||||
}
|
||||
|
||||
void StoreContextHelper(IR::OpSize Size, RegisterClassType Class, Ref Value, uint32_t Offset) {
|
||||
void StoreContextHelper(IR::OpSize Size, RegClass Class, Ref Value, uint32_t Offset) {
|
||||
// For i128Bit, we won't see a normal Constant to inline, but as a special
|
||||
// case we can replace with a 2x64-bit store which can use inline zeroes.
|
||||
if (Size == OpSize::i128Bit) {
|
||||
@@ -1173,7 +1174,7 @@ public:
|
||||
|
||||
if (Const->Constant == IR::NamedVectorConstant::NAMED_VECTOR_ZERO) {
|
||||
Ref Zero = _Constant(0);
|
||||
Ref STP = _StoreContextPair(IR::OpSize::i64Bit, GPRClass, Zero, Zero, Offset);
|
||||
Ref STP = _StoreContextPair(IR::OpSize::i64Bit, RegClass::GPR, Zero, Zero, Offset);
|
||||
|
||||
// XXX: This works around InlineConstant not having an associated
|
||||
// register class, else we'd just do InlineConstant above.
|
||||
@@ -1229,16 +1230,16 @@ public:
|
||||
|
||||
if (Index >= GPR0Index && Index <= GPR15Index) {
|
||||
Ref R = _StoreRegister(Value, GPRSize);
|
||||
R->Reg = PhysicalRegister(GPRFixedClass, Index - GPR0Index).Raw;
|
||||
R->Reg = PhysicalRegister(RegClass::GPRFixed, Index - GPR0Index).Raw;
|
||||
} else if (Index == PFIndex) {
|
||||
_StorePF(Value, GPRSize);
|
||||
} else if (Index == AFIndex) {
|
||||
_StoreAF(Value, GPRSize);
|
||||
} else if (Index >= FPR0Index && Index <= FPR15Index) {
|
||||
Ref R = _StoreRegister(Value, VectorSize);
|
||||
R->Reg = PhysicalRegister(FPRFixedClass, Index - FPR0Index).Raw;
|
||||
R->Reg = PhysicalRegister(RegClass::FPRFixed, Index - FPR0Index).Raw;
|
||||
} else if (Index == DFIndex) {
|
||||
_StoreContext(OpSize::i8Bit, GPRClass, Value, offsetof(Core::CPUState, flags[X86State::RFLAG_DF_RAW_LOC]));
|
||||
_StoreContextGPR(OpSize::i8Bit, Value, offsetof(Core::CPUState, flags[X86State::RFLAG_DF_RAW_LOC]));
|
||||
} else {
|
||||
bool Partial = RegCache.Partial & (1ull << Index);
|
||||
auto Size = Partial ? OpSize::i64Bit : CacheIndexToOpSize(Index);
|
||||
@@ -1263,7 +1264,7 @@ public:
|
||||
StoreContextHelper(Size, Class, Value, Offset);
|
||||
// If Partial and MMX register, then we need to store all 1s in bits 64-80
|
||||
if (Partial && Index >= MM0Index && Index <= MM7Index) {
|
||||
_StoreContext(OpSize::i16Bit, IR::GPRClass, Constant(0xFFFF), Offset + 8);
|
||||
_StoreContextGPR(OpSize::i16Bit, Constant(0xFFFF), Offset + 8);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1553,23 +1554,63 @@ private:
|
||||
|
||||
AddressMode DecodeAddress(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, MemoryAccessType AccessType, bool IsLoad);
|
||||
|
||||
Ref LoadSource(RegisterClassType Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags,
|
||||
Ref LoadSource(RegClass Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags,
|
||||
const LoadSourceOptions& Options = {});
|
||||
Ref LoadSource_WithOpSize(RegisterClassType Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand,
|
||||
IR::OpSize OpSize, uint32_t Flags, const LoadSourceOptions& Options = {});
|
||||
void StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op,
|
||||
const FEXCore::X86Tables::DecodedOperand& Operand, const Ref Src, IR::OpSize OpSize, IR::OpSize Align,
|
||||
MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
|
||||
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, const FEXCore::X86Tables::DecodedOperand& Operand,
|
||||
const Ref Src, IR::OpSize Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
|
||||
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, const Ref Src, IR::OpSize Align,
|
||||
Ref LoadSourceGPR(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags,
|
||||
const LoadSourceOptions& Options = {}) {
|
||||
return LoadSource(RegClass::GPR, Op, Operand, Flags, Options);
|
||||
}
|
||||
Ref LoadSourceFPR(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags,
|
||||
const LoadSourceOptions& Options = {}) {
|
||||
return LoadSource(RegClass::FPR, Op, Operand, Flags, Options);
|
||||
}
|
||||
|
||||
Ref LoadSource_WithOpSize(RegClass Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, IR::OpSize OpSize,
|
||||
uint32_t Flags, const LoadSourceOptions& Options = {});
|
||||
Ref LoadSourceGPR_WithOpSize(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, IR::OpSize OpSize, uint32_t Flags,
|
||||
const LoadSourceOptions& Options = {}) {
|
||||
return LoadSource_WithOpSize(RegClass::GPR, Op, Operand, OpSize, Flags, Options);
|
||||
}
|
||||
Ref LoadSourceFPR_WithOpSize(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, IR::OpSize OpSize, uint32_t Flags,
|
||||
const LoadSourceOptions& Options = {}) {
|
||||
return LoadSource_WithOpSize(RegClass::FPR, Op, Operand, OpSize, Flags, Options);
|
||||
}
|
||||
|
||||
void StoreResult_WithOpSize(RegClass Class, X86Tables::DecodedOp Op, const X86Tables::DecodedOperand& Operand, Ref Src, IR::OpSize OpSize,
|
||||
IR::OpSize Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
|
||||
void StoreResultGPR_WithOpSize(X86Tables::DecodedOp Op, const X86Tables::DecodedOperand& Operand, Ref Src, IR::OpSize OpSize,
|
||||
IR::OpSize Align = IR::OpSize::iInvalid, MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
|
||||
StoreResult_WithOpSize(RegClass::GPR, Op, Operand, Src, OpSize, Align, AccessType);
|
||||
}
|
||||
void StoreResultFPR_WithOpSize(X86Tables::DecodedOp Op, const X86Tables::DecodedOperand& Operand, Ref Src, IR::OpSize OpSize,
|
||||
IR::OpSize Align = IR::OpSize::iInvalid, MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
|
||||
StoreResult_WithOpSize(RegClass::FPR, Op, Operand, Src, OpSize, Align, AccessType);
|
||||
}
|
||||
|
||||
void StoreResult(RegClass Class, X86Tables::DecodedOp Op, const X86Tables::DecodedOperand& Operand, Ref Src, OpSize Align,
|
||||
MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
|
||||
void StoreResultGPR(X86Tables::DecodedOp Op, const X86Tables::DecodedOperand& Operand, Ref Src, OpSize Align = OpSize::iInvalid,
|
||||
MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
|
||||
StoreResult(RegClass::GPR, Op, Operand, Src, Align, AccessType);
|
||||
}
|
||||
void StoreResultFPR(X86Tables::DecodedOp Op, const X86Tables::DecodedOperand& Operand, Ref Src, OpSize Align = OpSize::iInvalid,
|
||||
MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
|
||||
StoreResult(RegClass::FPR, Op, Operand, Src, Align, AccessType);
|
||||
}
|
||||
|
||||
void StoreResult(RegClass Class, X86Tables::DecodedOp Op, Ref Src, OpSize Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
|
||||
void StoreResultGPR(X86Tables::DecodedOp Op, Ref Src, OpSize Align = OpSize::iInvalid, MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
|
||||
StoreResult(RegClass::GPR, Op, Src, Align, AccessType);
|
||||
}
|
||||
void StoreResultFPR(X86Tables::DecodedOp Op, Ref Src, OpSize Align = OpSize::iInvalid, MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
|
||||
StoreResult(RegClass::FPR, Op, Src, Align, AccessType);
|
||||
}
|
||||
|
||||
// In several instances, it's desirable to get a base address with the segment offset
|
||||
// applied to it. This pulls all the common-case appending into a single set of functions.
|
||||
[[nodiscard]]
|
||||
Ref MakeSegmentAddress(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, IR::OpSize OpSize) {
|
||||
Ref Mem = LoadSource_WithOpSize(GPRClass, Op, Operand, OpSize, Op->Flags, {.LoadData = false});
|
||||
Ref Mem = LoadSourceGPR_WithOpSize(Op, Operand, OpSize, Op->Flags, {.LoadData = false});
|
||||
return AppendSegmentOffset(Mem, Op->Flags);
|
||||
}
|
||||
[[nodiscard]]
|
||||
@@ -1614,6 +1655,9 @@ private:
|
||||
return IR::SizeToOpSize(GetSrcSize(Op));
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
IR::OpSize GetStringOpSize(X86Tables::DecodedOp Op) const;
|
||||
|
||||
// Set flag tracking to prepare for an operation that directly writes NZCV.
|
||||
void HandleNZCVWrite() {
|
||||
CachedNZCV = nullptr;
|
||||
@@ -1805,14 +1849,15 @@ private:
|
||||
// For DF, we need to transform 0/1 into 1/-1
|
||||
StoreDF(_SubShift(OpSize::i64Bit, Constant(1), Value, ShiftType::LSL, 1));
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_TF_RAW_LOC) {
|
||||
auto PackedTF = _LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
|
||||
auto PackedTF = _LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
|
||||
// An exception should still be raised after an instruction that unsets TF, leave the unblocked bit set but unset
|
||||
// the TF bit to cause such behaviour. The handling code at the start of the next block will then unset the
|
||||
// unblocked bit before raising the exception.
|
||||
auto NewPackedTF = _Select(FEXCore::IR::COND_EQ, Value, Constant(0), _And(OpSize::i32Bit, PackedTF, Constant(~1)), Constant(1));
|
||||
_StoreContext(OpSize::i8Bit, GPRClass, NewPackedTF, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
|
||||
auto NewPackedTF =
|
||||
_Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, Value, Constant(0), _And(OpSize::i32Bit, PackedTF, Constant(~1)), Constant(1));
|
||||
_StoreContextGPR(OpSize::i8Bit, NewPackedTF, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
|
||||
} else {
|
||||
_StoreContext(OpSize::i8Bit, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
|
||||
_StoreContextGPR(OpSize::i8Bit, Value, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1838,12 +1883,12 @@ private:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static CondClassType CondForNZCVBit(unsigned BitOffset, bool Invert) {
|
||||
static CondClass CondForNZCVBit(unsigned BitOffset, bool Invert) {
|
||||
switch (BitOffset) {
|
||||
case X86State::RFLAG_SF_RAW_LOC: return {Invert ? COND_PL : COND_MI};
|
||||
case X86State::RFLAG_ZF_RAW_LOC: return {Invert ? COND_NEQ : COND_EQ};
|
||||
case X86State::RFLAG_CF_RAW_LOC: return {Invert ? COND_ULT : COND_UGE};
|
||||
case X86State::RFLAG_OF_RAW_LOC: return {Invert ? COND_FNU : COND_FU};
|
||||
case X86State::RFLAG_SF_RAW_LOC: return Invert ? CondClass::PL : CondClass::MI;
|
||||
case X86State::RFLAG_ZF_RAW_LOC: return Invert ? CondClass::NEQ : CondClass::EQ;
|
||||
case X86State::RFLAG_CF_RAW_LOC: return Invert ? CondClass::ULT : CondClass::UGE;
|
||||
case X86State::RFLAG_OF_RAW_LOC: return Invert ? CondClass::FNU : CondClass::FU;
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
@@ -1874,11 +1919,11 @@ private:
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static RegisterClassType CacheIndexClass(int Index) {
|
||||
static RegClass CacheIndexClass(int Index) {
|
||||
if ((Index >= MM0Index && Index <= MM7Index) || Index >= FPR0Index) {
|
||||
return FPRClass;
|
||||
return RegClass::FPR;
|
||||
} else {
|
||||
return GPRClass;
|
||||
return RegClass::GPR;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1910,14 +1955,14 @@ private:
|
||||
RegCache.Written &= ~Bit;
|
||||
}
|
||||
|
||||
Ref LoadRegCache(uint64_t Offset, uint8_t Index, RegisterClassType RegClass, IR::OpSize Size) {
|
||||
Ref LoadRegCache(uint64_t Offset, uint8_t Index, RegClass Class, IR::OpSize Size) {
|
||||
LOGMAN_THROW_A_FMT(Index < 64, "valid index");
|
||||
uint64_t Bit = (1ull << (uint64_t)Index);
|
||||
|
||||
if (Size == OpSize::i128Bit && (RegCache.Partial & Bit)) {
|
||||
// We need to load the full register extend if we previously did a partial access.
|
||||
Ref Value = RegCache.Value[Index];
|
||||
Ref Full = _LoadContext(Size, RegClass, Offset);
|
||||
Ref Full = _LoadContext(Size, Class, Offset);
|
||||
|
||||
// If we did a partial store, we're inserting into the full register
|
||||
if (RegCache.Written & Bit) {
|
||||
@@ -1931,7 +1976,7 @@ private:
|
||||
if (Index == DFIndex) {
|
||||
RegCache.Value[Index] = _LoadDF();
|
||||
} else if ((Index >= MM0Index && Index <= MM7Index) || Index >= AVXHigh0Index) {
|
||||
RegCache.Value[Index] = _LoadContext(Size, RegClass, Offset);
|
||||
RegCache.Value[Index] = _LoadContext(Size, Class, Offset);
|
||||
|
||||
// We may have done a partial load, this requires special handling.
|
||||
if (Size == OpSize::i64Bit) {
|
||||
@@ -1942,7 +1987,7 @@ private:
|
||||
} else if (Index == AFIndex) {
|
||||
RegCache.Value[Index] = _LoadAF(Size);
|
||||
} else {
|
||||
RegCache.Value[Index] = _LoadRegister(Offset, RegClass, Size);
|
||||
RegCache.Value[Index] = _LoadRegister(Offset, Class, Size);
|
||||
}
|
||||
|
||||
RegCache.Cached |= Bit;
|
||||
@@ -1951,21 +1996,21 @@ private:
|
||||
return RegCache.Value[Index];
|
||||
}
|
||||
|
||||
RefPair AllocatePair(FEXCore::IR::RegisterClassType Class, IR::OpSize Size) {
|
||||
if (Class == FPRClass) {
|
||||
RefPair AllocatePair(RegClass Class, IR::OpSize Size) {
|
||||
if (Class == RegClass::FPR) {
|
||||
return {_AllocateFPR(Size, Size), _AllocateFPR(Size, Size)};
|
||||
} else {
|
||||
return {_AllocateGPR(false), _AllocateGPR(false)};
|
||||
}
|
||||
}
|
||||
|
||||
RefPair LoadContextPair_Uncached(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, unsigned Offset) {
|
||||
RefPair LoadContextPair_Uncached(RegClass Class, IR::OpSize Size, unsigned Offset) {
|
||||
RefPair Values = AllocatePair(Class, Size);
|
||||
_LoadContextPair(Size, Class, Offset, Values.Low, Values.High);
|
||||
return Values;
|
||||
}
|
||||
|
||||
RefPair LoadRegCachePair(uint64_t Offset, uint8_t Index, RegisterClassType RegClass, IR::OpSize Size) {
|
||||
RefPair LoadRegCachePair(uint64_t Offset, uint8_t Index, RegClass Class, IR::OpSize Size) {
|
||||
LOGMAN_THROW_A_FMT(Index != DFIndex, "must be pairable");
|
||||
LOGMAN_THROW_A_FMT(Size != IR::OpSize::iUnsized, "Invalid size!");
|
||||
|
||||
@@ -1973,7 +2018,7 @@ private:
|
||||
uint64_t Bits = (3ull << (uint64_t)Index);
|
||||
const auto SizeInt = IR::OpSizeToSize(Size);
|
||||
if (((RegCache.Partial | RegCache.Cached) & Bits) == 0 && ((Offset / SizeInt) < 64)) {
|
||||
auto Values = LoadContextPair_Uncached(RegClass, Size, Offset);
|
||||
auto Values = LoadContextPair_Uncached(Class, Size, Offset);
|
||||
RegCache.Value[Index] = Values.Low;
|
||||
RegCache.Value[Index + 1] = Values.High;
|
||||
RegCache.Cached |= Bits;
|
||||
@@ -1985,13 +2030,13 @@ private:
|
||||
|
||||
// Fallback on a pair of loads
|
||||
return {
|
||||
.Low = LoadRegCache(Offset, Index, RegClass, Size),
|
||||
.High = LoadRegCache(Offset + SizeInt, Index + 1, RegClass, Size),
|
||||
.Low = LoadRegCache(Offset, Index, Class, Size),
|
||||
.High = LoadRegCache(Offset + SizeInt, Index + 1, Class, Size),
|
||||
};
|
||||
}
|
||||
|
||||
Ref LoadGPR(uint8_t Reg) {
|
||||
return LoadRegCache(Reg, GPR0Index + Reg, GPRClass, GetGPROpSize());
|
||||
return LoadRegCache(Reg, GPR0Index + Reg, RegClass::GPR, GetGPROpSize());
|
||||
}
|
||||
|
||||
Ref LoadContext(IR::OpSize Size, uint8_t Index) {
|
||||
@@ -2007,7 +2052,7 @@ private:
|
||||
}
|
||||
|
||||
Ref LoadXMMRegister(uint8_t Reg) {
|
||||
return LoadRegCache(Reg, FPR0Index + Reg, FPRClass, GetGuestVectorLength());
|
||||
return LoadRegCache(Reg, FPR0Index + Reg, RegClass::FPR, GetGuestVectorLength());
|
||||
}
|
||||
|
||||
Ref LoadDF() {
|
||||
@@ -2062,7 +2107,7 @@ private:
|
||||
// Recover the sign bit, it is the logical DF value
|
||||
return _Lshr(OpSize::i64Bit, LoadDF(), Constant(63));
|
||||
} else {
|
||||
return _LoadContext(OpSize::i8Bit, GPRClass, offsetof(Core::CPUState, flags[BitOffset]));
|
||||
return _LoadContextGPR(OpSize::i8Bit, offsetof(Core::CPUState, flags[BitOffset]));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2079,18 +2124,18 @@ private:
|
||||
}
|
||||
|
||||
// Safe version of NZCVSelect that handles inverted carries automatically.
|
||||
Ref NZCVSelect(OpSize OpSize, CondClassType Cond, Ref TrueV, Ref FalseV, bool CarryIsInverted = false) {
|
||||
Ref NZCVSelect(OpSize OpSize, CondClass Cond, Ref TrueV, Ref FalseV, bool CarryIsInverted = false) {
|
||||
switch (Cond) {
|
||||
case IR::COND_UGE: /* cs */
|
||||
case IR::COND_ULT: /* cc */
|
||||
case CondClass::UGE: /* cs */
|
||||
case CondClass::ULT: /* cc */
|
||||
// Invert the condition to match our expectations.
|
||||
if (CarryIsInverted != CFInverted) {
|
||||
Cond = {Cond == COND_UGE ? COND_ULT : COND_UGE};
|
||||
Cond = (Cond == CondClass::UGE) ? CondClass::ULT : CondClass::UGE;
|
||||
}
|
||||
break;
|
||||
|
||||
case IR::COND_UGT: /* hi */
|
||||
case IR::COND_ULE: /* ls */
|
||||
case CondClass::UGT: /* hi */
|
||||
case CondClass::ULE: /* ls */
|
||||
// No clever optimization we can do here, rectify carry itself.
|
||||
RectifyCarryInvert(CarryIsInverted);
|
||||
break;
|
||||
@@ -2179,7 +2224,7 @@ private:
|
||||
|
||||
HandleNZCV_RMW();
|
||||
CalculatePF(_ShiftFlags(OpSizeFromSrc(Op), Result, Dest, Shift, Src, OldPF, CFInverted));
|
||||
StoreResult(GPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultGPR(Op, Result);
|
||||
}
|
||||
|
||||
// Helper to derive Dest by a given builder-using Expression with the opcode
|
||||
@@ -2252,8 +2297,7 @@ private:
|
||||
CachedIndexedNamedVectorConstants.clear();
|
||||
}
|
||||
|
||||
std::optional<CondClassType> DecodeNZCVCondition(uint8_t OP);
|
||||
Ref SelectBit(Ref Cmp, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue);
|
||||
std::optional<CondClass> DecodeNZCVCondition(uint8_t OP);
|
||||
Ref SelectCC0All1(uint8_t OP);
|
||||
|
||||
/**
|
||||
@@ -2267,8 +2311,8 @@ private:
|
||||
if (Size != OpSize::i32Bit) {
|
||||
return;
|
||||
}
|
||||
auto Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags);
|
||||
StoreResult(GPRClass, Op, Dest, OpSize::iInvalid);
|
||||
auto Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags);
|
||||
StoreResultGPR(Op, Dest);
|
||||
}
|
||||
|
||||
using ZeroShiftFunctionPtr = void (OpDispatchBuilder::*)(FEXCore::X86Tables::DecodedOp Op);
|
||||
@@ -2303,7 +2347,7 @@ private:
|
||||
|
||||
///< Jump to zeroshift block or end block depending on if it was provided.
|
||||
IRPair<IROp_CodeBlock> TailHandling = ZeroShiftResult ? ZeroShiftBlock : EndBlock;
|
||||
CondJump(Shift, Zero, TailHandling, SetBlock, {COND_EQ});
|
||||
CondJump(Shift, Zero, TailHandling, SetBlock, CondClass::EQ);
|
||||
|
||||
SetCurrentCodeBlock(SetBlock);
|
||||
StartNewBlock();
|
||||
@@ -2346,9 +2390,7 @@ private:
|
||||
void CalculateFlags_MUL(IR::OpSize SrcSize, Ref Res, Ref High);
|
||||
void CalculateFlags_UMUL(Ref High);
|
||||
void CalculateFlags_Logical(IR::OpSize SrcSize, Ref Res);
|
||||
void CalculateFlags_ShiftLeft(IR::OpSize SrcSize, Ref Res, Ref Src1, Ref Src2);
|
||||
void CalculateFlags_ShiftLeftImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRight(IR::OpSize SrcSize, Ref Res, Ref Src1, Ref Src2);
|
||||
void CalculateFlags_ShiftRightImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRightDoubleImmediate(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRightImmediateCommon(IR::OpSize SrcSize, Ref Res, Ref Src1, uint64_t Shift);
|
||||
@@ -2365,7 +2407,7 @@ private:
|
||||
LOGMAN_THROW_A_FMT(MMXState == MMXState_X87, "Expected state to be x87");
|
||||
_StackForceSlow();
|
||||
SetX87Top(Constant(0)); // top reset to zero
|
||||
_StoreContext(OpSize::i8Bit, GPRClass, Constant(0xFFFFUL), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
_StoreContextGPR(OpSize::i8Bit, Constant(0xFFFFUL), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
MMXState = MMXState_MMX;
|
||||
}
|
||||
|
||||
@@ -2402,44 +2444,62 @@ private:
|
||||
IROp_IRHeader* CurrentHeader {};
|
||||
|
||||
[[nodiscard]]
|
||||
bool IsTSOEnabled(FEXCore::IR::RegisterClassType Class) const {
|
||||
bool IsTSOEnabled(RegClass Class) const {
|
||||
if (ForceTSO == ForceTSOMode::ForceEnabled) {
|
||||
return true;
|
||||
} else if (ForceTSO == ForceTSOMode::ForceDisabled) {
|
||||
return false;
|
||||
} else if (Class == FPRClass) {
|
||||
} else if (Class == RegClass::FPR) {
|
||||
return CTX->IsVectorAtomicTSOEnabled();
|
||||
} else {
|
||||
return CTX->IsAtomicTSOEnabled();
|
||||
}
|
||||
}
|
||||
|
||||
Ref _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, Ref Addr, Ref Value, IR::OpSize Align = IR::OpSize::i8Bit) {
|
||||
Ref _StoreMemAutoTSO(RegClass Class, OpSize Size, Ref Addr, Ref Value, OpSize Align = OpSize::i8Bit) {
|
||||
if (IsTSOEnabled(Class)) {
|
||||
return _StoreMemTSO(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
return _StoreMemTSO(Class, Size, Value, Addr, Invalid(), Align, MemOffsetType::SXTX, 1);
|
||||
} else {
|
||||
return _StoreMem(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
return _StoreMem(Class, Size, Value, Addr, Invalid(), Align, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
}
|
||||
|
||||
Ref _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, Ref ssa0, IR::OpSize Align = IR::OpSize::i8Bit) {
|
||||
if (IsTSOEnabled(Class)) {
|
||||
return _LoadMemTSO(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
} else {
|
||||
return _LoadMem(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
Ref _StoreMemGPRAutoTSO(OpSize Size, Ref Addr, Ref Value, OpSize Align = OpSize::i8Bit) {
|
||||
return _StoreMemAutoTSO(RegClass::GPR, Size, Addr, Value, Align);
|
||||
}
|
||||
Ref _StoreMemFPRAutoTSO(OpSize Size, Ref Addr, Ref Value, OpSize Align = OpSize::i8Bit) {
|
||||
return _StoreMemAutoTSO(RegClass::FPR, Size, Addr, Value, Align);
|
||||
}
|
||||
|
||||
Ref _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, AddressMode A, IR::OpSize Align = IR::OpSize::i8Bit) {
|
||||
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
A = SelectAddressMode(this, A, GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, AtomicTSO, Class != GPRClass, Size);
|
||||
Ref _LoadMemAutoTSO(RegClass Class, OpSize Size, Ref ssa0, OpSize Align = OpSize::i8Bit) {
|
||||
if (IsTSOEnabled(Class)) {
|
||||
return _LoadMemTSO(Class, Size, ssa0, Invalid(), Align, MemOffsetType::SXTX, 1);
|
||||
} else {
|
||||
return _LoadMem(Class, Size, ssa0, Invalid(), Align, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
}
|
||||
Ref _LoadMemGPRAutoTSO(OpSize Size, Ref ssa0, OpSize Align = OpSize::i8Bit) {
|
||||
return _LoadMemAutoTSO(RegClass::GPR, Size, ssa0, Align);
|
||||
}
|
||||
Ref _LoadMemFPRAutoTSO(OpSize Size, Ref ssa0, OpSize Align = OpSize::i8Bit) {
|
||||
return _LoadMemAutoTSO(RegClass::FPR, Size, ssa0, Align);
|
||||
}
|
||||
|
||||
Ref _LoadMemAutoTSO(RegClass Class, OpSize Size, const AddressMode& A, OpSize Align = OpSize::i8Bit) {
|
||||
const bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
const auto B = SelectAddressMode(this, A, GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, AtomicTSO, Class != RegClass::GPR, Size);
|
||||
|
||||
if (AtomicTSO) {
|
||||
return _LoadMemTSO(Class, Size, A.Base, A.Index, Align, A.IndexType, A.IndexScale);
|
||||
return _LoadMemTSO(Class, Size, B.Base, B.Index, Align, B.IndexType, B.IndexScale);
|
||||
} else {
|
||||
return _LoadMem(Class, Size, A.Base, A.Index, Align, A.IndexType, A.IndexScale);
|
||||
return _LoadMem(Class, Size, B.Base, B.Index, Align, B.IndexType, B.IndexScale);
|
||||
}
|
||||
}
|
||||
Ref _LoadMemGPRAutoTSO(OpSize Size, const AddressMode& A, OpSize Align = OpSize::i8Bit) {
|
||||
return _LoadMemAutoTSO(RegClass::GPR, Size, A, Align);
|
||||
}
|
||||
Ref _LoadMemFPRAutoTSO(OpSize Size, const AddressMode& A, OpSize Align = OpSize::i8Bit) {
|
||||
return _LoadMemAutoTSO(RegClass::FPR, Size, A, Align);
|
||||
}
|
||||
|
||||
AddressMode SelectPairAddressMode(AddressMode A, IR::OpSize Size) {
|
||||
LOGMAN_THROW_A_FMT(Size != IR::OpSize::iUnsized, "Invalid size!");
|
||||
@@ -2457,56 +2517,72 @@ private:
|
||||
}
|
||||
|
||||
|
||||
RefPair LoadMemPair(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, Ref Base, unsigned Offset) {
|
||||
RefPair LoadMemPair(RegClass Class, OpSize Size, Ref Base, uint32_t Offset) {
|
||||
RefPair Values = AllocatePair(Class, Size);
|
||||
_LoadMemPair(Class, Size, Base, Offset, Values.Low, Values.High);
|
||||
return Values;
|
||||
}
|
||||
RefPair LoadMemPairFPR(OpSize Size, Ref Base, uint32_t Offset) {
|
||||
return LoadMemPair(RegClass::FPR, Size, Base, Offset);
|
||||
}
|
||||
|
||||
RefPair _LoadMemPairAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, AddressMode A, IR::OpSize Align = IR::OpSize::i8Bit) {
|
||||
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
RefPair _LoadMemPairAutoTSO(RegClass Class, OpSize Size, const AddressMode& A, OpSize Align = OpSize::i8Bit) {
|
||||
const bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
|
||||
// Use ldp if possible, otherwise fallback on two loads.
|
||||
if (!AtomicTSO && !A.Segment && Size >= OpSize::i32Bit & Size <= OpSize::i128Bit) {
|
||||
A = SelectPairAddressMode(A, Size);
|
||||
return LoadMemPair(Class, Size, A.Base, A.Offset);
|
||||
} else {
|
||||
AddressMode HighA = A;
|
||||
HighA.Offset += 16;
|
||||
|
||||
return {
|
||||
.Low = _LoadMemAutoTSO(Class, Size, A, Align),
|
||||
.High = _LoadMemAutoTSO(Class, Size, HighA, Align),
|
||||
};
|
||||
if (!AtomicTSO && !A.Segment && Size >= OpSize::i32Bit && Size <= OpSize::i128Bit) {
|
||||
const auto B = SelectPairAddressMode(A, Size);
|
||||
return LoadMemPair(Class, Size, B.Base, B.Offset);
|
||||
}
|
||||
|
||||
AddressMode HighA = A;
|
||||
HighA.Offset += 16;
|
||||
|
||||
return {
|
||||
.Low = _LoadMemAutoTSO(Class, Size, A, Align),
|
||||
.High = _LoadMemAutoTSO(Class, Size, HighA, Align),
|
||||
};
|
||||
}
|
||||
RefPair _LoadMemPairFPRAutoTSO(OpSize Size, const AddressMode& A, OpSize Align = OpSize::i8Bit) {
|
||||
return _LoadMemPairAutoTSO(RegClass::FPR, Size, A, Align);
|
||||
}
|
||||
|
||||
Ref _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, AddressMode A, Ref Value, IR::OpSize Align = IR::OpSize::i8Bit) {
|
||||
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
A = SelectAddressMode(this, A, GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, AtomicTSO, Class != GPRClass, Size);
|
||||
Ref _StoreMemAutoTSO(RegClass Class, OpSize Size, const AddressMode& A, Ref Value, OpSize Align = OpSize::i8Bit) {
|
||||
const bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
const auto B = SelectAddressMode(this, A, GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, AtomicTSO, Class != RegClass::GPR, Size);
|
||||
|
||||
if (AtomicTSO) {
|
||||
return _StoreMemTSO(Class, Size, Value, A.Base, A.Index, Align, A.IndexType, A.IndexScale);
|
||||
return _StoreMemTSO(Class, Size, Value, B.Base, B.Index, Align, B.IndexType, B.IndexScale);
|
||||
} else {
|
||||
return _StoreMem(Class, Size, Value, A.Base, A.Index, Align, A.IndexType, A.IndexScale);
|
||||
return _StoreMem(Class, Size, Value, B.Base, B.Index, Align, B.IndexType, B.IndexScale);
|
||||
}
|
||||
}
|
||||
Ref _StoreMemGPRAutoTSO(OpSize Size, const AddressMode& A, Ref Value, OpSize Align = OpSize::i8Bit) {
|
||||
return _StoreMemAutoTSO(RegClass::GPR, Size, A, Value, Align);
|
||||
}
|
||||
Ref _StoreMemFPRAutoTSO(OpSize Size, const AddressMode& A, Ref Value, OpSize Align = OpSize::i8Bit) {
|
||||
return _StoreMemAutoTSO(RegClass::FPR, Size, A, Value, Align);
|
||||
}
|
||||
|
||||
void _StoreMemPairAutoTSO(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, AddressMode A, Ref Value1, Ref Value2,
|
||||
IR::OpSize Align = IR::OpSize::i8Bit) {
|
||||
void _StoreMemPairAutoTSO(RegClass Class, OpSize Size, const AddressMode& A, Ref Value1, Ref Value2, OpSize Align = OpSize::i8Bit) {
|
||||
const auto SizeInt = IR::OpSizeToSize(Size);
|
||||
bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
const bool AtomicTSO = IsTSOEnabled(Class) && !A.NonTSO;
|
||||
|
||||
// Use stp if possible, otherwise fallback on two stores.
|
||||
if (!AtomicTSO && !A.Segment && Size >= OpSize::i32Bit & Size <= OpSize::i128Bit) {
|
||||
A = SelectPairAddressMode(A, Size);
|
||||
_StoreMemPair(Class, Size, Value1, Value2, A.Base, A.Offset);
|
||||
if (!AtomicTSO && !A.Segment && Size >= OpSize::i32Bit && Size <= OpSize::i128Bit) {
|
||||
const auto B = SelectPairAddressMode(A, Size);
|
||||
_StoreMemPair(Class, Size, Value1, Value2, B.Base, B.Offset);
|
||||
} else {
|
||||
_StoreMemAutoTSO(Class, Size, A, Value1, OpSize::i8Bit);
|
||||
A.Offset += SizeInt;
|
||||
_StoreMemAutoTSO(Class, Size, A, Value2, OpSize::i8Bit);
|
||||
auto B = A;
|
||||
|
||||
_StoreMemAutoTSO(Class, Size, B, Value1, OpSize::i8Bit);
|
||||
B.Offset += SizeInt;
|
||||
_StoreMemAutoTSO(Class, Size, B, Value2, OpSize::i8Bit);
|
||||
}
|
||||
}
|
||||
void _StoreMemPairFPRAutoTSO(OpSize Size, const AddressMode& A, Ref Value1, Ref Value2, OpSize Align = OpSize::i8Bit) {
|
||||
return _StoreMemPairAutoTSO(RegClass::FPR, Size, A, Value1, Value2, Align);
|
||||
}
|
||||
|
||||
Ref Pop(IR::OpSize Size, Ref SP_RMW) {
|
||||
Ref Value = _AllocateGPR(false);
|
||||
|
||||
@@ -35,20 +35,16 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_LoadSource_WithOpSize(
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(IsOperandMem(Operand, true), "only memory sources");
|
||||
|
||||
AddressMode A = DecodeAddress(Op, Operand, AccessType, true /* IsLoad */);
|
||||
|
||||
AddressMode HighA = A;
|
||||
HighA.Offset += 16;
|
||||
|
||||
if (Operand.IsSIB()) {
|
||||
const bool IsVSIB = (Op->Flags & X86Tables::DecodeFlags::FLAG_VSIB_BYTE) != 0;
|
||||
LOGMAN_THROW_A_FMT(!IsVSIB, "VSIB uses LoadVSIB instead");
|
||||
}
|
||||
|
||||
const AddressMode A = DecodeAddress(Op, Operand, AccessType, true /* IsLoad */);
|
||||
if (NeedsHigh) {
|
||||
return _LoadMemPairAutoTSO(FPRClass, OpSize::i128Bit, A, OpSize::i8Bit);
|
||||
return _LoadMemPairFPRAutoTSO(OpSize::i128Bit, A, OpSize::i8Bit);
|
||||
} else {
|
||||
return {.Low = _LoadMemAutoTSO(FPRClass, OpSize::i128Bit, A, OpSize::i8Bit)};
|
||||
return {.Low = _LoadMemFPRAutoTSO(OpSize::i128Bit, A, OpSize::i8Bit)};
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -95,9 +91,9 @@ void OpDispatchBuilder::AVX128_StoreResult_WithOpSize(FEXCore::X86Tables::Decode
|
||||
AddressMode A = DecodeAddress(Op, Operand, AccessType, false /* IsLoad */);
|
||||
|
||||
if (Src.High) {
|
||||
_StoreMemPairAutoTSO(FPRClass, OpSize::i128Bit, A, Src.Low, Src.High, OpSize::i8Bit);
|
||||
_StoreMemPairFPRAutoTSO(OpSize::i128Bit, A, Src.Low, Src.High, OpSize::i8Bit);
|
||||
} else {
|
||||
_StoreMemAutoTSO(FPRClass, OpSize::i128Bit, A, Src.Low, OpSize::i8Bit);
|
||||
_StoreMemFPRAutoTSO(OpSize::i128Bit, A, Src.Low, OpSize::i8Bit);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -151,13 +147,13 @@ void OpDispatchBuilder::AVX128_VMOVScalarImpl(OpcodeArgs, IR::OpSize ElementSize
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, RefPair {.Low = Result, .High = High});
|
||||
} else if (Op->Dest.IsGPR()) {
|
||||
// VMOVSS/SD xmm1, mem32/mem64
|
||||
Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], ElementSize, Op->Flags);
|
||||
Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[1], ElementSize, Op->Flags);
|
||||
auto High = LoadZeroVector(OpSize::i128Bit);
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, RefPair {.Low = Src, .High = High});
|
||||
} else {
|
||||
// VMOVSS/SD mem32/mem64, xmm1
|
||||
auto Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src.Low, ElementSize, OpSize::iInvalid);
|
||||
StoreResultFPR_WithOpSize(Op, Op->Dest, Src.Low, ElementSize);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -351,7 +347,7 @@ void OpDispatchBuilder::AVX128_MOVVectorNT(OpcodeArgs) {
|
||||
if (Op->Dest.IsGPR()) {
|
||||
///< MOVNTDQA load non-temporal comes from SSE4.1 and is extended by AVX/AVX2.
|
||||
RefPair Src {};
|
||||
Ref SrcAddr = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Ref SrcAddr = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Src.Low = _VLoadNonTemporal(OpSize::i128Bit, SrcAddr, 0);
|
||||
|
||||
if (Is128Bit) {
|
||||
@@ -362,7 +358,7 @@ void OpDispatchBuilder::AVX128_MOVVectorNT(OpcodeArgs) {
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Src);
|
||||
} else {
|
||||
auto Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, !Is128Bit, MemoryAccessType::STREAM);
|
||||
Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
|
||||
if (Is128Bit) {
|
||||
// Single store non-temporal for 128-bit operations.
|
||||
@@ -379,7 +375,7 @@ void OpDispatchBuilder::AVX128_MOVQ(OpcodeArgs) {
|
||||
if (Op->Src[0].IsGPR()) {
|
||||
Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false);
|
||||
} else {
|
||||
Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::i64Bit, Op->Flags);
|
||||
Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::i64Bit, Op->Flags);
|
||||
}
|
||||
|
||||
// This instruction is a bit special that if the destination is a register then it'll ZEXT the 64bit source to 256bit
|
||||
@@ -390,7 +386,7 @@ void OpDispatchBuilder::AVX128_MOVQ(OpcodeArgs) {
|
||||
Src.High = ZeroVector;
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Src);
|
||||
} else {
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src.Low, OpSize::i64Bit, OpSize::i64Bit);
|
||||
StoreResultFPR_WithOpSize(Op, Op->Dest, Src.Low, OpSize::i64Bit, OpSize::i64Bit);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -399,7 +395,7 @@ void OpDispatchBuilder::AVX128_VMOVLP(OpcodeArgs) {
|
||||
|
||||
if (!Op->Dest.IsGPR()) {
|
||||
///< VMOVLPS/PD mem64, xmm1
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src1.Low, OpSize::i64Bit, OpSize::i64Bit);
|
||||
StoreResultFPR_WithOpSize(Op, Op->Dest, Src1.Low, OpSize::i64Bit, OpSize::i64Bit);
|
||||
} else if (!Op->Src[1].IsGPR()) {
|
||||
///< VMOVLPS/PD xmm1, xmm2, mem64
|
||||
// Bits[63:0] come from Src2[63:0]
|
||||
@@ -463,7 +459,7 @@ void OpDispatchBuilder::AVX128_VMOVDDUP(OpcodeArgs) {
|
||||
// 128-bit operation only loads 8-bytes.
|
||||
// 256-bit operation loads a full 32-bytes.
|
||||
if (Is128Bit) {
|
||||
Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::i64Bit, Op->Flags);
|
||||
Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::i64Bit, Op->Flags);
|
||||
} else {
|
||||
Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, true);
|
||||
}
|
||||
@@ -558,18 +554,18 @@ void OpDispatchBuilder::AVX128_InsertCVTGPR_To_FPR(OpcodeArgs, IR::OpSize DstEle
|
||||
|
||||
if (Op->Src[1].IsGPR()) {
|
||||
// If the source is a GPR then convert directly from the GPR.
|
||||
auto Src2 = LoadSource_WithOpSize(GPRClass, Op, Op->Src[1], GetGPROpSize(), Op->Flags);
|
||||
auto Src2 = LoadSourceGPR_WithOpSize(Op, Op->Src[1], GetGPROpSize(), Op->Flags);
|
||||
Result.Low = _VSToFGPRInsert(OpSize::i128Bit, DstElementSize, SrcSize, Src1.Low, Src2, false);
|
||||
} else if (SrcSize != DstElementSize) {
|
||||
// If the source is from memory but the Source size and destination size aren't the same,
|
||||
// then it is more optimal to load in to a GPR and convert between GPR->FPR.
|
||||
// ARM GPR->FPR conversion supports different size source and destinations while FPR->FPR doesn't.
|
||||
auto Src2 = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags);
|
||||
auto Src2 = LoadSourceGPR(Op, Op->Src[1], Op->Flags);
|
||||
Result.Low = _VSToFGPRInsert(DstSize, DstElementSize, SrcSize, Src1.Low, Src2, false);
|
||||
} else {
|
||||
// In the case of cvtsi2s{s,d} where the source and destination are the same size,
|
||||
// then it is more optimal to load in to the FPR register directly and convert there.
|
||||
auto Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
auto Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
// Always signed
|
||||
Result.Low = _VSToFVectorInsert(DstSize, DstElementSize, DstElementSize, Src1.Low, Src2, false, false);
|
||||
}
|
||||
@@ -589,11 +585,11 @@ void OpDispatchBuilder::AVX128_CVTFPR_To_GPR(OpcodeArgs, IR::OpSize SrcElementSi
|
||||
if (Op->Src[0].IsGPR()) {
|
||||
Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false);
|
||||
} else {
|
||||
Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcElementSize, Op->Flags);
|
||||
Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcElementSize, Op->Flags);
|
||||
}
|
||||
|
||||
Ref Result = CVTFPR_To_GPRImpl(Op, Src.Low, SrcElementSize, HostRoundingMode);
|
||||
StoreResult(GPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultGPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VANDN(OpcodeArgs) {
|
||||
@@ -636,7 +632,7 @@ void OpDispatchBuilder::AVX128_UCOMISx(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
if (Op->Src[0].IsGPR()) {
|
||||
Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false);
|
||||
} else {
|
||||
Src2.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
Src2.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
}
|
||||
|
||||
Comiss(ElementSize, Src1.Low, Src2.Low);
|
||||
@@ -653,7 +649,7 @@ void OpDispatchBuilder::AVX128_VectorScalarInsertALU(OpcodeArgs, FEXCore::IR::IR
|
||||
if (Op->Src[1].IsGPR()) {
|
||||
Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false);
|
||||
} else {
|
||||
Src2.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags);
|
||||
Src2.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags);
|
||||
}
|
||||
|
||||
// If OpSize == ElementSize then it only does the lower scalar op
|
||||
@@ -690,7 +686,7 @@ void OpDispatchBuilder::AVX128_InsertScalarFCMP(OpcodeArgs, IR::OpSize ElementSi
|
||||
if (Op->Src[1].IsGPR()) {
|
||||
Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false);
|
||||
} else {
|
||||
Src2.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags);
|
||||
Src2.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags);
|
||||
}
|
||||
|
||||
const uint8_t CompType = Op->Src[2].Literal();
|
||||
@@ -708,12 +704,12 @@ void OpDispatchBuilder::AVX128_MOVBetweenGPR_FPR(OpcodeArgs) {
|
||||
RefPair Result {};
|
||||
if (Op->Src[0].IsGPR()) {
|
||||
// Loading from GPR and moving to Vector.
|
||||
Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], GetGPROpSize(), Op->Flags);
|
||||
Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[0], GetGPROpSize(), Op->Flags);
|
||||
// zext to 128bit
|
||||
Result.Low = _VCastFromGPR(OpSize::i128Bit, OpSizeFromSrc(Op), Src);
|
||||
} else {
|
||||
// Loading from Memory as a scalar. Zero extend
|
||||
Result.Low = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Result.Low = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
|
||||
Result.High = LoadZeroVector(OpSize::i128Bit);
|
||||
@@ -726,11 +722,11 @@ void OpDispatchBuilder::AVX128_MOVBetweenGPR_FPR(OpcodeArgs) {
|
||||
auto ElementSize = OpSizeFromDst(Op);
|
||||
// Extract element from GPR. Zero extending in the process.
|
||||
Src.Low = _VExtractToGPR(OpSizeFromSrc(Op), ElementSize, Src.Low, 0);
|
||||
StoreResult(GPRClass, Op, Op->Dest, Src.Low, OpSize::iInvalid);
|
||||
StoreResultGPR(Op, Op->Dest, Src.Low);
|
||||
} else {
|
||||
// Storing first element to memory.
|
||||
Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
_StoreMem(FPRClass, OpSizeFromDst(Op), Dest, Src.Low, OpSize::i8Bit);
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
_StoreMemFPR(OpSizeFromDst(Op), Dest, Src.Low, OpSize::i8Bit);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -758,7 +754,7 @@ void OpDispatchBuilder::AVX128_PExtr(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
// Extract already zero extends the result.
|
||||
Ref Result = _VExtractToGPR(OpSize::i128Bit, OverridenElementSize, Src.Low, Index);
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Result, GPRSize, OpSize::iInvalid);
|
||||
StoreResultGPR_WithOpSize(Op, Op->Dest, Result, GPRSize);
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -779,7 +775,7 @@ void OpDispatchBuilder::AVX128_ExtendVectorElements(OpcodeArgs, IR::OpSize Eleme
|
||||
const auto SrcSize = OpSizeFromSrc(Op);
|
||||
const auto LoadSize = Is256Bit ? IR::SizeToOpSize(IR::OpSizeToSize(SrcSize) * 2) : SrcSize;
|
||||
|
||||
return LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], LoadSize, Op->Flags);
|
||||
return LoadSourceFPR_WithOpSize(Op, Op->Src[0], LoadSize, Op->Flags);
|
||||
}
|
||||
};
|
||||
|
||||
@@ -868,7 +864,7 @@ void OpDispatchBuilder::AVX128_MOVMSK(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
auto GPRHigh = Mask8Byte(Src.High);
|
||||
GPR = _Orlshl(OpSize::i64Bit, GPRLow, GPRHigh, 2);
|
||||
}
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, GPR, GetGPROpSize(), OpSize::iInvalid);
|
||||
StoreResultGPR_WithOpSize(Op, Op->Dest, GPR, GetGPROpSize());
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_MOVMSKB(OpcodeArgs) {
|
||||
@@ -897,7 +893,7 @@ void OpDispatchBuilder::AVX128_MOVMSKB(OpcodeArgs) {
|
||||
Result = _Orlshl(OpSize::i64Bit, Result, ResultHigh, 16);
|
||||
}
|
||||
|
||||
StoreResult(GPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultGPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_PINSRImpl(OpcodeArgs, IR::OpSize ElementSize, const X86Tables::DecodedOperand& Src1Op,
|
||||
@@ -910,7 +906,7 @@ void OpDispatchBuilder::AVX128_PINSRImpl(OpcodeArgs, IR::OpSize ElementSize, con
|
||||
|
||||
if (Src2Op.IsGPR()) {
|
||||
// If the source is a GPR then convert directly from the GPR.
|
||||
auto Src2 = LoadSource_WithOpSize(GPRClass, Op, Src2Op, GetGPROpSize(), Op->Flags);
|
||||
auto Src2 = LoadSourceGPR_WithOpSize(Op, Src2Op, GetGPROpSize(), Op->Flags);
|
||||
Result.Low = _VInsGPR(OpSize::i128Bit, ElementSize, Index, Src1.Low, Src2);
|
||||
} else {
|
||||
// If loading from memory then we only load the element size
|
||||
@@ -1047,7 +1043,7 @@ void OpDispatchBuilder::AVX128_InsertScalar_CVT_Float_To_Float(OpcodeArgs, IR::O
|
||||
// Then zero extends the top 128-bit.
|
||||
const auto SrcSize = Op->Src[1].IsGPR() ? OpSize::i128Bit : SrcElementSize;
|
||||
auto Src1 = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false);
|
||||
Ref Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags, {.AllowUpperGarbage = true});
|
||||
Ref Src2 = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
Ref Result = _VFToFScalarInsert(OpSize::i128Bit, DstElementSize, SrcElementSize, Src1.Low, Src2, false);
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, AVX128_Zext(Result));
|
||||
@@ -1076,7 +1072,7 @@ void OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Float(OpcodeArgs, IR::OpSize
|
||||
} else {
|
||||
// Handle 64-bit memory source.
|
||||
// In the case of cvtps2pd xmm, m64.
|
||||
Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], LoadSize, Op->Flags);
|
||||
Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], LoadSize, Op->Flags);
|
||||
}
|
||||
|
||||
RefPair Result {};
|
||||
@@ -1154,7 +1150,7 @@ void OpDispatchBuilder::AVX128_Vector_CVT_Int_To_Float(OpcodeArgs, IR::OpSize Sr
|
||||
// unnecessarily zero extend the vector. Otherwise, if
|
||||
// memory, then we want to load the element size exactly.
|
||||
const auto LoadSize = IR::SizeToOpSize(8 * (IR::OpSizeToSize(Size) / 16));
|
||||
return RefPair {.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], LoadSize, Op->Flags)};
|
||||
return RefPair {.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], LoadSize, Op->Flags)};
|
||||
} else {
|
||||
return AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, !Is128Bit);
|
||||
}
|
||||
@@ -1304,7 +1300,7 @@ void OpDispatchBuilder::AVX128_InsertScalarRound(OpcodeArgs, IR::OpSize ElementS
|
||||
if (Op->Src[1].IsGPR()) {
|
||||
Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false);
|
||||
} else {
|
||||
Src2.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags);
|
||||
Src2.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags);
|
||||
}
|
||||
|
||||
// If OpSize == ElementSize then it only does the lower scalar op
|
||||
@@ -1573,20 +1569,20 @@ void OpDispatchBuilder::AVX128_VMASKMOVImpl(OpcodeArgs, IR::OpSize ElementSize,
|
||||
auto Address = MakeAddress(Op->Dest);
|
||||
|
||||
auto Data = AVX128_LoadSource_WithOpSize(Op, DataOp, Op->Flags, !Is128Bit);
|
||||
_VStoreVectorMasked(OpSize::i128Bit, ElementSize, Mask.Low, Data.Low, Address, Invalid(), MEM_OFFSET_SXTX, 1);
|
||||
_VStoreVectorMasked(OpSize::i128Bit, ElementSize, Mask.Low, Data.Low, Address, Invalid(), MemOffsetType::SXTX, 1);
|
||||
if (!Is128Bit) {
|
||||
_VStoreVectorMasked(OpSize::i128Bit, ElementSize, Mask.High, Data.High, Address, _InlineConstant(16), MEM_OFFSET_SXTX, 1);
|
||||
_VStoreVectorMasked(OpSize::i128Bit, ElementSize, Mask.High, Data.High, Address, _InlineConstant(16), MemOffsetType::SXTX, 1);
|
||||
}
|
||||
} else {
|
||||
auto Address = MakeAddress(DataOp);
|
||||
|
||||
RefPair Result {};
|
||||
Result.Low = _VLoadVectorMasked(OpSize::i128Bit, ElementSize, Mask.Low, Address, Invalid(), MEM_OFFSET_SXTX, 1);
|
||||
Result.Low = _VLoadVectorMasked(OpSize::i128Bit, ElementSize, Mask.Low, Address, Invalid(), MemOffsetType::SXTX, 1);
|
||||
|
||||
if (Is128Bit) {
|
||||
Result.High = LoadZeroVector(OpSize::i128Bit);
|
||||
} else {
|
||||
Result.High = _VLoadVectorMasked(OpSize::i128Bit, ElementSize, Mask.High, Address, _InlineConstant(16), MEM_OFFSET_SXTX, 1);
|
||||
Result.High = _VLoadVectorMasked(OpSize::i128Bit, ElementSize, Mask.High, Address, _InlineConstant(16), MemOffsetType::SXTX, 1);
|
||||
}
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
}
|
||||
@@ -1616,11 +1612,11 @@ void OpDispatchBuilder::AVX128_MASKMOV(OpcodeArgs) {
|
||||
// RDI source (DS prefix by default)
|
||||
auto MemDest = MakeSegmentAddress(X86State::REG_RDI, Op->Flags, X86Tables::DecodeFlags::FLAG_DS_PREFIX);
|
||||
|
||||
Ref XMMReg = _LoadMem(FPRClass, Size, MemDest, OpSize::i8Bit);
|
||||
Ref XMMReg = _LoadMemFPR(Size, MemDest, OpSize::i8Bit);
|
||||
|
||||
// If the Mask element high bit is set then overwrite the element with the source, else keep the memory variant
|
||||
XMMReg = _VBSL(Size, MaskSrc.Low, VectorSrc.Low, XMMReg);
|
||||
_StoreMem(FPRClass, Size, MemDest, XMMReg, OpSize::i8Bit);
|
||||
_StoreMemFPR(Size, MemDest, XMMReg, OpSize::i8Bit);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VectorVariableBlend(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
@@ -1660,7 +1656,7 @@ void OpDispatchBuilder::AVX128_SaveAVXState(Ref MemBase) {
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; i += 2) {
|
||||
RefPair Pair = LoadContextPair(OpSize::i128Bit, AVXHigh0Index + i);
|
||||
_StoreMemPair(FPRClass, OpSize::i128Bit, Pair.Low, Pair.High, MemBase, i * 16 + 576);
|
||||
_StoreMemPairFPR(OpSize::i128Bit, Pair.Low, Pair.High, MemBase, i * 16 + 576);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1668,7 +1664,7 @@ void OpDispatchBuilder::AVX128_RestoreAVXState(Ref MemBase) {
|
||||
const auto NumRegs = Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; i += 2) {
|
||||
auto YMMHRegs = LoadMemPair(FPRClass, OpSize::i128Bit, MemBase, i * 16 + 576);
|
||||
auto YMMHRegs = LoadMemPairFPR(OpSize::i128Bit, MemBase, i * 16 + 576);
|
||||
|
||||
AVX128_StoreXMMRegister(i, YMMHRegs.Low, true);
|
||||
AVX128_StoreXMMRegister(i + 1, YMMHRegs.High, true);
|
||||
@@ -1960,7 +1956,7 @@ void OpDispatchBuilder::AVX128_VFMAImpl(OpcodeArgs, IROps IROp, uint8_t Src1Idx,
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VFMAScalarImpl(OpcodeArgs, IROps IROp, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx) {
|
||||
const auto SrcSize = OpSizeFromSrc(Op);
|
||||
const OpSize ElementSize = Op->Flags & X86Tables::DecodeFlags::FLAG_OPTION_AVX_W ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
|
||||
auto Dest = AVX128_LoadSource_WithOpSize(Op, Op->Dest, Op->Flags, false).Low;
|
||||
auto Src1 = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false).Low;
|
||||
@@ -1968,13 +1964,13 @@ void OpDispatchBuilder::AVX128_VFMAScalarImpl(OpcodeArgs, IROps IROp, uint8_t Sr
|
||||
if (Op->Src[1].IsGPR()) {
|
||||
Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false).Low;
|
||||
} else {
|
||||
Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags);
|
||||
Src2 = LoadSourceFPR_WithOpSize(Op, Op->Src[1], ElementSize, Op->Flags);
|
||||
}
|
||||
|
||||
Ref Sources[3] = {Dest, Src1, Src2};
|
||||
|
||||
DeriveOp(Result_Low, IROp,
|
||||
_VFMLAScalarInsert(OpSize::i128Bit, SrcSize, Dest, Sources[Src1Idx - 1], Sources[Src2Idx - 1], Sources[AddendIdx - 1]));
|
||||
_VFMLAScalarInsert(OpSize::i128Bit, ElementSize, Dest, Sources[Src1Idx - 1], Sources[Src2Idx - 1], Sources[AddendIdx - 1]));
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, AVX128_Zext(Result_Low));
|
||||
}
|
||||
|
||||
@@ -2016,8 +2012,8 @@ void OpDispatchBuilder::AVX128_VFMAddSubImpl(OpcodeArgs, bool AddSub, uint8_t Sr
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
}
|
||||
|
||||
OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherImpl(OpSize Size, OpSize ElementLoadSize, OpSize AddrElementSize, RefPair Dest,
|
||||
RefPair Mask, RefVSIB VSIB) {
|
||||
OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherImpl(OpcodeArgs, OpSize Size, OpSize ElementLoadSize, OpSize AddrElementSize,
|
||||
RefPair Dest, RefPair Mask, RefVSIB VSIB) {
|
||||
LOGMAN_THROW_A_FMT(AddrElementSize == OpSize::i32Bit || AddrElementSize == OpSize::i64Bit, "Unknown address element size");
|
||||
const auto Is128Bit = Size == OpSize::i128Bit;
|
||||
|
||||
@@ -2061,10 +2057,13 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherImpl(OpSize Size, O
|
||||
}
|
||||
}
|
||||
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
auto AddrSize = (Op->Flags & X86Tables::DecodeFlags::FLAG_ADDRESS_SIZE) != 0 ? (GPRSize >> 1) : GPRSize;
|
||||
|
||||
RefPair Result {};
|
||||
///< Calculate the low-half.
|
||||
Result.Low = _VLoadVectorGatherMasked(OpSize::i128Bit, ElementLoadSize, Dest.Low, Mask.Low, BaseAddr, VSIB.Low, VSIB.High,
|
||||
AddrElementSize, VSIB.Scale, 0, 0);
|
||||
AddrElementSize, VSIB.Scale, 0, 0, AddrSize);
|
||||
|
||||
if (Is128Bit) {
|
||||
Result.High = LoadZeroVector(OpSize::i128Bit);
|
||||
@@ -2101,7 +2100,7 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherImpl(OpSize Size, O
|
||||
|
||||
///< Calculate the high-half.
|
||||
auto ResultHigh = _VLoadVectorGatherMasked(OpSize::i128Bit, ElementLoadSize, DestReg, MaskReg, BaseAddr, AddrAddressing.Low,
|
||||
AddrAddressing.High, AddrElementSize, VSIB.Scale, DataElementOffset, IndexElementOffset);
|
||||
AddrAddressing.High, AddrElementSize, VSIB.Scale, DataElementOffset, IndexElementOffset, AddrSize);
|
||||
|
||||
if (AddrElementSize == OpSize::i64Bit && ElementLoadSize == OpSize::i32Bit) {
|
||||
// If we only fetched 128-bits worth of data then the upper-result is all zero.
|
||||
@@ -2114,7 +2113,7 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherImpl(OpSize Size, O
|
||||
return Result;
|
||||
}
|
||||
|
||||
OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherQPSImpl(Ref Dest, Ref Mask, RefVSIB VSIB) {
|
||||
OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherQPSImpl(OpcodeArgs, Ref Dest, Ref Mask, RefVSIB VSIB) {
|
||||
|
||||
///< BaseAddr doesn't need to exist, calculate that here.
|
||||
Ref BaseAddr = VSIB.BaseAddr;
|
||||
@@ -2142,8 +2141,11 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherQPSImpl(Ref Dest, R
|
||||
|
||||
RefPair Result {};
|
||||
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
auto AddrSize = (Op->Flags & X86Tables::DecodeFlags::FLAG_ADDRESS_SIZE) != 0 ? (GPRSize >> 1) : GPRSize;
|
||||
|
||||
///< Calculate the low-half.
|
||||
Result.Low = _VLoadVectorGatherMaskedQPS(OpSize::i128Bit, OpSize::i32Bit, Dest, Mask, BaseAddr, VSIB.Low, VSIB.High, VSIB.Scale);
|
||||
Result.Low = _VLoadVectorGatherMaskedQPS(OpSize::i128Bit, OpSize::i32Bit, Dest, Mask, BaseAddr, VSIB.Low, VSIB.High, VSIB.Scale, AddrSize);
|
||||
Result.High = LoadZeroVector(OpSize::i128Bit);
|
||||
if (VSIB.High == Invalid()) {
|
||||
// Special case for only loading two floats.
|
||||
@@ -2202,15 +2204,15 @@ void OpDispatchBuilder::AVX128_VPGATHER(OpcodeArgs, OpSize AddrElementSize) {
|
||||
}
|
||||
|
||||
///< AddressElementSize is now OpSize::i64Bit
|
||||
Result = AVX128_VPGatherQPSImpl(Dest.Low, Mask.Low, VSIBLow);
|
||||
Result = AVX128_VPGatherQPSImpl(Op, Dest.Low, Mask.Low, VSIBLow);
|
||||
if (NeedsHighAddrBytes) {
|
||||
auto Res = AVX128_VPGatherQPSImpl(Dest.High, Mask.High, VSIBHigh);
|
||||
auto Res = AVX128_VPGatherQPSImpl(Op, Dest.High, Mask.High, VSIBHigh);
|
||||
Result.High = Res.Low;
|
||||
}
|
||||
} else if (AddrElementSize == OpSize::i64Bit && ElementLoadSize == OpSize::i32Bit) {
|
||||
Result = AVX128_VPGatherQPSImpl(Dest.Low, Mask.Low, VSIB);
|
||||
Result = AVX128_VPGatherQPSImpl(Op, Dest.Low, Mask.Low, VSIB);
|
||||
} else {
|
||||
Result = AVX128_VPGatherImpl(Size, ElementLoadSize, AddrElementSize, Dest, Mask, VSIB);
|
||||
Result = AVX128_VPGatherImpl(Op, Size, ElementLoadSize, AddrElementSize, Dest, Mask, VSIB);
|
||||
}
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
|
||||
@@ -2234,7 +2236,7 @@ void OpDispatchBuilder::AVX128_VCVTPH2PS(OpcodeArgs) {
|
||||
// In the event that a memory operand is used as the source operand,
|
||||
// the access width will always be half the size of the destination vector width
|
||||
// (i.e. 128-bit vector -> 64-bit mem, 256-bit vector -> 128-bit mem)
|
||||
Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
}
|
||||
|
||||
RefPair Result {};
|
||||
@@ -2289,7 +2291,7 @@ void OpDispatchBuilder::AVX128_VCVTPS2PH(OpcodeArgs) {
|
||||
}
|
||||
|
||||
if (!Op->Dest.IsGPR()) {
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result.Low, StoreSize, OpSize::iInvalid);
|
||||
StoreResultFPR_WithOpSize(Op, Op->Dest, Result.Low, StoreSize);
|
||||
} else {
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
}
|
||||
|
||||
@@ -23,8 +23,8 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
// ARMv8 SHA1 extension provides a `SHA1H` instruction which does a fixed rotate by 30.
|
||||
// This only operates on element 0 rather than element 3. We don't have the luxury of rewriting the x86 SHA algorithm to take advantage of this.
|
||||
@@ -36,7 +36,7 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
|
||||
auto Tmp = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src, RotatedNode);
|
||||
auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 3, 3, Src, Tmp);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
|
||||
@@ -44,15 +44,15 @@ void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
Ref NewVec = _VExtr(OpSize::i128Bit, OpSize::i64Bit, Dest, Src, 1);
|
||||
|
||||
// [W0, W1, W2, W3] ^ [W2, W3, W4, W5]
|
||||
Ref Result = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, NewVec);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
@@ -60,8 +60,8 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
// ARM SHA1 mostly matches x86 semantics, except the input and outputs are both flipped from elements 0,1,2,3 to 3,2,1,0.
|
||||
auto Src1 = SHADataShuffle(Dest);
|
||||
@@ -70,7 +70,7 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
// The result is swizzled differently than expected
|
||||
auto Result = SHADataShuffle(_VSha1SU1(Src1, Src2));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
@@ -79,8 +79,8 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
return;
|
||||
}
|
||||
const uint64_t Imm8 = Op->Src[1].Literal() & 0b11;
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
Ref Result {};
|
||||
Ref ConstantVector {};
|
||||
@@ -112,7 +112,7 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
case 3: Result = SHADataShuffle(_VSha1P(Src1, ZeroRegister, Src2)); break;
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
|
||||
@@ -120,12 +120,12 @@ void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto Result = _VSha256U0(Dest, Src);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
@@ -133,8 +133,8 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto Src1 = _VExtr(OpSize::i128Bit, OpSize::i32Bit, Dest, Dest, 3);
|
||||
auto DupDst = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
@@ -142,7 +142,7 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
|
||||
auto Result = _VSha256U1(Src1, Src2);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
@@ -150,8 +150,8 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
// Hardcoded to XMM0
|
||||
auto XMM0 = LoadXMMRegister(0);
|
||||
|
||||
@@ -177,7 +177,7 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
auto B = _VSha256H2(EFGH, ABCD, Key);
|
||||
auto Result = shuffle_abcd(A, B);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
|
||||
@@ -185,9 +185,9 @@ void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESImc(Src);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
|
||||
@@ -195,10 +195,10 @@ void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESEnc(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
|
||||
@@ -208,11 +208,11 @@ void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESENC.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENC unimplemented");
|
||||
|
||||
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
Ref Result = _VAESEnc(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
|
||||
@@ -220,10 +220,10 @@ void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESEncLast(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
|
||||
@@ -233,11 +233,11 @@ void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESENCLAST.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENCLAST unimplemented");
|
||||
|
||||
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
Ref Result = _VAESEncLast(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
|
||||
@@ -245,10 +245,10 @@ void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESDec(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
|
||||
@@ -258,11 +258,11 @@ void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESDEC.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDEC unimplemented");
|
||||
|
||||
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
Ref Result = _VAESDec(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
|
||||
@@ -270,10 +270,10 @@ void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESDecLast(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
|
||||
@@ -283,15 +283,15 @@ void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESDECLAST.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDECLAST unimplemented");
|
||||
|
||||
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
Ref Result = _VAESDecLast(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) {
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
const uint64_t RCON = Op->Src[1].Literal();
|
||||
|
||||
auto KeyGenSwizzle = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE);
|
||||
@@ -305,7 +305,7 @@ void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) {
|
||||
}
|
||||
|
||||
Ref Result = AESKeyGenAssistImpl(Op);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
|
||||
@@ -313,12 +313,12 @@ void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[1].Literal());
|
||||
|
||||
auto Res = _PCLMUL(OpSize::i128Bit, Dest, Src, Selector & 0b1'0001);
|
||||
StoreResult(FPRClass, Op, Res, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Res);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) {
|
||||
@@ -328,12 +328,12 @@ void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) {
|
||||
}
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
|
||||
Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[2].Literal());
|
||||
|
||||
Ref Res = _PCLMUL(DstSize, Src1, Src2, Selector & 0b1'0001);
|
||||
StoreResult(FPRClass, Op, Res, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Res);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -263,7 +263,7 @@ void OpDispatchBuilder::CalculateDeferredFlags() {
|
||||
Ref OpDispatchBuilder::IncrementByCarry(OpSize OpSize, Ref Src) {
|
||||
// If CF not inverted, we use .cc since the increment happens when the
|
||||
// condition is false. If CF inverted, invert to use .cs. A bit mindbendy.
|
||||
return _NZCVSelectIncrement(OpSize, {CFInverted ? COND_UGE : COND_ULT}, Src, Src);
|
||||
return _NZCVSelectIncrement(OpSize, CFInverted ? CondClass::UGE : CondClass::ULT, Src, Src);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::CalculateFlags_ADC(IR::OpSize SrcSize, Ref Src1, Ref Src2) {
|
||||
@@ -290,7 +290,7 @@ Ref OpDispatchBuilder::CalculateFlags_ADC(IR::OpSize SrcSize, Ref Src1, Ref Src2
|
||||
Res = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Res);
|
||||
|
||||
// TODO: We can fold that second Bfe in (cmp uxth).
|
||||
auto SelectCFInv = Select01(OpSize, CondClassType {COND_UGE}, Res, Src2PlusCF);
|
||||
auto SelectCFInv = Select01(OpSize, CondClass::UGE, Res, Src2PlusCF);
|
||||
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
SetCFInverted(SelectCFInv);
|
||||
@@ -324,7 +324,7 @@ Ref OpDispatchBuilder::CalculateFlags_SBB(IR::OpSize SrcSize, Ref Src1, Ref Src2
|
||||
Res = Sub(OpSize, Src1, Src2PlusCF);
|
||||
Res = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Res);
|
||||
|
||||
auto SelectCFInv = Select01(OpSize, CondClassType {COND_UGE}, Src1, Src2PlusCF);
|
||||
auto SelectCFInv = Select01(OpSize, CondClass::UGE, Src1, Src2PlusCF);
|
||||
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
SetCFInverted(SelectCFInv);
|
||||
@@ -406,7 +406,7 @@ void OpDispatchBuilder::CalculateFlags_MUL(IR::OpSize SrcSize, Ref Res, Ref High
|
||||
// If High = SignBit, then sets to nZCv. Else sets to nzcV. Since SF/ZF
|
||||
// undefined, this does what we need after inverting carry.
|
||||
auto Zero = _InlineConstant(0);
|
||||
_CondSubNZCV(OpSize::i64Bit, Zero, Zero, CondClassType {COND_EQ}, 0x1 /* nzcV */);
|
||||
_CondSubNZCV(OpSize::i64Bit, Zero, Zero, CondClass::EQ, 0x1 /* nzcV */);
|
||||
CFInverted = true;
|
||||
}
|
||||
|
||||
@@ -423,7 +423,7 @@ void OpDispatchBuilder::CalculateFlags_UMUL(Ref High) {
|
||||
|
||||
// If High = 0, then sets to nZCv. Else sets to nzcV. Since SF/ZF undefined,
|
||||
// this does what we need.
|
||||
_CondSubNZCV(Size, Zero, Zero, CondClassType {COND_EQ}, 0x1 /* nzcV */);
|
||||
_CondSubNZCV(Size, Zero, Zero, CondClass::EQ, 0x1 /* nzcV */);
|
||||
CFInverted = true;
|
||||
}
|
||||
|
||||
|
||||
@@ -151,6 +151,9 @@ constexpr DispatchTableEntry OpDispatch_SecondaryGroupTables[] = {
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F2, 3), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 3>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_16, PF_F2, 4), 4, &OpDispatchBuilder::NOPOp},
|
||||
|
||||
// GROUP 17
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_17, PF_66, 0), 1, &OpDispatchBuilder::Extrq_imm},
|
||||
|
||||
// GROUP P
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_NONE, 0), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, false, false, 1>},
|
||||
{OPD(FEXCore::X86Tables::TYPE_GROUP_P, PF_NONE, 1), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::Prefetch, true, false, 1>},
|
||||
|
||||
@@ -145,7 +145,7 @@ constexpr DispatchTableEntry OpDispatch_TwoByteOpTable[] = {
|
||||
|
||||
#ifndef _WIN32
|
||||
// FEX reserved instructions
|
||||
{0x37, 1, &OpDispatchBuilder::CallbackReturnOp},
|
||||
{0x3E, 1, &OpDispatchBuilder::CallbackReturnOp},
|
||||
{0x3F, 1, &OpDispatchBuilder::ThunkOp},
|
||||
#endif
|
||||
};
|
||||
@@ -198,6 +198,8 @@ constexpr DispatchTableEntry OpDispatch_SecondaryRepNEModTables[] = {
|
||||
{0x5E, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x5F, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, OpSize::i64Bit>},
|
||||
{0x70, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSHUFWOp, true>},
|
||||
{0x78, 1, &OpDispatchBuilder::Insertq_imm},
|
||||
{0x79, 1, &OpDispatchBuilder::Insertq},
|
||||
{0x7C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, OpSize::i32Bit>},
|
||||
{0x7D, 1, &OpDispatchBuilder::HSUBP<OpSize::i32Bit>},
|
||||
{0xD0, 1, &OpDispatchBuilder::ADDSUBPOp<OpSize::i32Bit>},
|
||||
@@ -256,6 +258,7 @@ constexpr DispatchTableEntry OpDispatch_SecondaryOpSizeModTables[] = {
|
||||
{0x75, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i16Bit>},
|
||||
{0x76, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i32Bit>},
|
||||
{0x78, 1, nullptr}, // GROUP 17
|
||||
{0x79, 1, &OpDispatchBuilder::Extrq},
|
||||
{0x7C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, OpSize::i64Bit>},
|
||||
{0x7D, 1, &OpDispatchBuilder::HSUBP<OpSize::i64Bit>},
|
||||
{0x7E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVBetweenGPR_FPR, OpDispatchBuilder::VectorOpType::SSE>},
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -17,7 +17,6 @@ $end_info$
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/FPState.h>
|
||||
|
||||
#include <cmath>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
|
||||
@@ -28,7 +27,7 @@ class OrderedNode;
|
||||
Ref OpDispatchBuilder::GetX87Top() {
|
||||
// Yes, we are storing 3 bits in a single flag register.
|
||||
// Deal with it
|
||||
return _LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
return _LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SetX87FTW(Ref FTW) {
|
||||
@@ -52,24 +51,22 @@ void OpDispatchBuilder::SetX87FTW(Ref FTW) {
|
||||
FTW = _Orlshr(OpSize::i32Bit, FTW, FTW, 4);
|
||||
|
||||
// ...and that's it. StoreContext implicitly does the final masking.
|
||||
_StoreContext(OpSize::i8Bit, GPRClass, FTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
_StoreContextGPR(OpSize::i8Bit, FTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SetX87Top(Ref Value) {
|
||||
_StoreContext(OpSize::i8Bit, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
_StoreContextGPR(OpSize::i8Bit, Value, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
}
|
||||
|
||||
// Float LoaD operation with memory operand
|
||||
void OpDispatchBuilder::FLD(OpcodeArgs, IR::OpSize Width) {
|
||||
const auto ReadWidth = (Width == OpSize::f80Bit) ? OpSize::i128Bit : Width;
|
||||
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], Width, Op->Flags);
|
||||
Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], Width, Op->Flags);
|
||||
Ref ConvertedData = Data;
|
||||
// Convert to 80bit float
|
||||
if (Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
ConvertedData = _F80CVTTo(Data, ReadWidth);
|
||||
ConvertedData = _F80CVTTo(Data, Width);
|
||||
}
|
||||
_PushStack(ConvertedData, Data, ReadWidth, true);
|
||||
_PushStack(ConvertedData, Data, Width);
|
||||
}
|
||||
|
||||
// Float LoaD operation with memory operand
|
||||
@@ -79,27 +76,27 @@ void OpDispatchBuilder::FLDFromStack(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::FBLD(OpcodeArgs) {
|
||||
// Read from memory
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::f80Bit, Op->Flags);
|
||||
Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::f80Bit, Op->Flags);
|
||||
Ref ConvertedData = _F80BCDLoad(Data);
|
||||
_PushStack(ConvertedData, Data, OpSize::i128Bit, true);
|
||||
_PushStack(ConvertedData, Invalid(), OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FBSTP(OpcodeArgs) {
|
||||
Ref converted = _F80BCDStore(_ReadStackValue(0));
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, converted, OpSize::f80Bit, OpSize::i8Bit);
|
||||
StoreResultFPR_WithOpSize(Op, Op->Dest, converted, OpSize::f80Bit, OpSize::i8Bit);
|
||||
_PopStackDestroy();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FLD_Const(OpcodeArgs, NamedVectorConstant K) {
|
||||
// Update TOP
|
||||
Ref Data = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, K);
|
||||
_PushStack(Data, Data, OpSize::i128Bit, true);
|
||||
_PushStack(Data, Data, OpSize::f80Bit);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
const auto ReadWidth = OpSizeFromSrc(Op);
|
||||
// Read from memory
|
||||
Ref Data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], ReadWidth, Op->Flags);
|
||||
Ref Data = LoadSourceGPR_WithOpSize(Op, Op->Src[0], ReadWidth, Op->Flags);
|
||||
|
||||
// Sign extend to 64bits
|
||||
if (ReadWidth != OpSize::i64Bit) {
|
||||
@@ -112,27 +109,28 @@ void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
// Extract sign and make integer absolute
|
||||
auto zero = Constant(0);
|
||||
_SubNZCV(OpSize::i64Bit, Data, zero);
|
||||
auto sign = _NZCVSelect(OpSize::i64Bit, CondClassType {COND_SLT}, Constant(0x8000), zero);
|
||||
auto absolute = _Neg(OpSize::i64Bit, Data, CondClassType {COND_MI});
|
||||
auto sign = _NZCVSelect(OpSize::i64Bit, CondClass::SLT, Constant(0x8000), zero);
|
||||
auto absolute = _Neg(OpSize::i64Bit, Data, CondClass::MI);
|
||||
|
||||
// left justify the absolute integer
|
||||
auto shift = Sub(OpSize::i64Bit, Constant(63), _FindMSB(IR::OpSize::i64Bit, absolute));
|
||||
auto shifted = _Lshl(OpSize::i64Bit, absolute, shift);
|
||||
|
||||
auto adjusted_exponent = Sub(OpSize::i64Bit, Constant(0x3fff + 63), shift);
|
||||
auto zeroed_exponent = _Select(COND_EQ, absolute, zero, zero, adjusted_exponent);
|
||||
auto zeroed_exponent = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, absolute, zero, zero, adjusted_exponent);
|
||||
auto upper = _Or(OpSize::i64Bit, sign, zeroed_exponent);
|
||||
|
||||
Ref ConvertedData = _VLoadTwoGPRs(shifted, upper);
|
||||
_PushStack(ConvertedData, Data, ReadWidth, false);
|
||||
_PushStack(ConvertedData, Invalid(), OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FST(OpcodeArgs, IR::OpSize Width) {
|
||||
const auto SourceSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
LOGMAN_THROW_A_FMT(Width == OpSize::i32Bit || Width == OpSize::i64Bit || Width == OpSize::f80Bit, "Invalid store width for FST");
|
||||
const auto SourceSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::f80Bit;
|
||||
AddressMode A = DecodeAddress(Op, Op->Dest, MemoryAccessType::DEFAULT, false);
|
||||
|
||||
A = SelectAddressMode(this, A, GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, false, false, Width);
|
||||
_StoreStackMem(SourceSize, Width, A.Base, A.Index, OpSize::iInvalid, A.IndexType, A.IndexScale, /*Float=*/true);
|
||||
_StoreStackMem(SourceSize, Width, A.Base, A.Index, OpSize::iInvalid, A.IndexType, A.IndexScale);
|
||||
|
||||
if (Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) {
|
||||
_PopStackDestroy();
|
||||
@@ -166,12 +164,12 @@ void OpDispatchBuilder::FIST(OpcodeArgs, bool Truncate) {
|
||||
// Check for NaN/Infinity: exponent = 0x7fff
|
||||
SaveNZCV();
|
||||
_TestNZ(OpSize::i64Bit, Exponent, Constant(0x7fff));
|
||||
Ref IsSpecial = _NZCVSelect01({COND_EQ});
|
||||
Ref IsSpecial = _NZCVSelect01(CondClass::EQ);
|
||||
|
||||
// For overflow detection, check if exponent indicates a value >= 2^15
|
||||
// Biased exponent for 2^15 is 0x3fff + 15 = 0x400e
|
||||
SubWithFlags(OpSize::i64Bit, Exponent, 0x400e);
|
||||
Ref IsOverflow = _NZCVSelect01({COND_UGE});
|
||||
Ref IsOverflow = _NZCVSelect01(CondClass::UGE);
|
||||
|
||||
// Set Invalid Operation flag if overflow or special value
|
||||
Ref InvalidFlag = _Or(OpSize::i64Bit, IsSpecial, IsOverflow);
|
||||
@@ -180,7 +178,7 @@ void OpDispatchBuilder::FIST(OpcodeArgs, bool Truncate) {
|
||||
|
||||
Data = _F80CVTInt(Size, Data, Truncate);
|
||||
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Data, Size, OpSize::i8Bit);
|
||||
StoreResultGPR_WithOpSize(Op, Op->Dest, Data, Size, OpSize::i8Bit);
|
||||
|
||||
if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) {
|
||||
_PopStackDestroy();
|
||||
@@ -206,10 +204,10 @@ void OpDispatchBuilder::FADD(OpcodeArgs, IR::OpSize Width, bool Integer, OpDispa
|
||||
// We have one memory argument
|
||||
Ref Arg {};
|
||||
if (Integer) {
|
||||
Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTToInt(Arg, Width);
|
||||
} else {
|
||||
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTTo(Arg, Width);
|
||||
}
|
||||
|
||||
@@ -236,10 +234,10 @@ void OpDispatchBuilder::FMUL(OpcodeArgs, IR::OpSize Width, bool Integer, OpDispa
|
||||
// We have one memory argument
|
||||
Ref arg {};
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTToInt(arg, Width);
|
||||
} else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTTo(arg, Width);
|
||||
}
|
||||
|
||||
@@ -273,10 +271,10 @@ void OpDispatchBuilder::FDIV(OpcodeArgs, IR::OpSize Width, bool Integer, bool Re
|
||||
// We have one memory argument
|
||||
Ref arg {};
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTToInt(arg, Width);
|
||||
} else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTTo(arg, Width);
|
||||
}
|
||||
|
||||
@@ -314,10 +312,10 @@ void OpDispatchBuilder::FSUB(OpcodeArgs, IR::OpSize Width, bool Integer, bool Re
|
||||
// We have one memory argument
|
||||
Ref Arg {};
|
||||
if (Integer) {
|
||||
Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTToInt(Arg, Width);
|
||||
} else {
|
||||
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTTo(Arg, Width);
|
||||
}
|
||||
|
||||
@@ -340,7 +338,7 @@ Ref OpDispatchBuilder::GetX87FTW_Helper() {
|
||||
// bytes, we use the well-known bit twiddling algorithm:
|
||||
//
|
||||
// https://graphics.stanford.edu/~seander/bithacks.html#InterleaveBMN
|
||||
Ref X = _LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
Ref X = _LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
X = _Orlshl(OpSize::i32Bit, X, X, 4);
|
||||
X = _And(OpSize::i32Bit, X, Constant(0x0f0f0f0f));
|
||||
X = _Orlshl(OpSize::i32Bit, X, X, 2);
|
||||
@@ -381,41 +379,41 @@ void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) {
|
||||
_SyncStackToSlow();
|
||||
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
Ref Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
Ref Mem = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
{
|
||||
auto FCW = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreMem(GPRClass, Size, Mem, FCW, Size);
|
||||
auto FCW = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreMemGPR(Size, Mem, FCW, Size);
|
||||
}
|
||||
|
||||
{ _StoreMem(GPRClass, Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MEM_OFFSET_SXTX, 1); }
|
||||
{ _StoreMemGPR(Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1); }
|
||||
|
||||
auto ZeroConst = Constant(0);
|
||||
|
||||
{
|
||||
// FTW
|
||||
_StoreMem(GPRClass, Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMemGPR(Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction Offset
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction CS selector (+ Opcode)
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer offset
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer selector
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -441,20 +439,20 @@ void OpDispatchBuilder::X87LDENV(OpcodeArgs) {
|
||||
_StackForceSlow();
|
||||
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
Ref Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Ref Mem = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, Mem, OpSize::i16Bit);
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
auto NewFCW = _LoadMemGPR(OpSize::i16Bit, Mem, OpSize::i16Bit);
|
||||
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
Ref MemLocation = Add(OpSize::i64Bit, Mem, IR::OpSizeToSize(Size) * 1);
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, MemLocation, Size);
|
||||
auto NewFSW = _LoadMemGPR(Size, MemLocation, Size);
|
||||
ReconstructX87StateFromFSW_Helper(NewFSW);
|
||||
|
||||
{
|
||||
// FTW
|
||||
Ref MemLocation = Add(OpSize::i64Bit, Mem, IR::OpSizeToSize(Size) * 2);
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, MemLocation, Size));
|
||||
SetX87FTW(_LoadMemGPR(Size, MemLocation, Size));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -483,61 +481,61 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
Ref Mem = MakeSegmentAddress(Op, Op->Dest);
|
||||
Ref Top = GetX87Top();
|
||||
{
|
||||
auto FCW = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreMem(GPRClass, Size, Mem, FCW, Size);
|
||||
auto FCW = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreMemGPR(Size, Mem, FCW, Size);
|
||||
}
|
||||
|
||||
{ _StoreMem(GPRClass, Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MEM_OFFSET_SXTX, 1); }
|
||||
{ _StoreMemGPR(Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1); }
|
||||
|
||||
auto ZeroConst = Constant(0);
|
||||
|
||||
{
|
||||
// FTW
|
||||
_StoreMem(GPRClass, Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMemGPR(Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction Offset
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction CS selector (+ Opcode)
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer offset
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer selector
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
auto SevenConst = Constant(7);
|
||||
const auto LoadSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
for (int i = 0; i < 7; ++i) {
|
||||
Ref data = _LoadContextIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
|
||||
Ref data = _LoadContextFPRIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit));
|
||||
if (ReducedPrecisionMode) {
|
||||
data = _F80CVTTo(data, OpSize::i64Bit);
|
||||
}
|
||||
_StoreMem(FPRClass, OpSize::i128Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMemFPR(OpSize::i128Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
Top = _And(OpSize::i32Bit, Add(OpSize::i32Bit, Top, 1), SevenConst);
|
||||
}
|
||||
|
||||
// The final st(7) needs a bit of special handling here
|
||||
Ref data = _LoadContextIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
|
||||
Ref data = _LoadContextFPRIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit));
|
||||
if (ReducedPrecisionMode) {
|
||||
data = _F80CVTTo(data, OpSize::i64Bit);
|
||||
}
|
||||
// ST7 broken in to two parts
|
||||
// Lower 64bits [63:0]
|
||||
// upper 16 bits [79:64]
|
||||
_StoreMem(FPRClass, OpSize::i64Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMemFPR(OpSize::i64Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
auto topBytes = _VDupElement(OpSize::i128Bit, OpSize::i16Bit, data, 4);
|
||||
_StoreMem(FPRClass, OpSize::i16Bit, topBytes, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10) + 8), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
_StoreMemFPR(OpSize::i16Bit, topBytes, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10) + 8), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
|
||||
// reset to default
|
||||
FNINIT(Op);
|
||||
@@ -548,8 +546,8 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
Ref Mem = MakeSegmentAddress(Op, Op->Src[0]);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, Mem, OpSize::i16Bit);
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
auto NewFCW = _LoadMemGPR(OpSize::i16Bit, Mem, OpSize::i16Bit);
|
||||
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
if (ReducedPrecisionMode) {
|
||||
// ignore the rounding precision, we're always 64-bit in F64.
|
||||
// extract rounding mode
|
||||
@@ -561,11 +559,11 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
_SetRoundingMode(roundingMode, false, roundingMode);
|
||||
}
|
||||
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MEM_OFFSET_SXTX, 1);
|
||||
auto NewFSW = _LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1);
|
||||
Ref Top = ReconstructX87StateFromFSW_Helper(NewFSW);
|
||||
{
|
||||
// FTW
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MEM_OFFSET_SXTX, 1));
|
||||
SetX87FTW(_LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1));
|
||||
}
|
||||
|
||||
auto SevenConst = Constant(7);
|
||||
@@ -574,14 +572,14 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
Ref Mask = _VLoadTwoGPRs(low, high);
|
||||
const auto StoreSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
for (int i = 0; i < 7; ++i) {
|
||||
Ref Reg = _LoadMem(FPRClass, OpSize::i128Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
Ref Reg = _LoadMemFPR(OpSize::i128Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
// Mask off the top bits
|
||||
Reg = _VAnd(OpSize::i128Bit, OpSize::i128Bit, Reg, Mask);
|
||||
if (ReducedPrecisionMode) {
|
||||
// Convert to double precision
|
||||
Reg = _F80CVT(OpSize::i64Bit, Reg);
|
||||
}
|
||||
_StoreContextIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
|
||||
_StoreContextFPRIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit));
|
||||
|
||||
Top = _And(OpSize::i32Bit, Add(OpSize::i32Bit, Top, 1), SevenConst);
|
||||
}
|
||||
@@ -590,19 +588,19 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
// ST7 broken in to two parts
|
||||
// Lower 64bits [63:0]
|
||||
// upper 16 bits [79:64]
|
||||
Ref Reg = _LoadMem(FPRClass, OpSize::i64Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
Ref RegHigh = _LoadMem(FPRClass, OpSize::i16Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7) + 8), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
Ref Reg = _LoadMemFPR(OpSize::i64Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
Ref RegHigh = _LoadMemFPR(OpSize::i16Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7) + 8), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
Reg = _VInsElement(OpSize::i128Bit, OpSize::i16Bit, 4, 0, Reg, RegHigh);
|
||||
if (ReducedPrecisionMode) {
|
||||
Reg = _F80CVT(OpSize::i64Bit, Reg); // Convert to double precision
|
||||
}
|
||||
_StoreContextIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
|
||||
_StoreContextFPRIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit));
|
||||
}
|
||||
|
||||
// Load / Store Control Word
|
||||
void OpDispatchBuilder::X87FSTCW(OpcodeArgs) {
|
||||
auto FCW = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
StoreResult(GPRClass, Op, FCW, OpSize::iInvalid);
|
||||
auto FCW = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
StoreResultGPR(Op, FCW);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FLDCW(OpcodeArgs) {
|
||||
@@ -610,8 +608,8 @@ void OpDispatchBuilder::X87FLDCW(OpcodeArgs) {
|
||||
// to switch for now to slow mode whenever these are manually changed.
|
||||
// Remove the next line and try DF_04.asm in fast path.
|
||||
_StackForceSlow();
|
||||
Ref NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
Ref NewFCW = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FXCH(OpcodeArgs) {
|
||||
@@ -647,10 +645,10 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs, IR::OpSize Width, bool Integer, OpDisp
|
||||
if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
// Memory arg
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTToInt(arg, Width);
|
||||
} else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTTo(arg, Width);
|
||||
}
|
||||
} else {
|
||||
@@ -766,7 +764,7 @@ Ref OpDispatchBuilder::ReconstructFSW_Helper(Ref T) {
|
||||
void OpDispatchBuilder::X87FNSTSW(OpcodeArgs) {
|
||||
Ref TopValue = _SyncStackToSlow();
|
||||
Ref StatusWord = ReconstructFSW_Helper(TopValue);
|
||||
StoreResult(GPRClass, Op, StatusWord, OpSize::iInvalid);
|
||||
StoreResultGPR(Op, StatusWord);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FNCLEX(OpcodeArgs) {
|
||||
@@ -785,12 +783,12 @@ void OpDispatchBuilder::FNINIT(OpcodeArgs) {
|
||||
|
||||
// Init FCW to 0x037F
|
||||
auto NewFCW = Constant(0x037F);
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
// Set top to zero
|
||||
SetX87Top(Zero);
|
||||
// Tags all get marked as invalid
|
||||
_StoreContext(OpSize::i8Bit, GPRClass, Zero, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
_StoreContextGPR(OpSize::i8Bit, Zero, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
|
||||
// Reinits the simulated stack
|
||||
_InitStack();
|
||||
@@ -863,7 +861,7 @@ void OpDispatchBuilder::X87FXAM(OpcodeArgs) {
|
||||
auto TopValid = _StackValidTag(0);
|
||||
|
||||
// In the case of top being invalid then C3:C2:C0 is 0b101
|
||||
auto C3 = Select01(OpSize::i32Bit, CondClassType {COND_NEQ}, TopValid, Constant(1));
|
||||
auto C3 = Select01(OpSize::i32Bit, CondClass::NEQ, TopValid, Constant(1));
|
||||
|
||||
auto C2 = TopValid;
|
||||
auto C0 = C3; // Mirror C3 until something other than zero is supported
|
||||
@@ -878,8 +876,8 @@ void OpDispatchBuilder::X87FXTRACT(OpcodeArgs) {
|
||||
_PopStackDestroy();
|
||||
auto Exp = _F80XTRACT_EXP(Top);
|
||||
auto Sig = _F80XTRACT_SIG(Top);
|
||||
_PushStack(Exp, Exp, OpSize::f80Bit, true);
|
||||
_PushStack(Sig, Sig, OpSize::f80Bit, true);
|
||||
_PushStack(Exp, Invalid(), OpSize::iInvalid);
|
||||
_PushStack(Sig, Invalid(), OpSize::iInvalid);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -29,38 +29,37 @@ void OpDispatchBuilder::X87LDENVF64(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
Ref Mem = MakeSegmentAddress(Op, Op->Src[0]);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, Mem, OpSize::i16Bit);
|
||||
auto NewFCW = _LoadMemGPR(OpSize::i16Bit, Mem, OpSize::i16Bit);
|
||||
// ignore the rounding precision, we're always 64-bit in F64.
|
||||
// extract rounding mode
|
||||
Ref roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
|
||||
_SetRoundingMode(roundingMode, false, roundingMode);
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, Mem, Constant(IR::OpSizeToSize(Size)), Size, MEM_OFFSET_SXTX, 1);
|
||||
auto NewFSW = _LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size)), Size, MemOffsetType::SXTX, 1);
|
||||
ReconstructX87StateFromFSW_Helper(NewFSW);
|
||||
|
||||
{
|
||||
// FTW
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MEM_OFFSET_SXTX, 1));
|
||||
SetX87FTW(_LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1));
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FLDCWF64(OpcodeArgs) {
|
||||
_StackForceSlow();
|
||||
|
||||
Ref NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref NewFCW = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
// ignore the rounding precision, we're always 64-bit in F64.
|
||||
// extract rounding mode
|
||||
Ref roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
|
||||
_SetRoundingMode(roundingMode, false, roundingMode);
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
}
|
||||
|
||||
// F64 ops
|
||||
// Float load op with memory operand
|
||||
void OpDispatchBuilder::FLDF64(OpcodeArgs, IR::OpSize Width) {
|
||||
const auto ReadWidth = (Width == OpSize::f80Bit) ? OpSize::i128Bit : Width;
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], Width, Op->Flags);
|
||||
Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], Width, Op->Flags);
|
||||
// Convert to 64bit float
|
||||
Ref ConvertedData = Data;
|
||||
if (Width == OpSize::i32Bit) {
|
||||
@@ -68,39 +67,39 @@ void OpDispatchBuilder::FLDF64(OpcodeArgs, IR::OpSize Width) {
|
||||
} else if (Width == OpSize::f80Bit) {
|
||||
ConvertedData = _F80CVT(OpSize::i64Bit, Data);
|
||||
}
|
||||
_PushStack(ConvertedData, Data, ReadWidth, true);
|
||||
_PushStack(ConvertedData, Data, Width);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FBLDF64(OpcodeArgs) {
|
||||
// Read from memory
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::f80Bit, Op->Flags);
|
||||
Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::f80Bit, Op->Flags);
|
||||
Ref ConvertedData = _F80BCDLoad(Data);
|
||||
ConvertedData = _F80CVT(OpSize::i64Bit, ConvertedData);
|
||||
_PushStack(ConvertedData, Data, OpSize::i64Bit, true);
|
||||
_PushStack(ConvertedData, Invalid(), OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FBSTPF64(OpcodeArgs) {
|
||||
Ref converted = _F80CVTTo(_ReadStackValue(0), OpSize::i64Bit);
|
||||
converted = _F80BCDStore(converted);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, converted, OpSize::f80Bit, OpSize::i8Bit);
|
||||
StoreResultFPR_WithOpSize(Op, Op->Dest, converted, OpSize::f80Bit, OpSize::i8Bit);
|
||||
_PopStackDestroy();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FLDF64_Const(OpcodeArgs, uint64_t Num) {
|
||||
auto Data = _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, Constant(Num));
|
||||
_PushStack(Data, Data, OpSize::i64Bit, true);
|
||||
_PushStack(Data, Data, OpSize::i64Bit);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FILDF64(OpcodeArgs) {
|
||||
const auto ReadWidth = OpSizeFromSrc(Op);
|
||||
|
||||
// Read from memory
|
||||
Ref Data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], ReadWidth, Op->Flags);
|
||||
Ref Data = LoadSourceGPR_WithOpSize(Op, Op->Src[0], ReadWidth, Op->Flags);
|
||||
if (ReadWidth == OpSize::i16Bit) {
|
||||
Data = _Sbfe(OpSize::i64Bit, IR::OpSizeAsBits(ReadWidth), 0, Data);
|
||||
}
|
||||
auto ConvertedData = _Float_FromGPR_S(OpSize::i64Bit, ReadWidth == OpSize::i32Bit ? OpSize::i32Bit : OpSize::i64Bit, Data);
|
||||
_PushStack(ConvertedData, Data, ReadWidth, false);
|
||||
_PushStack(ConvertedData, Invalid(), OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FISTF64(OpcodeArgs, bool Truncate) {
|
||||
@@ -112,7 +111,7 @@ void OpDispatchBuilder::FISTF64(OpcodeArgs, bool Truncate) {
|
||||
} else {
|
||||
data = _Float_ToGPR_S(Size == OpSize::i32Bit ? OpSize::i32Bit : OpSize::i64Bit, OpSize::i64Bit, data);
|
||||
}
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, data, Size, OpSize::i8Bit);
|
||||
StoreResultGPR_WithOpSize(Op, Op->Dest, data, Size, OpSize::i8Bit);
|
||||
|
||||
if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) {
|
||||
_PopStackDestroy();
|
||||
@@ -138,16 +137,16 @@ void OpDispatchBuilder::FADDF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpDi
|
||||
Ref arg {};
|
||||
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
if (Width == OpSize::i16Bit) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
@@ -176,16 +175,16 @@ void OpDispatchBuilder::FMULF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpDi
|
||||
Ref arg {};
|
||||
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
if (Width == OpSize::i16Bit) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
@@ -228,16 +227,16 @@ void OpDispatchBuilder::FDIVF64(OpcodeArgs, IR::OpSize Width, bool Integer, bool
|
||||
|
||||
if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
if (Integer) {
|
||||
Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
if (Width == OpSize::i16Bit) {
|
||||
Arg = _Sbfe(OpSize::i64Bit, 16, 0, Arg);
|
||||
}
|
||||
Arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, Arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, Arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
@@ -285,16 +284,16 @@ void OpDispatchBuilder::FSUBF64(OpcodeArgs, IR::OpSize Width, bool Integer, bool
|
||||
|
||||
if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
if (Width == OpSize::i16Bit) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
@@ -332,16 +331,16 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpD
|
||||
} else if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
// Memory arg
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
if (Width == OpSize::i16Bit) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
b = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
@@ -393,11 +392,11 @@ void OpDispatchBuilder::X87FXTRACTF64(OpcodeArgs) {
|
||||
SaveNZCV();
|
||||
_TestNZ(OpSize::i64Bit, Gpr, Constant(0x7fff'ffff'ffff'ffffUL));
|
||||
|
||||
Ref Sig = _NZCVSelectV(OpSize::i64Bit, {COND_EQ}, SigZV, SigNZV);
|
||||
Ref Exp = _NZCVSelectV(OpSize::i64Bit, {COND_EQ}, ExpZV, ExpNZV);
|
||||
Ref Sig = _NZCVSelectV(OpSize::i64Bit, CondClass::EQ, SigZV, SigNZV);
|
||||
Ref Exp = _NZCVSelectV(OpSize::i64Bit, CondClass::EQ, ExpZV, ExpNZV);
|
||||
|
||||
_PopStackDestroy();
|
||||
_PushStack(Exp, Exp, OpSize::i64Bit, true);
|
||||
_PushStack(Sig, Sig, OpSize::i64Bit, true);
|
||||
_PushStack(Exp, Invalid(), OpSize::iInvalid);
|
||||
_PushStack(Sig, Invalid(), OpSize::iInvalid);
|
||||
}
|
||||
} // namespace FEXCore::IR
|
||||
@@ -1,87 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: glue|x86-guest-code
|
||||
desc: Guest-side assembly helpers used by the backends
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "FEXCore/Utils/AllocatorHooks.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <cstring>
|
||||
|
||||
namespace FEXCore {
|
||||
constexpr size_t CODE_SIZE = 0x1000;
|
||||
|
||||
X86GeneratedCode::X86GeneratedCode() {
|
||||
#ifdef _WIN32
|
||||
// No need to allocate anything in this config.
|
||||
#else
|
||||
|
||||
// Allocate a page for our emulated guest
|
||||
CodePtr = AllocateGuestCodeSpace(CODE_SIZE);
|
||||
|
||||
constexpr std::array<uint8_t, 2> SignalReturnCode = {
|
||||
0x0F, 0x37, // CALLBACKRET FEX Instruction
|
||||
};
|
||||
|
||||
CallbackReturn = reinterpret_cast<uint64_t>(CodePtr);
|
||||
|
||||
memcpy(reinterpret_cast<void*>(CallbackReturn), SignalReturnCode.data(), SignalReturnCode.size());
|
||||
|
||||
mprotect(CodePtr, CODE_SIZE, PROT_READ);
|
||||
#endif
|
||||
}
|
||||
|
||||
X86GeneratedCode::~X86GeneratedCode() {
|
||||
#ifndef _WIN32
|
||||
FEXCore::Allocator::VirtualFree(CodePtr, CODE_SIZE);
|
||||
#endif
|
||||
}
|
||||
|
||||
void* X86GeneratedCode::AllocateGuestCodeSpace(size_t Size) {
|
||||
#ifndef _WIN32
|
||||
FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE);
|
||||
|
||||
if (Is64BitMode()) {
|
||||
// 64bit mode can have its sigret handler anywhere
|
||||
return FEXCore::Allocator::VirtualAlloc(Size);
|
||||
}
|
||||
|
||||
// First 64bit page
|
||||
constexpr uintptr_t LOCATION_MAX = 0x1'0000'0000;
|
||||
|
||||
// 32bit mode
|
||||
// We need to have the sigret handler in the lower 32bits of memory space
|
||||
// Scan top down and try to allocate a location
|
||||
for (size_t Location = 0xFFFF'E000; Location != 0x0; Location -= 0x1000) {
|
||||
void* Ptr = ::mmap(reinterpret_cast<void*>(Location), Size, PROT_READ | PROT_WRITE, MAP_FIXED_NOREPLACE | MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
|
||||
if (Ptr != MAP_FAILED && reinterpret_cast<uintptr_t>(Ptr) >= LOCATION_MAX) {
|
||||
// Failed to map in the lower 32bits
|
||||
// Try again
|
||||
// Can happen in the case that host kernel ignores MAP_FIXED_NOREPLACE
|
||||
::munmap(Ptr, Size);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (Ptr != MAP_FAILED) {
|
||||
return Ptr;
|
||||
}
|
||||
}
|
||||
|
||||
// Can't do anything about this
|
||||
// Here's hoping the application doesn't use signals
|
||||
return MAP_FAILED;
|
||||
#else
|
||||
return nullptr;
|
||||
#endif
|
||||
}
|
||||
|
||||
} // namespace FEXCore
|
||||
@@ -1,25 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: glue|x86-guest-code
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore {
|
||||
class X86GeneratedCode final {
|
||||
public:
|
||||
X86GeneratedCode();
|
||||
~X86GeneratedCode();
|
||||
|
||||
uint64_t CallbackReturn {};
|
||||
|
||||
private:
|
||||
void* CodePtr {};
|
||||
void* AllocateGuestCodeSpace(size_t Size);
|
||||
};
|
||||
} // namespace FEXCore
|
||||
@@ -100,10 +100,11 @@ constexpr std::array<X86InstInfo, MAX_SECOND_TABLE_SIZE> SecondBaseOps = []() co
|
||||
{0x34, 1, X86InstInfo{"SYSENTER", TYPE_INST, FLAGS_NO_OVERLAY, 0}},
|
||||
{0x35, 1, X86InstInfo{"SYSEXIT", TYPE_INST, FLAGS_NO_OVERLAY, 0}},
|
||||
{0x36, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0}},
|
||||
{0x37, 1, X86InstInfo{"GETSEC", TYPE_INVALID, FLAGS_NO_OVERLAY, 0}},
|
||||
{0x38, 1, X86InstInfo{"", TYPE_0F38_TABLE, FLAGS_NO_OVERLAY, 0}},
|
||||
{0x39, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0}},
|
||||
{0x3A, 1, X86InstInfo{"", TYPE_0F3A_TABLE, FLAGS_NO_OVERLAY, 0}},
|
||||
{0x3B, 4, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0}},
|
||||
{0x3B, 3, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0}},
|
||||
|
||||
{0x40, 1, X86InstInfo{"CMOVO", TYPE_INST, FLAGS_MODRM | FLAGS_NO_OVERLAY, 0}},
|
||||
{0x41, 1, X86InstInfo{"CMOVNO", TYPE_INST, FLAGS_MODRM | FLAGS_NO_OVERLAY, 0}},
|
||||
@@ -299,7 +300,7 @@ constexpr std::array<X86InstInfo, MAX_SECOND_TABLE_SIZE> SecondBaseOps = []() co
|
||||
// FEX reserved instructions
|
||||
// Unused x86 encoding instruction.
|
||||
|
||||
{0x37, 1, X86InstInfo{"CALLBACKRET", TYPE_INST, FLAGS_BLOCK_END | FLAGS_NO_OVERLAY | FLAGS_SETS_RIP, 0}},
|
||||
{0x3E, 1, X86InstInfo{"CALLBACKRET", TYPE_INST, FLAGS_BLOCK_END | FLAGS_NO_OVERLAY | FLAGS_SETS_RIP, 0}},
|
||||
|
||||
// This was originally used by VIA to jump to its alternative instruction set. Used for OP_THUNK
|
||||
{0x3F, 1, X86InstInfo{"ALTINST", TYPE_INST, FLAGS_BLOCK_END | FLAGS_NO_OVERLAY | FLAGS_SETS_RIP, 0}},
|
||||
@@ -442,7 +443,7 @@ constexpr std::array<X86InstInfo, MAX_REPNE_MOD_TABLE_SIZE> RepNEModOps = []() c
|
||||
{0x70, 1, X86InstInfo{"PSHUFLW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1}},
|
||||
{0x71, 3, X86InstInfo{"", TYPE_COPY_OTHER, FLAGS_NONE, 0}},
|
||||
{0x74, 4, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{0x78, 1, X86InstInfo{"INSERTQ", TYPE_INST, GenFlagsSameSize(SIZE_64BIT) | FLAGS_MODRM | FLAGS_SF_MOD_REG_ONLY | FLAGS_XMM_FLAGS,2}},
|
||||
{0x78, 1, X86InstInfo{"INSERTQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_REG_ONLY | FLAGS_XMM_FLAGS,2}},
|
||||
{0x79, 1, X86InstInfo{"INSERTQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_SF_MOD_REG_ONLY | FLAGS_XMM_FLAGS, 0}},
|
||||
{0x7A, 2, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0}},
|
||||
{0x7C, 1, X86InstInfo{"HADDPS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 0}},
|
||||
|
||||
@@ -7,6 +7,7 @@ $end_info$
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include <array>
|
||||
@@ -559,21 +560,6 @@ constexpr static inline void GenerateTableWithCopy(X86InstInfo *FinalTable, X86T
|
||||
}
|
||||
};
|
||||
|
||||
template<typename OpcodeType>
|
||||
static inline void LateInitCopyTable(X86InstInfo *FinalTable, X86TablesInfoStruct<OpcodeType> const *OtherLocal, size_t OtherTableSize) {
|
||||
for (size_t j = 0; j < OtherTableSize; ++j) {
|
||||
X86TablesInfoStruct<OpcodeType> const &OtherOp = OtherLocal[j];
|
||||
auto OtherOpNum = OtherOp.first;
|
||||
X86InstInfo const &OtherInfo = OtherOp.Info;
|
||||
for (uint32_t i = 0; i < OtherOp.second; ++i) {
|
||||
X86InstInfo &FinalOp = FinalTable[OtherOpNum + i];
|
||||
if (FinalOp.Type == TYPE_COPY_OTHER) {
|
||||
FinalOp = OtherInfo;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template<typename OpcodeType>
|
||||
constexpr static inline void GenerateX87Table(X86InstInfo *FinalTable, X86TablesInfoStruct<OpcodeType> const *LocalTable, size_t TableSize) {
|
||||
for (size_t j = 0; j < TableSize; ++j) {
|
||||
@@ -604,12 +590,6 @@ constexpr static inline void GenerateX87Table(X86InstInfo *FinalTable, X86Tables
|
||||
}
|
||||
};
|
||||
|
||||
}
|
||||
FEX_DEFINE_ENUM_FMT_PASSTHROUGH(FEXCore::X86Tables::DecodedOperand::OpType);
|
||||
|
||||
template <>
|
||||
struct fmt::formatter<FEXCore::X86Tables::DecodedOperand::OpType> : formatter<uint32_t> {
|
||||
template <typename FormatContext>
|
||||
auto format(FEXCore::X86Tables::DecodedOperand::OpType type, FormatContext& ctx) const {
|
||||
return fmt::formatter<uint32_t>::format(static_cast<uint32_t>(type), ctx);
|
||||
}
|
||||
};
|
||||
} // namespace FEXCore::X86Tables
|
||||
@@ -20,7 +20,6 @@
|
||||
namespace FEXCore::IR {
|
||||
|
||||
class OrderedNode;
|
||||
class RegisterAllocationPass;
|
||||
|
||||
/**
|
||||
* @brief The IROp_Header is an dynamically sized array
|
||||
@@ -61,24 +60,7 @@ struct NodeID final {
|
||||
Value = 0;
|
||||
}
|
||||
|
||||
[[nodiscard]] friend constexpr bool operator==(NodeID, NodeID) noexcept = default;
|
||||
|
||||
[[nodiscard]]
|
||||
friend constexpr bool operator<(NodeID lhs, NodeID rhs) noexcept {
|
||||
return lhs.Value < rhs.Value;
|
||||
}
|
||||
[[nodiscard]]
|
||||
friend constexpr bool operator>(NodeID lhs, NodeID rhs) noexcept {
|
||||
return operator<(rhs, lhs);
|
||||
}
|
||||
[[nodiscard]]
|
||||
friend constexpr bool operator<=(NodeID lhs, NodeID rhs) noexcept {
|
||||
return !operator>(lhs, rhs);
|
||||
}
|
||||
[[nodiscard]]
|
||||
friend constexpr bool operator>=(NodeID lhs, NodeID rhs) noexcept {
|
||||
return !operator<(lhs, rhs);
|
||||
}
|
||||
[[nodiscard]] constexpr auto operator<=>(const NodeID&) const noexcept = default;
|
||||
|
||||
friend std::ostream& operator<<(std::ostream& out, NodeID ID) {
|
||||
out << ID.Value;
|
||||
@@ -431,94 +413,6 @@ static_assert(sizeof(OrderedNode) == (sizeof(OrderedNodeHeader) + 2 * sizeof(uin
|
||||
// };
|
||||
using Ref = OrderedNode*;
|
||||
|
||||
struct FEX_PACKED RegisterClassType final {
|
||||
using value_type = uint32_t;
|
||||
|
||||
value_type Val;
|
||||
[[nodiscard]] constexpr operator value_type() const {
|
||||
return Val;
|
||||
}
|
||||
[[nodiscard]]
|
||||
friend constexpr bool operator==(const RegisterClassType&, const RegisterClassType&) = default;
|
||||
};
|
||||
|
||||
struct FEX_PACKED CondClassType final {
|
||||
uint8_t Val;
|
||||
[[nodiscard]] constexpr operator uint8_t() const {
|
||||
return Val;
|
||||
}
|
||||
[[nodiscard]]
|
||||
friend constexpr bool operator==(const CondClassType&, const CondClassType&) = default;
|
||||
};
|
||||
|
||||
struct FEX_PACKED MemOffsetType final {
|
||||
uint8_t Val;
|
||||
[[nodiscard]] constexpr operator uint8_t() const {
|
||||
return Val;
|
||||
}
|
||||
[[nodiscard]]
|
||||
friend constexpr bool operator==(const MemOffsetType&, const MemOffsetType&) = default;
|
||||
};
|
||||
|
||||
struct FEX_PACKED TypeDefinition final {
|
||||
uint16_t Val;
|
||||
|
||||
[[nodiscard]] constexpr operator uint16_t() const {
|
||||
return Val;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static constexpr TypeDefinition Create(uint8_t Bytes) {
|
||||
TypeDefinition Type {};
|
||||
Type.Val = Bytes << 8;
|
||||
return Type;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static constexpr TypeDefinition Create(uint8_t Bytes, uint8_t Elements) {
|
||||
TypeDefinition Type {};
|
||||
Type.Val = (Bytes << 8) | (Elements & 255);
|
||||
return Type;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
constexpr uint8_t Bytes() const {
|
||||
return Val >> 8;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
constexpr uint8_t Elements() const {
|
||||
return Val & 255;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
friend constexpr bool operator==(const TypeDefinition&, const TypeDefinition&) = default;
|
||||
};
|
||||
|
||||
static_assert(std::is_trivially_copyable_v<TypeDefinition>);
|
||||
|
||||
struct FEX_PACKED FenceType final {
|
||||
using value_type = uint8_t;
|
||||
|
||||
value_type Val;
|
||||
[[nodiscard]] constexpr operator value_type() const {
|
||||
return Val;
|
||||
}
|
||||
[[nodiscard]]
|
||||
friend constexpr bool operator==(const FenceType&, const FenceType&) = default;
|
||||
};
|
||||
|
||||
struct FEX_PACKED RoundType final {
|
||||
uint8_t Val;
|
||||
[[nodiscard]] constexpr operator uint8_t() const {
|
||||
return Val;
|
||||
}
|
||||
[[nodiscard]]
|
||||
friend constexpr bool operator==(const RoundType&, const RoundType&) = default;
|
||||
};
|
||||
|
||||
class NodeIterator;
|
||||
|
||||
/* This iterator can be used to step though nodes.
|
||||
* Due to how our IR is laid out, this can be used to either step
|
||||
* though the CodeBlocks or though the code within a single block.
|
||||
@@ -782,6 +676,15 @@ inline NodeID NodeWrapperBase<Type>::ID() const {
|
||||
bool IsBlockExit(FEXCore::IR::IROps Op);
|
||||
|
||||
void Dump(fextl::stringstream* out, const IRListView* IR);
|
||||
|
||||
constexpr auto format_as(FEXCore::IR::NodeID ID) {
|
||||
return ID.Value;
|
||||
}
|
||||
|
||||
FEX_DEFINE_ENUM_FMT_PASSTHROUGH(FEXCore::IR::FenceType)
|
||||
FEX_DEFINE_ENUM_FMT_PASSTHROUGH(FEXCore::IR::MemOffsetType)
|
||||
FEX_DEFINE_ENUM_FMT_PASSTHROUGH(FEXCore::IR::OpSize)
|
||||
FEX_DEFINE_ENUM_FMT_PASSTHROUGH(FEXCore::IR::RegClass)
|
||||
} // namespace FEXCore::IR
|
||||
|
||||
template<>
|
||||
@@ -790,45 +693,3 @@ struct std::hash<FEXCore::IR::NodeID> {
|
||||
return std::hash<FEXCore::IR::NodeID::value_type> {}(ID.Value);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct fmt::formatter<FEXCore::IR::NodeID> : fmt::formatter<FEXCore::IR::NodeID::value_type> {
|
||||
using Base = fmt::formatter<FEXCore::IR::NodeID::value_type>;
|
||||
|
||||
// Pass-through the underlying value, so IDs can
|
||||
// be formatted like any integral value.
|
||||
template<typename FormatContext>
|
||||
auto format(const FEXCore::IR::NodeID& ID, FormatContext& ctx) const {
|
||||
return Base::format(ID.Value, ctx);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct fmt::formatter<FEXCore::IR::RegisterClassType> : fmt::formatter<FEXCore::IR::RegisterClassType::value_type> {
|
||||
using Base = fmt::formatter<FEXCore::IR::RegisterClassType::value_type>;
|
||||
|
||||
template<typename FormatContext>
|
||||
auto format(const FEXCore::IR::RegisterClassType& Class, FormatContext& ctx) const {
|
||||
return Base::format(Class.Val, ctx);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct fmt::formatter<FEXCore::IR::FenceType> : fmt::formatter<FEXCore::IR::FenceType::value_type> {
|
||||
using Base = fmt::formatter<FEXCore::IR::FenceType::value_type>;
|
||||
|
||||
template<typename FormatContext>
|
||||
auto format(const FEXCore::IR::FenceType& Fence, FormatContext& ctx) const {
|
||||
return Base::format(Fence.Val, ctx);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct fmt::formatter<FEXCore::IR::OpSize> : fmt::formatter<std::underlying_type_t<FEXCore::IR::OpSize>> {
|
||||
using Base = fmt::formatter<std::underlying_type_t<FEXCore::IR::OpSize>>;
|
||||
|
||||
template<typename FormatContext>
|
||||
auto format(const FEXCore::IR::OpSize& OpSize, FormatContext& ctx) const {
|
||||
return Base::format(FEXCore::ToUnderlying(OpSize), ctx);
|
||||
}
|
||||
};
|
||||
+117
-142
@@ -52,80 +52,68 @@
|
||||
" * These are validations that can't be automatically inferred and need to be hand-written",
|
||||
""
|
||||
],
|
||||
"Enums": {
|
||||
"class CondClass : uint8_t": [
|
||||
"EQ = 0,",
|
||||
"NEQ = 1,",
|
||||
"UGE = 2,",
|
||||
"ULT = 3,",
|
||||
"MI = 4,",
|
||||
"PL = 5,",
|
||||
"VS = 6,",
|
||||
"VC = 7,",
|
||||
"UGT = 8,",
|
||||
"ULE = 9,",
|
||||
"SGE = 10,",
|
||||
"SLT = 11,",
|
||||
"SGT = 12,",
|
||||
"SLE = 13,",
|
||||
"TSTZ = 14, /* bit test zero */",
|
||||
"TSTNZ = 15, /* bit test nonzero */",
|
||||
"",
|
||||
"FLU = 16, /* float less or unordered */",
|
||||
"FGE = 17, /* float greater or equal */",
|
||||
"FLEU = 18, /* float less or equal or unordered */",
|
||||
"FGT = 19, /* float greater */",
|
||||
"FU = 20, /* float unordered */",
|
||||
"FNU = 21, /* float not unordered */",
|
||||
"",
|
||||
"AL = 32, /* always */"
|
||||
],
|
||||
"class FenceType : uint8_t": [
|
||||
"Load = 0,",
|
||||
"Store = 1,",
|
||||
"LoadStore = 2,",
|
||||
"Inst = 3,"
|
||||
],
|
||||
"class MemOffsetType : uint8_t": [
|
||||
"SXTX = 0,",
|
||||
"UXTW = 1,",
|
||||
"SXTW = 2,"
|
||||
],
|
||||
"class RegClass : uint32_t": [
|
||||
"Invalid = 0,",
|
||||
"GPR = 1,",
|
||||
"GPRFixed = 2,",
|
||||
"FPR = 3,",
|
||||
"FPRFixed = 4,",
|
||||
"Complex = 5,"
|
||||
],
|
||||
"class RoundMode : uint8_t": [
|
||||
"Nearest = 0,",
|
||||
"NegInfinity = 1,",
|
||||
"PosInfinity = 2,",
|
||||
"TowardsZero = 3, /* Truncate */",
|
||||
"Host = 4,"
|
||||
]
|
||||
},
|
||||
"Defines": [
|
||||
"constexpr uint8_t COND_EQ = 0",
|
||||
"constexpr uint8_t COND_NEQ = 1",
|
||||
"constexpr uint8_t COND_UGE = 2",
|
||||
"constexpr uint8_t COND_ULT = 3",
|
||||
"constexpr uint8_t COND_MI = 4",
|
||||
"constexpr uint8_t COND_PL = 5",
|
||||
"constexpr uint8_t COND_VS = 6",
|
||||
"constexpr uint8_t COND_VC = 7",
|
||||
"constexpr uint8_t COND_UGT = 8",
|
||||
"constexpr uint8_t COND_ULE = 9",
|
||||
"constexpr uint8_t COND_SGE = 10",
|
||||
"constexpr uint8_t COND_SLT = 11",
|
||||
"constexpr uint8_t COND_SGT = 12",
|
||||
"constexpr uint8_t COND_SLE = 13",
|
||||
"constexpr uint8_t COND_TSTZ = 14 /* bit test zero */",
|
||||
"constexpr uint8_t COND_TSTNZ = 15 /* bit test nonzero */",
|
||||
|
||||
"constexpr uint8_t COND_FLU = 16 /* float less or unordred */",
|
||||
"constexpr uint8_t COND_FGE = 17 /* float greater or equal */",
|
||||
"constexpr uint8_t COND_FLEU = 18 /* float less or equal or unordred */",
|
||||
"constexpr uint8_t COND_FGT = 19 /* float greater */",
|
||||
"constexpr uint8_t COND_FU = 20 /* float unordred */",
|
||||
"constexpr uint8_t COND_FNU = 21 /* float not unordred */",
|
||||
|
||||
"constexpr uint8_t COND_AL = 32 /* always */",
|
||||
|
||||
"constexpr FEXCore::IR::RegisterClassType InvalidClass {0}",
|
||||
"constexpr FEXCore::IR::RegisterClassType GPRClass {1}",
|
||||
"constexpr FEXCore::IR::RegisterClassType GPRFixedClass {2}",
|
||||
"constexpr FEXCore::IR::RegisterClassType FPRClass {3}",
|
||||
"constexpr FEXCore::IR::RegisterClassType FPRFixedClass {4}",
|
||||
"constexpr FEXCore::IR::RegisterClassType ComplexClass {5}",
|
||||
"constexpr uint8_t NumClasses {6}",
|
||||
"",
|
||||
"constexpr FEXCore::IR::TypeDefinition i8 {TypeDefinition::Create(1, 0)}",
|
||||
"constexpr FEXCore::IR::TypeDefinition i16 {TypeDefinition::Create(2, 0)}",
|
||||
"constexpr FEXCore::IR::TypeDefinition i32 {TypeDefinition::Create(4, 0)}",
|
||||
"constexpr FEXCore::IR::TypeDefinition i64 {TypeDefinition::Create(8, 0)}",
|
||||
"constexpr FEXCore::IR::TypeDefinition i128 {TypeDefinition::Create(16, 0)}",
|
||||
"",
|
||||
"constexpr FEXCore::IR::TypeDefinition i8v8 {TypeDefinition::Create(1, 8)}",
|
||||
"constexpr FEXCore::IR::TypeDefinition i8v16 {TypeDefinition::Create(1, 16)}",
|
||||
"constexpr FEXCore::IR::TypeDefinition i16v4 {TypeDefinition::Create(2, 4)}",
|
||||
"constexpr FEXCore::IR::TypeDefinition i16v8 {TypeDefinition::Create(2, 8)}",
|
||||
"constexpr FEXCore::IR::TypeDefinition i32v2 {TypeDefinition::Create(4, 2)}",
|
||||
"constexpr FEXCore::IR::TypeDefinition i32v4 {TypeDefinition::Create(4, 4)}",
|
||||
"constexpr FEXCore::IR::TypeDefinition i64v2 {TypeDefinition::Create(8, 2)}",
|
||||
"",
|
||||
|
||||
"constexpr uint8_t FCMP_FLAG_EQ = 0",
|
||||
"constexpr uint8_t FCMP_FLAG_LT = 1",
|
||||
"constexpr uint8_t FCMP_FLAG_UNORDERED = 2",
|
||||
|
||||
"constexpr FEXCore::IR::FenceType Fence_Load {0}",
|
||||
"constexpr FEXCore::IR::FenceType Fence_Store {1}",
|
||||
"constexpr FEXCore::IR::FenceType Fence_LoadStore {2}",
|
||||
"constexpr FEXCore::IR::FenceType Fence_Inst {3}",
|
||||
|
||||
"constexpr uint8_t ROUND_MODE_NEAREST = 0",
|
||||
"constexpr uint8_t ROUND_MODE_NEGATIVE_INFINITY = 1",
|
||||
"constexpr uint8_t ROUND_MODE_POSITIVE_INFINITY = 2",
|
||||
"constexpr uint8_t ROUND_MODE_TOWARDS_ZERO = 3",
|
||||
"constexpr uint8_t ROUND_MODE_FLUSH_TO_ZERO = 1 << 2",
|
||||
|
||||
"constexpr FEXCore::IR::RoundType Round_Nearest {ROUND_MODE_NEAREST}",
|
||||
"constexpr FEXCore::IR::RoundType Round_Negative_Infinity {ROUND_MODE_NEGATIVE_INFINITY}",
|
||||
"constexpr FEXCore::IR::RoundType Round_Positive_Infinity {ROUND_MODE_POSITIVE_INFINITY}",
|
||||
"constexpr FEXCore::IR::RoundType Round_Towards_Zero {ROUND_MODE_TOWARDS_ZERO} /* Truncate */",
|
||||
"constexpr FEXCore::IR::RoundType Round_Host {ROUND_MODE_TOWARDS_ZERO + 1}",
|
||||
|
||||
"constexpr FEXCore::IR::MemOffsetType MEM_OFFSET_SXTX {0}",
|
||||
"constexpr FEXCore::IR::MemOffsetType MEM_OFFSET_UXTW {1}",
|
||||
"constexpr FEXCore::IR::MemOffsetType MEM_OFFSET_SXTW {2}",
|
||||
|
||||
"struct BreakDefinition {",
|
||||
" uint16_t ErrorRegister;",
|
||||
" uint8_t Signal;",
|
||||
@@ -148,13 +136,12 @@
|
||||
"GPR": "OrderedNode*",
|
||||
"FPR": "OrderedNode*",
|
||||
"FenceType": "FenceType",
|
||||
"RegisterClass": "RegisterClassType",
|
||||
"CondClass": "CondClassType",
|
||||
"SyscallFlags": "FEXCore::IR::SyscallFlags",
|
||||
"RegisterClass": "RegClass",
|
||||
"CondClass": "CondClass",
|
||||
"SHA256Sum": "SHA256Sum",
|
||||
"MemOffsetType": "MemOffsetType",
|
||||
"BreakDefinition": "BreakDefinition",
|
||||
"RoundType": "RoundType",
|
||||
"RoundType": "RoundMode",
|
||||
"FloatCompareOp": "FloatCompareOp",
|
||||
"NamedVectorConstant": "FEXCore::IR::NamedVectorConstant",
|
||||
"IndexNamedVectorConstant": "FEXCore::IR::IndexNamedVectorConstant",
|
||||
@@ -307,7 +294,7 @@
|
||||
"HasSideEffects": true,
|
||||
"RAOverride": "0"
|
||||
},
|
||||
"CondJump SSA:$Cmp1, SSA:$Cmp2, SSA:$TrueBlock, SSA:$FalseBlock, CondClass:$Cond{{COND_NEQ}}, OpSize:$CompareSize{OpSize::iInvalid}, i1:$FromNZCV{false}": {
|
||||
"CondJump SSA:$Cmp1, SSA:$Cmp2, SSA:$TrueBlock, SSA:$FalseBlock, CondClass:$Cond{CondClass::NEQ}, OpSize:$CompareSize{OpSize::iInvalid}, i1:$FromNZCV{false}": {
|
||||
"Inline": ["", "AddSub"],
|
||||
"HasSideEffects": true,
|
||||
"RAOverride": "2"
|
||||
@@ -326,25 +313,13 @@
|
||||
"CallbackReturn": {
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"GPR = Syscall GPR:$SyscallID, GPR:$Arg0, GPR:$Arg1, GPR:$Arg2, GPR:$Arg3, GPR:$Arg4, GPR:$Arg5, SyscallFlags:$Flags": {
|
||||
"GPR = Syscall GPR:$SyscallID, GPR:$Arg0, GPR:$Arg1, GPR:$Arg2, GPR:$Arg3, GPR:$Arg4, GPR:$Arg5": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Dispatches a guest syscall through to the SyscallHandler class"
|
||||
],
|
||||
"DestSize": "OpSize::i64Bit"
|
||||
},
|
||||
|
||||
"GPR = InlineSyscall GPR:$Arg0, GPR:$Arg1, GPR:$Arg2, GPR:$Arg3, GPR:$Arg4, GPR:$Arg5, i32:$HostSyscallNumber, SyscallFlags:$Flags": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Dispatches a guest syscall directly to the host syscall interface,",
|
||||
"bypassing the SyscallHandler class used by Syscall.",
|
||||
"This has significantly less overhead than Syscall, which needs to save JIT state first.",
|
||||
"Can only be used for syscalls that match across architecture,",
|
||||
"such as gettid (matches on x86/x86-64/Arm64)."
|
||||
],
|
||||
|
||||
"DestSize": "OpSize::i64Bit"
|
||||
},
|
||||
|
||||
"Thunk GPR:$ArgPtr, SHA256Sum:$ThunkNameHash": {
|
||||
"HasSideEffects": true
|
||||
},
|
||||
@@ -364,20 +339,6 @@
|
||||
"GPR = Copy GPR:$Source": {
|
||||
"Desc": ["GPR copy, generated by RA to split live ranges"],
|
||||
"DestSize": "OpSize::i64Bit"
|
||||
},
|
||||
|
||||
"GPR = Swap1 GPR:$A, GPR:$B": {
|
||||
"Desc": ["GPR swap part 1, generated by RA. Returns value of first source.",
|
||||
"Destination must be second GPR."],
|
||||
"DestSize": "OpSize::i64Bit"
|
||||
},
|
||||
|
||||
"GPR = Swap2": {
|
||||
"Desc": ["GPR swap part 2, generated by RA. Returns source source.",
|
||||
"Must immediately succeed Swap1 with no intervening instructions",
|
||||
"Kludge to workaround single destination restriction on IR",
|
||||
"Hopefully temporary"],
|
||||
"DestSize": "OpSize::i64Bit"
|
||||
}
|
||||
},
|
||||
"StaticRA": {
|
||||
@@ -423,8 +384,8 @@
|
||||
],
|
||||
"DestSize": "ByteSize",
|
||||
"EmitValidation": [
|
||||
"($Class == GPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == GPRClass",
|
||||
"($Class == RegClass::GPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == RegClass::FPR",
|
||||
"($Class == RegClass::FPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == RegClass::GPR",
|
||||
"!($Offset >= offsetof(Core::CPUState, gregs[0]) && $Offset < offsetof(Core::CPUState, gregs[16])) && \"Can't LoadContext to GPR\"",
|
||||
"!($Offset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $Offset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't LoadContext to XMM\""
|
||||
]
|
||||
@@ -437,8 +398,8 @@
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "ByteSize",
|
||||
"EmitValidation": [
|
||||
"($Class == GPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == GPRClass",
|
||||
"($Class == RegClass::GPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == RegClass::FPR",
|
||||
"($Class == RegClass::FPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == RegClass::GPR",
|
||||
"!($Offset >= offsetof(Core::CPUState, gregs[0]) && $Offset < offsetof(Core::CPUState, gregs[16])) && \"Can't LoadContext to GPR\"",
|
||||
"!($Offset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $Offset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't LoadContext to XMM\""
|
||||
]
|
||||
@@ -454,8 +415,8 @@
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "ByteSize",
|
||||
"EmitValidation": [
|
||||
"($Class == GPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == GPRClass",
|
||||
"($Class == RegClass::GPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == RegClass::FPR",
|
||||
"($Class == RegClass::FPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == RegClass::GPR",
|
||||
"!($Offset >= offsetof(Core::CPUState, gregs[0]) && $Offset < offsetof(Core::CPUState, gregs[16])) && \"Can't StoreContext to GPR\"",
|
||||
"!($Offset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $Offset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't StoreContext to XMM\""
|
||||
]
|
||||
@@ -472,8 +433,8 @@
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value1) == $Class",
|
||||
"WalkFindRegClass($Value2) == $Class",
|
||||
"($Class == GPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == GPRClass",
|
||||
"($Class == RegClass::GPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == RegClass::FPR",
|
||||
"($Class == RegClass::FPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == RegClass::GPR",
|
||||
"!($Offset >= offsetof(Core::CPUState, gregs[0]) && $Offset < offsetof(Core::CPUState, gregs[16])) && \"Can't StoreContext to GPR\"",
|
||||
"!($Offset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $Offset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't StoreContext to XMM\""
|
||||
]
|
||||
@@ -485,8 +446,8 @@
|
||||
],
|
||||
"DestSize": "ByteSize",
|
||||
"EmitValidation": [
|
||||
"($Class == GPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == GPRClass",
|
||||
"($Class == RegClass::GPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == RegClass::FPR",
|
||||
"($Class == RegClass::FPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == RegClass::GPR",
|
||||
"!($BaseOffset >= offsetof(Core::CPUState, gregs[0]) && $BaseOffset < offsetof(Core::CPUState, gregs[16])) && \"Can't LoadContextIndexed to GPR\"",
|
||||
"!($BaseOffset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $BaseOffset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't LoadContextIndexed to XMM\""
|
||||
]
|
||||
@@ -499,8 +460,8 @@
|
||||
"DestSize": "ByteSize",
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value) == $Class",
|
||||
"($Class == GPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == GPRClass",
|
||||
"($Class == RegClass::GPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == RegClass::FPR",
|
||||
"($Class == RegClass::FPR && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == RegClass::GPR",
|
||||
"!($BaseOffset >= offsetof(Core::CPUState, gregs[0]) && $BaseOffset < offsetof(Core::CPUState, gregs[16])) && \"Can't StoreContextIndexed to GPR\"",
|
||||
"!($BaseOffset >= offsetof(Core::CPUState, xmm.avx.data[0]) && $BaseOffset < offsetof(Core::CPUState, xmm.avx.data[16])) && \"Can't StoreContextIndexed to XMM\""
|
||||
]
|
||||
@@ -630,7 +591,7 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
},
|
||||
"FPR = VLoadVectorGatherMasked OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Incoming, FPR:$Mask, GPR:$AddrBase, FPR:$VectorIndexLow, FPR:$VectorIndexHigh, OpSize:$VectorIndexElementSize, u8:$OffsetScale, u8:$DataElementOffsetStart, u8:$IndexElementOffsetStart": {
|
||||
"FPR = VLoadVectorGatherMasked OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Incoming, FPR:$Mask, GPR:$AddrBase, FPR:$VectorIndexLow, FPR:$VectorIndexHigh, OpSize:$VectorIndexElementSize, u8:$OffsetScale, u8:$DataElementOffsetStart, u8:$IndexElementOffsetStart, OpSize:$AddrSize": {
|
||||
"Desc": [
|
||||
"Does a masked load similar to VPGATHERD* where the upper bit of each element",
|
||||
"determines whether or not that element will be loaded from memory.",
|
||||
@@ -644,7 +605,7 @@
|
||||
"$VectorIndexElementSize == OpSize::i32Bit || $VectorIndexElementSize == OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"FPR = VLoadVectorGatherMaskedQPS OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Incoming, FPR:$MaskReg, GPR:$AddrBase, FPR:$VectorIndexLow, FPR:$VectorIndexHigh, u8:$OffsetScale": {
|
||||
"FPR = VLoadVectorGatherMaskedQPS OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Incoming, FPR:$MaskReg, GPR:$AddrBase, FPR:$VectorIndexLow, FPR:$VectorIndexHigh, u8:$OffsetScale, OpSize:$AddrSize": {
|
||||
"Desc": [
|
||||
"Does a masked load similar to VPGATHERQPS where the upper bit of each element",
|
||||
"determines whether or not that element will be loaded from memory.",
|
||||
@@ -759,9 +720,10 @@
|
||||
},
|
||||
"Fence FenceType:$Fence": {
|
||||
"Desc": ["Does a memory fence operation of the desired type",
|
||||
"Fence_Load: Ensures load memory operations are serialized",
|
||||
"Fence_Store: Ensures store memory operations are serialized",
|
||||
"Fence_LoadStore: Ensures loads and store memory operations are serialized",
|
||||
"FenceType::Load: Ensures load memory operations are serialized",
|
||||
"FenceType::Store: Ensures store memory operations are serialized",
|
||||
"FenceType::LoadStore: Ensures loads and store memory operations are serialized",
|
||||
"FenceType::Inst: Instruction barrier. Ensures all instructions after this point will be explicitly fetched",
|
||||
"Ensures the memory operations are globally visible"
|
||||
],
|
||||
"HasSideEffects": true
|
||||
@@ -842,16 +804,6 @@
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"AtomicXor OpSize:#Size, GPR:$Value, GPR:$Addr": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer xor",
|
||||
"IR layout must match Fetch-variant, otherwise DCE IR optimization breaks!"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i8Bit || Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = AtomicSwap OpSize:#Size, GPR:$Value, GPR:$Addr": {
|
||||
"HasSideEffects": true,
|
||||
"Desc": ["Atomic integer swap"
|
||||
@@ -1006,7 +958,7 @@
|
||||
"DestSize": "OpSize::i64Bit"
|
||||
},
|
||||
|
||||
"GPR = Neg OpSize:#Size, GPR:$Src, CondClass:$Cond{{COND_AL}}": {
|
||||
"GPR = Neg OpSize:#Size, GPR:$Src, CondClass:$Cond{CondClass::AL}": {
|
||||
"Desc": ["Integer negation, with optional predication",
|
||||
"Dest = Cond ? -Src : Src",
|
||||
"Will truncate to 64 or 32bits"
|
||||
@@ -1084,6 +1036,13 @@
|
||||
"Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = Rbit OpSize:#Size, GPR:$Src": {
|
||||
"Desc": ["Reverses the bit order of the register"],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = Add OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": [ "Integer Add",
|
||||
"Will truncate to 64 or 32bits"
|
||||
@@ -1551,6 +1510,15 @@
|
||||
"ResultSize == FEXCore::IR::OpSize::i32Bit || ResultSize == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = MaskGenerateFromBitWidth GPR:$BitWidth": {
|
||||
"Desc": ["Generates a bit mask from with a value from [0, 63]",
|
||||
"0 is special cased to full-mask",
|
||||
"Special operation for SSE4a bitmask generation."
|
||||
],
|
||||
"DestSize": "FEXCore::IR::OpSize::i64Bit",
|
||||
"ImplicitFlagClobber": true
|
||||
},
|
||||
|
||||
"GPR = Extr OpSize:#Size, GPR:$Upper, GPR:$Lower, u8:$LSB": {
|
||||
"Desc": ["Concats the two GPRs to create a value that is the size of the full two GPRs",
|
||||
"It then extracts a bitfield width that size of a GPR from the LSB",
|
||||
@@ -2159,22 +2127,34 @@
|
||||
|
||||
"FPR = VAnd OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
"ElementSize": "ElementSize",
|
||||
"EmitValidation": [
|
||||
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
|
||||
"FPR = VAndn OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
"ElementSize": "ElementSize",
|
||||
"EmitValidation": [
|
||||
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
|
||||
"FPR = VOr OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
"ElementSize": "ElementSize",
|
||||
"EmitValidation": [
|
||||
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
|
||||
"FPR = VXor OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize"
|
||||
"ElementSize": "ElementSize",
|
||||
"EmitValidation": [
|
||||
"RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
|
||||
"FPR = VUQAdd OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
@@ -2828,17 +2808,13 @@
|
||||
"X87": true,
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"PushStack FPR:$X80Src, SSA:$OriginalValue, OpSize:$LoadSize, i1:$Float": {
|
||||
"PushStack FPR:$X80Src, FPR:$OriginalValue, OpSize:$LoadSize": {
|
||||
"Desc": [
|
||||
"Pushes the provided X80Src source on to the x87 stack.",
|
||||
"Tracks OriginalValue as the original value of X80Src.",
|
||||
"Tracks OriginalValue as the original value of X80Src. OriginalValue can be Invalid() in which case no tracking is done.",
|
||||
"Opsize is 128bit for F80 values, 64-bit for low precision.",
|
||||
"LoadSize the original load size, i.e. of size of OriginalValue.",
|
||||
"Float: 80-bit, 64-bit, 32-bit",
|
||||
"Int: 64-bit, 32-bit, 16-bit"
|
||||
],
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($OriginalValue) == FPRClass || WalkFindRegClass($OriginalValue) == GPRClass"
|
||||
"Float: 80-bit, 64-bit, 32-bit"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"X87": true
|
||||
@@ -2850,13 +2826,12 @@
|
||||
"HasSideEffects": true,
|
||||
"X87": true
|
||||
},
|
||||
"StoreStackMem OpSize:$SourceSize, OpSize:$StoreSize, GPR:$Addr, GPR:$Offset, OpSize:$Align, MemOffsetType:$OffsetType, u8:$OffsetScale, i1:$Float": {
|
||||
"StoreStackMem OpSize:$SourceSize, OpSize:$StoreSize, GPR:$Addr, GPR:$Offset, OpSize:$Align, MemOffsetType:$OffsetType, u8:$OffsetScale": {
|
||||
"Desc": [
|
||||
"Takes the top value off the x87 stack and stores it to memory.",
|
||||
"SourceSize is 128bit for F80 values, 64-bit for low precision.",
|
||||
"StoreSize is the store size for conversion:",
|
||||
"Float: 80-bit, 64-bit, or 32-bit",
|
||||
"Int: 64-bit, 32-bit, 16-bit"
|
||||
"Float: 80-bit, 64-bit, or 32-bit"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"X87": true
|
||||
|
||||
@@ -38,8 +38,8 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, uint64_t Arg)
|
||||
*out << fextl::fmt::format("#{:#x}", Arg);
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, CondClassType Arg) {
|
||||
if (Arg == COND_AL) {
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, CondClass Arg) {
|
||||
if (Arg == CondClass::AL) {
|
||||
*out << "ALWAYS";
|
||||
return;
|
||||
}
|
||||
@@ -48,7 +48,7 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, CondClassType
|
||||
"UGT", "ULE", "SGE", "SLT", "SGT", "SLE", "TSTZ", "TSTNZ",
|
||||
"FLU", "FGE", "FLEU", "FGT", "FU", "FNU"};
|
||||
|
||||
*out << CondNames[Arg];
|
||||
*out << CondNames[FEXCore::ToUnderlying(Arg)];
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, MemOffsetType Arg) {
|
||||
@@ -58,39 +58,39 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, MemOffsetType
|
||||
"SXTW",
|
||||
};
|
||||
|
||||
*out << Names[Arg];
|
||||
*out << Names[FEXCore::ToUnderlying(Arg)];
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, RegisterClassType Arg) {
|
||||
if (Arg == GPRClass.Val) {
|
||||
*out << "GPR";
|
||||
} else if (Arg == GPRFixedClass.Val) {
|
||||
*out << "GPRFixed";
|
||||
} else if (Arg == FPRClass.Val) {
|
||||
*out << "FPR";
|
||||
} else if (Arg == FPRFixedClass.Val) {
|
||||
*out << "FPRFixed";
|
||||
} else {
|
||||
*out << "Unknown Registerclass " << Arg;
|
||||
}
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, RegClass Arg) {
|
||||
*out << [Arg] {
|
||||
switch (Arg) {
|
||||
case RegClass::Invalid: return "Invalid";
|
||||
case RegClass::GPR: return "GPR";
|
||||
case RegClass::GPRFixed: return "GPRFixed";
|
||||
case RegClass::FPR: return "FPR";
|
||||
case RegClass::FPRFixed: return "FPRFixed";
|
||||
case RegClass::Complex: return "Complex";
|
||||
}
|
||||
return "<Unknown RegClass Type>";
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNodeWrapper Arg) {
|
||||
if (Arg.IsImmediate()) {
|
||||
auto PhyReg = PhysicalRegister(Arg);
|
||||
|
||||
switch (PhyReg.Class) {
|
||||
case FEXCore::IR::GPRClass.Val: *out << "r"; break;
|
||||
case FEXCore::IR::GPRFixedClass.Val: *out << "R"; break;
|
||||
case FEXCore::IR::FPRClass.Val: *out << "v"; break;
|
||||
case FEXCore::IR::FPRFixedClass.Val: *out << "V"; break;
|
||||
case FEXCore::IR::ComplexClass.Val: *out << "c"; break;
|
||||
case FEXCore::IR::InvalidClass.Val: *out << "invalid"; break;
|
||||
switch (PhyReg.AsRegClass()) {
|
||||
case RegClass::GPR: *out << "r"; break;
|
||||
case RegClass::GPRFixed: *out << "R"; break;
|
||||
case RegClass::FPR: *out << "v"; break;
|
||||
case RegClass::FPRFixed: *out << "V"; break;
|
||||
case RegClass::Complex: *out << "c"; break;
|
||||
case RegClass::Invalid: *out << "invalid"; break;
|
||||
default: *out << "unknown"; break;
|
||||
}
|
||||
|
||||
if (PhyReg.Class != FEXCore::IR::InvalidClass.Val) {
|
||||
*out << std::dec << (uint32_t)PhyReg.Reg;
|
||||
if (PhyReg.AsRegClass() != RegClass::Invalid) {
|
||||
*out << std::dec << uint32_t(PhyReg.Reg);
|
||||
}
|
||||
|
||||
return;
|
||||
@@ -124,41 +124,32 @@ static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNode
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::FenceType Arg) {
|
||||
if (Arg == IR::Fence_Load) {
|
||||
*out << "Loads";
|
||||
} else if (Arg == IR::Fence_Store) {
|
||||
*out << "Stores";
|
||||
} else if (Arg == IR::Fence_LoadStore) {
|
||||
*out << "LoadStores";
|
||||
} else {
|
||||
*out << "<Unknown Fence Type>";
|
||||
}
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, FenceType Arg) {
|
||||
*out << [Arg] {
|
||||
switch (Arg) {
|
||||
case FenceType::Load: return "Loads";
|
||||
case FenceType::Store: return "Stores";
|
||||
case FenceType::LoadStore: return "LoadStores";
|
||||
case FenceType::Inst: return "Instruction";
|
||||
}
|
||||
return "<Unknown Fence Type>";
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::RoundType Arg) {
|
||||
switch (Arg) {
|
||||
case FEXCore::IR::Round_Nearest: *out << "Nearest"; break;
|
||||
case FEXCore::IR::Round_Negative_Infinity: *out << "-Inf"; break;
|
||||
case FEXCore::IR::Round_Positive_Infinity: *out << "+Inf"; break;
|
||||
case FEXCore::IR::Round_Towards_Zero: *out << "Towards Zero"; break;
|
||||
case FEXCore::IR::Round_Host: *out << "Host"; break;
|
||||
default: *out << "<Unknown Round Type>"; break;
|
||||
}
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, RoundMode Arg) {
|
||||
*out << [Arg] {
|
||||
switch (Arg) {
|
||||
case RoundMode::Nearest: return "Nearest";
|
||||
case RoundMode::NegInfinity: return "-Inf";
|
||||
case RoundMode::PosInfinity: return "+Inf";
|
||||
case RoundMode::TowardsZero: return "Towards Zero";
|
||||
case RoundMode::Host: return "Host";
|
||||
}
|
||||
return "<Unknown Round Type>";
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::SyscallFlags Arg) {
|
||||
switch (Arg) {
|
||||
case FEXCore::IR::SyscallFlags::DEFAULT: *out << "Default"; break;
|
||||
case FEXCore::IR::SyscallFlags::OPTIMIZETHROUGH: *out << "Optimize Through"; break;
|
||||
case FEXCore::IR::SyscallFlags::NOSYNCSTATEONENTRY: *out << "No Sync State on Entry"; break;
|
||||
case FEXCore::IR::SyscallFlags::NORETURN: *out << "No Return"; break;
|
||||
case FEXCore::IR::SyscallFlags::NOSIDEEFFECTS: *out << "No Side Effects"; break;
|
||||
default: *out << "<Unknown Round Type>"; break;
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::NamedVectorConstant Arg) {
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, NamedVectorConstant Arg) {
|
||||
*out << [Arg] {
|
||||
// clang-format off
|
||||
switch (Arg) {
|
||||
@@ -186,6 +177,22 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::N
|
||||
return "movmskps_shift";
|
||||
case NamedVectorConstant::NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE:
|
||||
return "aeskeygenassist_swizzle";
|
||||
case NamedVectorConstant::NAMED_VECTOR_BLENDPS_0110B:
|
||||
return "blendps_0110b";
|
||||
case NamedVectorConstant::NAMED_VECTOR_BLENDPS_0111B:
|
||||
return "blendps_0111b";
|
||||
case NamedVectorConstant::NAMED_VECTOR_BLENDPS_1001B:
|
||||
return "blendps_1001b";
|
||||
case NamedVectorConstant::NAMED_VECTOR_BLENDPS_1011B:
|
||||
return "blendps_1011b";
|
||||
case NamedVectorConstant::NAMED_VECTOR_BLENDPS_1101B:
|
||||
return "blendps_1101b";
|
||||
case NamedVectorConstant::NAMED_VECTOR_BLENDPS_1110B:
|
||||
return "blendps_1110b";
|
||||
case NamedVectorConstant::NAMED_VECTOR_MOVMASKB:
|
||||
return "movmaskb";
|
||||
case NamedVectorConstant::NAMED_VECTOR_MOVMASKB_UPPER:
|
||||
return "movmaskb_upper";
|
||||
case NamedVectorConstant::NAMED_VECTOR_ZERO:
|
||||
return "vectorzero";
|
||||
case NamedVectorConstant::NAMED_VECTOR_X87_ONE:
|
||||
@@ -216,9 +223,20 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::N
|
||||
return "cvtmax_i32";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_I64:
|
||||
return "cvtmax_i64";
|
||||
default:
|
||||
return "<Unknown Named Vector Constant>";
|
||||
case NamedVectorConstant::NAMED_VECTOR_F80_SIGN_MASK:
|
||||
return "f80_sign_mask";
|
||||
case NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K0:
|
||||
return "sha1rnds_k0";
|
||||
case NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K1:
|
||||
return "sha1rnds_k1";
|
||||
case NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K2:
|
||||
return "sha1rnds_k2";
|
||||
case NamedVectorConstant::NAMED_VECTOR_SHA1RNDS_K3:
|
||||
return "sha1rnds_k3";
|
||||
case NamedVectorConstant::NAMED_VECTOR_MAX:
|
||||
return "<Programming Error: Printing MAX value>";
|
||||
}
|
||||
return "<Unknown Named Vector Constant>";
|
||||
// clang-format on
|
||||
}();
|
||||
}
|
||||
@@ -241,36 +259,43 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, IndexNamedVect
|
||||
return "dppd_mask";
|
||||
case IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PBLENDW:
|
||||
return "pblendw";
|
||||
default:
|
||||
return "<Unknown Indexed Named Vector Constant>";
|
||||
case INDEXED_NAMED_VECTOR_MAX:
|
||||
return "<Programming Error: Printing MAX value>";
|
||||
}
|
||||
return "<Unknown Indexed Named Vector Constant>";
|
||||
// clang-format on
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::OpSize Arg) {
|
||||
switch (Arg) {
|
||||
case OpSize::i8Bit: *out << "i8"; break;
|
||||
case OpSize::i16Bit: *out << "i16"; break;
|
||||
case OpSize::i32Bit: *out << "i32"; break;
|
||||
case OpSize::i64Bit: *out << "i64"; break;
|
||||
case OpSize::i128Bit: *out << "i128"; break;
|
||||
case OpSize::i256Bit: *out << "i256"; break;
|
||||
case OpSize::f80Bit: *out << "f80"; break;
|
||||
default: *out << "<Unknown OpSize Type>"; break;
|
||||
}
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, OpSize Arg) {
|
||||
*out << [Arg] {
|
||||
switch (Arg) {
|
||||
case OpSize::iUnsized: return "Unsized";
|
||||
case OpSize::i8Bit: return "i8";
|
||||
case OpSize::i16Bit: return "i16";
|
||||
case OpSize::i32Bit: return "i32";
|
||||
case OpSize::i64Bit: return "i64";
|
||||
case OpSize::f80Bit: return "f80";
|
||||
case OpSize::i128Bit: return "i128";
|
||||
case OpSize::i256Bit: return "i256";
|
||||
case OpSize::iInvalid: return "Invalid";
|
||||
}
|
||||
return "<Unknown OpSize Type>";
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::FloatCompareOp Arg) {
|
||||
switch (Arg) {
|
||||
case FloatCompareOp::EQ: *out << "FEQ"; break;
|
||||
case FloatCompareOp::LT: *out << "FLT"; break;
|
||||
case FloatCompareOp::LE: *out << "FLE"; break;
|
||||
case FloatCompareOp::UNO: *out << "UNO"; break;
|
||||
case FloatCompareOp::NEQ: *out << "NEQ"; break;
|
||||
case FloatCompareOp::ORD: *out << "ORD"; break;
|
||||
default: *out << "<Unknown OpSize Type>"; break;
|
||||
}
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, FloatCompareOp Arg) {
|
||||
*out << [Arg] {
|
||||
switch (Arg) {
|
||||
case FloatCompareOp::EQ: return "FEQ";
|
||||
case FloatCompareOp::LT: return "FLT";
|
||||
case FloatCompareOp::LE: return "FLE";
|
||||
case FloatCompareOp::UNO: return "UNO";
|
||||
case FloatCompareOp::NEQ: return "NEQ";
|
||||
case FloatCompareOp::ORD: return "ORD";
|
||||
}
|
||||
return "<Unknown FloatCompareOp Type>";
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::BreakDefinition Arg) {
|
||||
@@ -280,23 +305,28 @@ static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::B
|
||||
*out << static_cast<uint32_t>(Arg.si_code) << "}";
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::ShiftType Arg) {
|
||||
switch (Arg) {
|
||||
case ShiftType::LSL: *out << "LSL"; break;
|
||||
case ShiftType::LSR: *out << "LSR"; break;
|
||||
case ShiftType::ASR: *out << "ASR"; break;
|
||||
case ShiftType::ROR: *out << "ROR"; break;
|
||||
default: *out << "<Unknown Shift Type>"; break;
|
||||
}
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, ShiftType Arg) {
|
||||
*out << [Arg] {
|
||||
switch (Arg) {
|
||||
case ShiftType::LSL: return "LSL";
|
||||
case ShiftType::LSR: return "LSR";
|
||||
case ShiftType::ASR: return "ASR";
|
||||
case ShiftType::ROR: return "ROR";
|
||||
}
|
||||
return "<Unknown Shift Type>";
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, FEXCore::IR::BranchHint Arg) {
|
||||
switch (Arg) {
|
||||
case BranchHint::None: *out << "None"; break;
|
||||
case BranchHint::Call: *out << "Call"; break;
|
||||
case BranchHint::Return: *out << "Return"; break;
|
||||
default: *out << "<Unknown Branch Hint>"; break;
|
||||
}
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, BranchHint Arg) {
|
||||
*out << [Arg] {
|
||||
switch (Arg) {
|
||||
case BranchHint::None: return "None";
|
||||
case BranchHint::Call: return "Call";
|
||||
case BranchHint::Return: return "Return";
|
||||
case BranchHint::CheckTF: return "CheckTF";
|
||||
}
|
||||
return "<Unknown Branch Hint>";
|
||||
}();
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream* out, const IRListView*, const std::array<uint8_t, 0x10>& Arg) {
|
||||
@@ -323,8 +353,7 @@ void Dump(fextl::stringstream* out, const IRListView* IR) {
|
||||
auto BlockIROp = BlockHeader->C<FEXCore::IR::IROp_CodeBlock>();
|
||||
|
||||
AddIndent();
|
||||
*out << "(%" << IR->GetID(BlockNode) << ") "
|
||||
<< "CodeBlock ";
|
||||
*out << "(%" << IR->GetID(BlockNode) << ") " << "CodeBlock ";
|
||||
|
||||
*out << "%" << BlockIROp->Begin.ID() << ", ";
|
||||
*out << "%" << BlockIROp->Last.ID() << std::endl;
|
||||
@@ -353,17 +382,17 @@ void Dump(fextl::stringstream* out, const IRListView* IR) {
|
||||
|
||||
auto PhyReg = PhysicalRegister(CodeNode);
|
||||
if (!PhyReg.IsInvalid()) {
|
||||
switch (PhyReg.Class) {
|
||||
case FEXCore::IR::GPRClass.Val: *out << "(r"; break;
|
||||
case FEXCore::IR::GPRFixedClass.Val: *out << "(R"; break;
|
||||
case FEXCore::IR::FPRClass.Val: *out << "(v"; break;
|
||||
case FEXCore::IR::FPRFixedClass.Val: *out << "(V"; break;
|
||||
case FEXCore::IR::ComplexClass.Val: *out << "(complex"; break;
|
||||
case FEXCore::IR::InvalidClass.Val: *out << "(invalid"; break;
|
||||
switch (PhyReg.AsRegClass()) {
|
||||
case RegClass::GPR: *out << "(r"; break;
|
||||
case RegClass::GPRFixed: *out << "(R"; break;
|
||||
case RegClass::FPR: *out << "(v"; break;
|
||||
case RegClass::FPRFixed: *out << "(V"; break;
|
||||
case RegClass::Complex: *out << "(complex"; break;
|
||||
case RegClass::Invalid: *out << "(invalid"; break;
|
||||
default: *out << "(unknown"; break;
|
||||
}
|
||||
if (PhyReg.Class != FEXCore::IR::InvalidClass.Val) {
|
||||
*out << std::dec << (uint32_t)PhyReg.Reg << ")";
|
||||
if (PhyReg.AsRegClass() != RegClass::Invalid) {
|
||||
*out << std::dec << uint32_t(PhyReg.Reg) << ")";
|
||||
} else {
|
||||
*out << ")";
|
||||
}
|
||||
|
||||
@@ -33,14 +33,14 @@ bool IsBlockExit(FEXCore::IR::IROps Op) {
|
||||
}
|
||||
}
|
||||
|
||||
FEXCore::IR::RegisterClassType IREmitter::WalkFindRegClass(Ref Node) {
|
||||
RegClass IREmitter::WalkFindRegClass(Ref Node) {
|
||||
auto Class = GetOpRegClass(Node);
|
||||
switch (Class) {
|
||||
case GPRClass:
|
||||
case FPRClass:
|
||||
case GPRFixedClass:
|
||||
case FPRFixedClass:
|
||||
case InvalidClass: return Class;
|
||||
case RegClass::GPR:
|
||||
case RegClass::FPR:
|
||||
case RegClass::GPRFixed:
|
||||
case RegClass::FPRFixed:
|
||||
case RegClass::Invalid: return Class;
|
||||
default: break;
|
||||
}
|
||||
|
||||
@@ -82,7 +82,7 @@ FEXCore::IR::RegisterClassType IREmitter::WalkFindRegClass(Ref Node) {
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled op type: {} {} in argument class validation", ToUnderlying(IROp->Op), GetOpName(Node)); break;
|
||||
}
|
||||
return InvalidClass;
|
||||
return RegClass::Invalid;
|
||||
}
|
||||
|
||||
void IREmitter::ResetWorkingList() {
|
||||
|
||||
@@ -46,12 +46,12 @@ public:
|
||||
*
|
||||
* @{ */
|
||||
|
||||
FEXCore::IR::RegisterClassType WalkFindRegClass(Ref Node);
|
||||
RegClass WalkFindRegClass(Ref Node);
|
||||
|
||||
// These inlining helpers are used by IRDefines.inc so define first.
|
||||
Ref InlineMem(OpSize Size, Ref Offset, MemOffsetType OffsetType, uint8_t& OffsetScale, bool TSO = false) {
|
||||
uint64_t Imm {};
|
||||
if (OffsetType != MEM_OFFSET_SXTX || !IsValueConstant(WrapNode(Offset), &Imm)) {
|
||||
if (OffsetType != MemOffsetType::SXTX || !IsValueConstant(WrapNode(Offset), &Imm)) {
|
||||
return Offset;
|
||||
}
|
||||
|
||||
@@ -108,36 +108,86 @@ public:
|
||||
IRPair<IROp_Jump> _Jump() {
|
||||
return _Jump(InvalidNode);
|
||||
}
|
||||
IRPair<IROp_CondJump> _CondJump(Ref ssa0, CondClassType cond = {COND_NEQ}) {
|
||||
IRPair<IROp_CondJump> _CondJump(Ref ssa0, CondClass cond = CondClass::NEQ) {
|
||||
return _CondJump(ssa0, _Constant(0), InvalidNode, InvalidNode, cond, GetOpSize(ssa0));
|
||||
}
|
||||
IRPair<IROp_CondJump> _CondJump(Ref ssa0, Ref ssa1, Ref ssa2, CondClassType cond = {COND_NEQ}) {
|
||||
IRPair<IROp_CondJump> _CondJump(Ref ssa0, Ref ssa1, Ref ssa2, CondClass cond = CondClass::NEQ) {
|
||||
return _CondJump(ssa0, _Constant(0), ssa1, ssa2, cond, GetOpSize(ssa0));
|
||||
}
|
||||
// TODO: Work to remove this implicit sized Select implementation.
|
||||
IRPair<IROp_Select> _Select(uint8_t Cond, Ref ssa0, Ref ssa1, Ref ssa2, Ref ssa3, IR::OpSize CompareSize = OpSize::iUnsized) {
|
||||
if (CompareSize == OpSize::iUnsized) {
|
||||
CompareSize = std::max(OpSize::i32Bit, std::max(GetOpSize(ssa0), GetOpSize(ssa1)));
|
||||
}
|
||||
|
||||
return _Select(std::max(OpSize::i32Bit, std::max(GetOpSize(ssa2), GetOpSize(ssa3))), CompareSize, CondClassType {Cond}, ssa0, ssa1, ssa2, ssa3);
|
||||
IRPair<IROp_LoadContext> _LoadContextGPR(OpSize ByteSize, uint32_t Offset) {
|
||||
return _LoadContext(ByteSize, RegClass::GPR, Offset);
|
||||
}
|
||||
IRPair<IROp_LoadMem> _LoadMem(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, Ref ssa0, IR::OpSize Align = OpSize::i8Bit) {
|
||||
return _LoadMem(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
IRPair<IROp_LoadContext> _LoadContextFPR(OpSize ByteSize, uint32_t Offset) {
|
||||
return _LoadContext(ByteSize, RegClass::FPR, Offset);
|
||||
}
|
||||
IRPair<IROp_StoreMem> _StoreMem(FEXCore::IR::RegisterClassType Class, IR::OpSize Size, Ref Addr, Ref Value, IR::OpSize Align = OpSize::i8Bit) {
|
||||
return _StoreMem(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
IRPair<IROp_StoreContext> _StoreContextGPR(OpSize ByteSize, Ref Value, uint32_t Offset) {
|
||||
return _StoreContext(ByteSize, RegClass::GPR, Value, Offset);
|
||||
}
|
||||
IRPair<IROp_StoreContext> _StoreContextFPR(OpSize ByteSize, Ref Value, uint32_t Offset) {
|
||||
return _StoreContext(ByteSize, RegClass::FPR, Value, Offset);
|
||||
}
|
||||
|
||||
IRPair<IROp_Select> Select01(FEXCore::IR::OpSize CompareSize, CondClassType Cond, OrderedNode* Cmp1, OrderedNode* Cmp2) {
|
||||
IRPair<IROp_LoadContextIndexed> _LoadContextGPRIndexed(Ref Index, OpSize ByteSize, uint32_t BaseOffset, uint32_t Stride) {
|
||||
return _LoadContextIndexed(Index, ByteSize, BaseOffset, Stride, RegClass::GPR);
|
||||
}
|
||||
IRPair<IROp_LoadContextIndexed> _LoadContextFPRIndexed(Ref Index, OpSize ByteSize, uint32_t BaseOffset, uint32_t Stride) {
|
||||
return _LoadContextIndexed(Index, ByteSize, BaseOffset, Stride, RegClass::FPR);
|
||||
}
|
||||
IRPair<IROp_StoreContextIndexed> _StoreContextGPRIndexed(Ref Value, Ref Index, OpSize ByteSize, uint32_t BaseOffset, uint32_t Stride) {
|
||||
return _StoreContextIndexed(Value, Index, ByteSize, BaseOffset, Stride, RegClass::GPR);
|
||||
}
|
||||
IRPair<IROp_StoreContextIndexed> _StoreContextFPRIndexed(Ref Value, Ref Index, OpSize ByteSize, uint32_t BaseOffset, uint32_t Stride) {
|
||||
return _StoreContextIndexed(Value, Index, ByteSize, BaseOffset, Stride, RegClass::FPR);
|
||||
}
|
||||
|
||||
IRPair<IROp_LoadMem> _LoadMem(RegClass Class, OpSize Size, Ref ssa0, OpSize Align = OpSize::i8Bit) {
|
||||
return _LoadMem(Class, Size, ssa0, Invalid(), Align, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
IRPair<IROp_LoadMem> _LoadMemGPR(OpSize Size, Ref ssa0, OpSize Align = OpSize::i8Bit) {
|
||||
return _LoadMem(RegClass::GPR, Size, ssa0, Invalid(), Align, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
IRPair<IROp_LoadMem> _LoadMemGPR(OpSize Size, Ref Addr, Ref Offset, OpSize Align, MemOffsetType OffsetType, uint8_t OffsetScale) {
|
||||
return _LoadMem(RegClass::GPR, Size, Addr, Offset, Align, OffsetType, OffsetScale);
|
||||
}
|
||||
IRPair<IROp_LoadMem> _LoadMemFPR(OpSize Size, Ref ssa0, OpSize Align = OpSize::i8Bit) {
|
||||
return _LoadMem(RegClass::FPR, Size, ssa0, Invalid(), Align, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
IRPair<IROp_LoadMem> _LoadMemFPR(OpSize Size, Ref Addr, Ref Offset, OpSize Align, MemOffsetType OffsetType, uint8_t OffsetScale) {
|
||||
return _LoadMem(RegClass::FPR, Size, Addr, Offset, Align, OffsetType, OffsetScale);
|
||||
}
|
||||
IRPair<IROp_StoreMem> _StoreMem(RegClass Class, OpSize Size, Ref Addr, Ref Value, OpSize Align = OpSize::i8Bit) {
|
||||
return _StoreMem(Class, Size, Value, Addr, Invalid(), Align, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
IRPair<IROp_StoreMem> _StoreMemGPR(OpSize Size, Ref Addr, Ref Value, OpSize Align = OpSize::i8Bit) {
|
||||
return _StoreMem(RegClass::GPR, Size, Value, Addr, Invalid(), Align, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
IRPair<IROp_StoreMem> _StoreMemGPR(OpSize Size, Ref Value, Ref Addr, Ref Offset, OpSize Align, MemOffsetType OffsetType, uint8_t OffsetScale) {
|
||||
return _StoreMem(RegClass::GPR, Size, Value, Addr, Offset, Align, OffsetType, OffsetScale);
|
||||
}
|
||||
IRPair<IROp_StoreMem> _StoreMemFPR(OpSize Size, Ref Addr, Ref Value, OpSize Align = OpSize::i8Bit) {
|
||||
return _StoreMem(RegClass::FPR, Size, Value, Addr, Invalid(), Align, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
IRPair<IROp_StoreMem> _StoreMemFPR(OpSize Size, Ref Value, Ref Addr, Ref Offset, OpSize Align, MemOffsetType OffsetType, uint8_t OffsetScale) {
|
||||
return _StoreMem(RegClass::FPR, Size, Value, Addr, Offset, Align, OffsetType, OffsetScale);
|
||||
}
|
||||
|
||||
IRPair<IROp_StoreMemPair> _StoreMemPairGPR(OpSize Size, Ref Value1, Ref Value2, Ref Addr, uint32_t Offset) {
|
||||
return _StoreMemPair(RegClass::GPR, Size, Value1, Value2, Addr, Offset);
|
||||
}
|
||||
IRPair<IROp_StoreMemPair> _StoreMemPairFPR(OpSize Size, Ref Value1, Ref Value2, Ref Addr, uint32_t Offset) {
|
||||
return _StoreMemPair(RegClass::FPR, Size, Value1, Value2, Addr, Offset);
|
||||
}
|
||||
|
||||
IRPair<IROp_Select> Select01(FEXCore::IR::OpSize CompareSize, CondClass Cond, OrderedNode* Cmp1, OrderedNode* Cmp2) {
|
||||
return _Select(OpSize::i64Bit, CompareSize, Cond, Cmp1, Cmp2, _InlineConstant(1), _InlineConstant(0));
|
||||
}
|
||||
|
||||
IRPair<IROp_Select> To01(FEXCore::IR::OpSize CompareSize, OrderedNode* Cmp1) {
|
||||
return Select01(CompareSize, CondClassType {COND_NEQ}, Cmp1, Constant(0));
|
||||
return Select01(CompareSize, CondClass::NEQ, Cmp1, Constant(0));
|
||||
}
|
||||
|
||||
IRPair<IROp_NZCVSelect> _NZCVSelect01(CondClassType Cond) {
|
||||
IRPair<IROp_NZCVSelect> _NZCVSelect01(CondClass Cond) {
|
||||
return _NZCVSelect(OpSize::i64Bit, Cond, _InlineConstant(1), _InlineConstant(0));
|
||||
}
|
||||
|
||||
@@ -250,7 +300,7 @@ public:
|
||||
}
|
||||
|
||||
/** @} */
|
||||
FEXCore::IR::RegisterClassType WalkFindRegClass(OrderedNodeWrapper ssa) {
|
||||
RegClass WalkFindRegClass(OrderedNodeWrapper ssa) {
|
||||
Ref RealNode = ssa.GetNode(DualListData.ListBegin());
|
||||
return WalkFindRegClass(RealNode);
|
||||
}
|
||||
|
||||
@@ -39,7 +39,7 @@ public:
|
||||
}
|
||||
|
||||
protected:
|
||||
PassManager* Manager;
|
||||
PassManager* Manager {};
|
||||
};
|
||||
|
||||
class PassManager final {
|
||||
|
||||
@@ -93,11 +93,11 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
// After RA, the destination needs to be assigned a register and class
|
||||
auto PhyReg = PhysicalRegister(CodeNode);
|
||||
|
||||
FEXCore::IR::RegisterClassType ExpectedClass = IR::GetRegClass(IROp->Op);
|
||||
FEXCore::IR::RegisterClassType AssignedClass = FEXCore::IR::RegisterClassType {PhyReg.Class};
|
||||
const auto ExpectedClass = IR::GetRegClass(IROp->Op);
|
||||
const auto AssignedClass = PhyReg.AsRegClass();
|
||||
|
||||
// If no register class was assigned
|
||||
if (AssignedClass == IR::InvalidClass) {
|
||||
if (AssignedClass == IR::RegClass::Invalid) {
|
||||
HadError |= true;
|
||||
Errors << "%" << ID << ": Had destination but with no register class assigned" << std::endl;
|
||||
}
|
||||
@@ -109,10 +109,10 @@ void IRValidation::Run(IREmitter* IREmit) {
|
||||
}
|
||||
|
||||
// Assigned class wasn't the expected class and it is a non-complex op
|
||||
if (AssignedClass != ExpectedClass && ExpectedClass != IR::ComplexClass) {
|
||||
if (AssignedClass != ExpectedClass && ExpectedClass != IR::RegClass::Complex) {
|
||||
HadWarning |= true;
|
||||
Warnings << "%" << ID << ": Destination had register class " << AssignedClass.Val << " When register class "
|
||||
<< ExpectedClass.Val << " Was expected" << std::endl;
|
||||
Warnings << "%" << ID << ": Destination had register class " << uint32_t(AssignedClass) << " When register class "
|
||||
<< uint32_t(ExpectedClass) << " Was expected" << std::endl;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2,22 +2,21 @@
|
||||
/*
|
||||
$info$
|
||||
tags: ir|opts
|
||||
desc: This is not used right now, possibly broken
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "FEXCore/Core/X86Enums.h"
|
||||
#include "FEXCore/Utils/CompilerDefs.h"
|
||||
#include "FEXCore/Utils/MathUtils.h"
|
||||
#include "FEXCore/fextl/deque.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
#include <FEXCore/Core/X86Enums.h>
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/fextl/deque.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
// Flag bit flags
|
||||
#define FLAG_V (1U << 0)
|
||||
#define FLAG_C (1U << 1)
|
||||
@@ -62,36 +61,36 @@ struct FlagInfo {
|
||||
return {.Raw = R};
|
||||
}
|
||||
|
||||
bool Trivial() {
|
||||
bool Trivial() const {
|
||||
return Raw == 0;
|
||||
}
|
||||
|
||||
unsigned Read() {
|
||||
unsigned Read() const {
|
||||
return Bits(0, 8);
|
||||
}
|
||||
|
||||
unsigned Write() {
|
||||
unsigned Write() const {
|
||||
return Bits(8, 8);
|
||||
}
|
||||
|
||||
bool CanEliminate() {
|
||||
bool CanEliminate() const {
|
||||
return Bits(16, 1);
|
||||
}
|
||||
|
||||
bool Special() {
|
||||
bool Special() const {
|
||||
return Bits(63, 1);
|
||||
}
|
||||
|
||||
IROps Replacement() {
|
||||
IROps Replacement() const {
|
||||
return (IROps)Bits(32, 16);
|
||||
}
|
||||
|
||||
IROps ReplacementNoWrite() {
|
||||
IROps ReplacementNoWrite() const {
|
||||
return (IROps)Bits(48, 16);
|
||||
}
|
||||
|
||||
private:
|
||||
unsigned Bits(unsigned Start, unsigned Count) {
|
||||
unsigned Bits(unsigned Start, unsigned Count) const {
|
||||
return (Raw >> Start) & ((1u << Count) - 1);
|
||||
}
|
||||
};
|
||||
@@ -154,45 +153,44 @@ public:
|
||||
|
||||
private:
|
||||
FlagInfo Classify(IROp_Header* Node);
|
||||
unsigned FlagForReg(unsigned Reg);
|
||||
unsigned FlagsForCondClassType(CondClassType Cond);
|
||||
unsigned FlagsForCondClassType(CondClass Cond);
|
||||
bool EliminateDeadCode(IREmitter* IREmit, Ref CodeNode, IROp_Header* IROp);
|
||||
void FoldBranch(IREmitter* IREmit, IRListView& CurrentIR, IROp_CondJump* Op, Ref CodeNode);
|
||||
CondClassType X86ToArmFloatCond(CondClassType X86);
|
||||
CondClass X86ToArmFloatCond(CondClass X86);
|
||||
bool ProcessBlock(IREmitter* IREmit, IRListView& CurrentIR, Ref Block, ControlFlowGraph& CFG);
|
||||
void OptimizeParity(IREmitter* IREmit, IRListView& CurrentIR, ControlFlowGraph& CFG);
|
||||
};
|
||||
|
||||
unsigned DeadFlagCalculationEliminination::FlagsForCondClassType(CondClassType Cond) {
|
||||
unsigned DeadFlagCalculationEliminination::FlagsForCondClassType(CondClass Cond) {
|
||||
switch (Cond) {
|
||||
case COND_AL: return 0;
|
||||
case CondClass::AL: return 0;
|
||||
|
||||
case COND_MI:
|
||||
case COND_PL: return FLAG_N;
|
||||
case CondClass::MI:
|
||||
case CondClass::PL: return FLAG_N;
|
||||
|
||||
case COND_EQ:
|
||||
case COND_NEQ: return FLAG_Z;
|
||||
case CondClass::EQ:
|
||||
case CondClass::NEQ: return FLAG_Z;
|
||||
|
||||
case COND_UGE:
|
||||
case COND_ULT: return FLAG_C;
|
||||
case CondClass::UGE:
|
||||
case CondClass::ULT: return FLAG_C;
|
||||
|
||||
case COND_VS:
|
||||
case COND_VC:
|
||||
case COND_FU:
|
||||
case COND_FNU: return FLAG_V;
|
||||
case CondClass::VS:
|
||||
case CondClass::VC:
|
||||
case CondClass::FU:
|
||||
case CondClass::FNU: return FLAG_V;
|
||||
|
||||
case COND_UGT:
|
||||
case COND_ULE: return FLAG_Z | FLAG_C;
|
||||
case CondClass::UGT:
|
||||
case CondClass::ULE: return FLAG_Z | FLAG_C;
|
||||
|
||||
case COND_SGE:
|
||||
case COND_SLT:
|
||||
case COND_FLU:
|
||||
case COND_FGE: return FLAG_N | FLAG_V;
|
||||
case CondClass::SGE:
|
||||
case CondClass::SLT:
|
||||
case CondClass::FLU:
|
||||
case CondClass::FGE: return FLAG_N | FLAG_V;
|
||||
|
||||
case COND_SGT:
|
||||
case COND_SLE:
|
||||
case COND_FLEU:
|
||||
case COND_FGT: return FLAG_N | FLAG_Z | FLAG_V;
|
||||
case CondClass::SGT:
|
||||
case CondClass::SLE:
|
||||
case CondClass::FLEU:
|
||||
case CondClass::FGT: return FLAG_N | FLAG_Z | FLAG_V;
|
||||
|
||||
default: LOGMAN_THROW_A_FMT(false, "unknown cond class type"); return FLAG_NZCV;
|
||||
}
|
||||
@@ -456,7 +454,7 @@ bool DeadFlagCalculationEliminination::EliminateDeadCode(IREmitter* IREmit, Ref
|
||||
return true;
|
||||
}
|
||||
|
||||
CondClassType DeadFlagCalculationEliminination::X86ToArmFloatCond(CondClassType X86) {
|
||||
CondClass DeadFlagCalculationEliminination::X86ToArmFloatCond(CondClass X86) {
|
||||
// Table of x86 condition codes that map to arm64 condition codes, in the
|
||||
// sense that fcmp+axflag+branch(x86) is equivalent to fcmp+branch(arm).
|
||||
//
|
||||
@@ -465,12 +463,12 @@ CondClassType DeadFlagCalculationEliminination::X86ToArmFloatCond(CondClassType
|
||||
//
|
||||
// SF/OF conditions are trivial and therefore shouldn't actually be generated
|
||||
switch (X86) {
|
||||
case COND_UGE /* A */: return {COND_FGE} /* GE */;
|
||||
case COND_UGT /* AE */: return {COND_FGT} /* GT */;
|
||||
case COND_ULT /* B */: return {COND_SLT} /* LT */;
|
||||
case COND_ULE /* BE */: return {COND_SLE} /* LE */;
|
||||
case COND_SLE /* LE */: return {COND_SLE} /* LE */;
|
||||
default: return {COND_AL};
|
||||
case CondClass::UGE /* A */: return CondClass::FGE /* GE */;
|
||||
case CondClass::UGT /* AE */: return CondClass::FGT /* GT */;
|
||||
case CondClass::ULT /* B */: return CondClass::SLT /* LT */;
|
||||
case CondClass::ULE /* BE */: return CondClass::SLE /* LE */;
|
||||
case CondClass::SLE /* LE */: return CondClass::SLE /* LE */;
|
||||
default: return CondClass::AL;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -485,8 +483,8 @@ void DeadFlagCalculationEliminination::FoldBranch(IREmitter* IREmit, IRListView&
|
||||
auto Prev = CurrentIR.GetOp<IR::IROp_Header>(PrevWrap);
|
||||
if (Prev->Op == OP_AXFLAG) {
|
||||
// Pattern match a branch fed by AXFLAG.
|
||||
CondClassType ArmCond = X86ToArmFloatCond(Op->Cond);
|
||||
if (ArmCond == COND_AL) {
|
||||
CondClass ArmCond = X86ToArmFloatCond(Op->Cond);
|
||||
if (ArmCond == CondClass::AL) {
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -495,7 +493,7 @@ void DeadFlagCalculationEliminination::FoldBranch(IREmitter* IREmit, IRListView&
|
||||
// Pattern match a branch fed by a compare. We could also handle bit tests
|
||||
// here, but tbz/tbnz has a limited offset range which we don't have a way to
|
||||
// deal with yet. Let's hope that's not a big deal.
|
||||
if (!(Op->Cond == COND_NEQ || Op->Cond == COND_EQ) || (Prev->Size < OpSize::i32Bit)) {
|
||||
if (!(Op->Cond == CondClass::NEQ || Op->Cond == CondClass::EQ) || (Prev->Size < OpSize::i32Bit)) {
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -629,9 +627,9 @@ void DeadFlagCalculationEliminination::OptimizeParity(IREmitter* IREmit, IRListV
|
||||
}
|
||||
|
||||
for (auto [Block, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
auto ID = BlockHeader->C<IROp_CodeBlock>()->ID;
|
||||
const auto ID = BlockHeader->C<IROp_CodeBlock>()->ID;
|
||||
const auto& Predecessors = CFG.Get(ID)->Predecessors;
|
||||
bool Full = false;
|
||||
auto Predecessors = CFG.Get(ID)->Predecessors;
|
||||
|
||||
if (Predecessors.empty()) {
|
||||
// Conservatively assume there was full parity before the start block
|
||||
|
||||
@@ -12,6 +12,7 @@ $end_info$
|
||||
#include "Interface/IR/Passes.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
@@ -22,7 +23,7 @@ using namespace FEXCore;
|
||||
|
||||
namespace FEXCore::IR {
|
||||
namespace {
|
||||
struct RegisterClass {
|
||||
struct RegisterClassData {
|
||||
uint32_t Available;
|
||||
uint32_t Count;
|
||||
|
||||
@@ -32,9 +33,9 @@ namespace {
|
||||
Ref RegToSSA[32];
|
||||
};
|
||||
|
||||
IR::RegisterClassType GetRegClassFromNode(IR::IRListView* IR, IR::IROp_Header* IROp) {
|
||||
IR::RegisterClassType Class = IR::GetRegClass(IROp->Op);
|
||||
if (Class != IR::ComplexClass) {
|
||||
IR::RegClass GetRegClassFromNode(IR::IRListView* IR, IR::IROp_Header* IROp) {
|
||||
const auto Class = IR::GetRegClass(IROp->Op);
|
||||
if (Class != IR::RegClass::Complex) {
|
||||
return Class;
|
||||
}
|
||||
|
||||
@@ -46,7 +47,7 @@ namespace {
|
||||
case IR::OP_LOADMEM:
|
||||
case IR::OP_LOADMEMTSO: return IROp->C<IR::IROp_LoadMem>()->Class;
|
||||
case IR::OP_FILLREGISTER: return IROp->C<IR::IROp_FillRegister>()->Class;
|
||||
default: return IR::InvalidClass;
|
||||
default: return IR::RegClass::Invalid;
|
||||
}
|
||||
};
|
||||
} // Anonymous namespace
|
||||
@@ -56,15 +57,15 @@ public:
|
||||
explicit ConstrainedRAPass(const FEXCore::CPUIDEmu* CPUID)
|
||||
: CPUID {CPUID} {}
|
||||
void Run(IREmitter* IREmit) override;
|
||||
void AddRegisters(IR::RegisterClassType Class, uint32_t RegisterCount) override;
|
||||
void AddRegisters(IR::RegClass Class, uint32_t RegisterCount) override;
|
||||
bool TryPostRAMerge(Ref LastNode, Ref CodeNode, IROp_Header* IROp);
|
||||
|
||||
private:
|
||||
RegisterClass Classes[IR::NumClasses];
|
||||
RegisterClassData Classes[IR::NumClasses];
|
||||
|
||||
IREmitter* IREmit;
|
||||
IRListView* IR;
|
||||
const FEXCore::CPUIDEmu* CPUID;
|
||||
IREmitter* IREmit {};
|
||||
IRListView* IR {};
|
||||
const FEXCore::CPUIDEmu* CPUID {};
|
||||
|
||||
// Map of nodes to their preferred register, to coalesce load/store reg.
|
||||
fextl::vector<PhysicalRegister> PreferredReg;
|
||||
@@ -82,7 +83,7 @@ private:
|
||||
fextl::vector<bool> Seen;
|
||||
|
||||
// SourcesNextUses is read backwards, this tracks the index
|
||||
int64_t SourceIndex;
|
||||
int64_t SourceIndex {};
|
||||
|
||||
bool Rematerializable(IROp_Header* IROp) {
|
||||
return IROp->Op == OP_CONSTANT;
|
||||
@@ -101,7 +102,7 @@ private:
|
||||
uint32_t SlotPlusOne = SpillSlots[IR->GetID(Node).Value];
|
||||
LOGMAN_THROW_A_FMT(SlotPlusOne >= 1, "Node must have been spilled");
|
||||
|
||||
RegisterClassType RegClass = GetRegClassFromNode(IR, IROp);
|
||||
const auto RegClass = GetRegClassFromNode(IR, IROp);
|
||||
return IREmit->_FillRegister(IROp->Size, IROp->ElementSize, SlotPlusOne - 1, RegClass);
|
||||
};
|
||||
|
||||
@@ -109,7 +110,7 @@ private:
|
||||
// block, so we don't need to size the block up-front.
|
||||
fextl::vector<uint32_t> NextUses;
|
||||
|
||||
bool AnySpilled;
|
||||
bool AnySpilled {};
|
||||
|
||||
bool IsValidArg(OrderedNodeWrapper Arg) {
|
||||
if (Arg.IsInvalid()) {
|
||||
@@ -120,7 +121,7 @@ private:
|
||||
return Op != OP_INLINECONSTANT && Op != OP_INLINEENTRYPOINTOFFSET;
|
||||
};
|
||||
|
||||
RegisterClass* GetClass(PhysicalRegister Reg) {
|
||||
RegisterClassData* GetClass(PhysicalRegister Reg) {
|
||||
return &Classes[Reg.Class];
|
||||
};
|
||||
|
||||
@@ -133,13 +134,13 @@ private:
|
||||
LOGMAN_THROW_A_FMT(ID < SSAToReg.size(), "Only old nodes looked up");
|
||||
|
||||
PhysicalRegister Reg = SSAToReg[ID];
|
||||
RegisterClass* Class = GetClass(Reg);
|
||||
RegisterClassData* Class = GetClass(Reg);
|
||||
|
||||
return (Class->Available & GetRegBits(Reg)) == 0 && Class->RegToSSA[Reg.Reg] == Node;
|
||||
};
|
||||
|
||||
void FreeReg(PhysicalRegister Reg) {
|
||||
RegisterClass* Class = GetClass(Reg);
|
||||
RegisterClassData* Class = GetClass(Reg);
|
||||
uint32_t RegBits = GetRegBits(Reg);
|
||||
|
||||
LOGMAN_THROW_A_FMT(!(Class->Available & RegBits), "Register double-free");
|
||||
@@ -187,22 +188,22 @@ private:
|
||||
};
|
||||
|
||||
PhysicalRegister DecodeSRAReg(const IROp_Header* IROp, Ref Node) {
|
||||
uint8_t FlagOffset = Classes[GPRFixedClass.Val].Count - 2;
|
||||
uint8_t FlagOffset = Classes[FEXCore::ToUnderlying(RegClass::GPRFixed)].Count - 2;
|
||||
|
||||
if (IROp->Op == OP_STOREREGISTER) {
|
||||
return PhysicalRegister(Node);
|
||||
} else if (IROp->Op == OP_LOADPF || IROp->Op == OP_STOREPF) {
|
||||
return PhysicalRegister {GPRFixedClass, FlagOffset};
|
||||
return PhysicalRegister {RegClass::GPRFixed, FlagOffset};
|
||||
} else if (IROp->Op == OP_LOADAF || IROp->Op == OP_STOREAF) {
|
||||
return PhysicalRegister {GPRFixedClass, (uint8_t)(FlagOffset + 1)};
|
||||
return PhysicalRegister {RegClass::GPRFixed, uint8_t(FlagOffset + 1)};
|
||||
} else {
|
||||
const IROp_LoadRegister* Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
|
||||
LOGMAN_THROW_A_FMT(Op->Class == GPRClass || Op->Class == FPRClass, "SRA classes");
|
||||
if (Op->Class == FPRClass) {
|
||||
return PhysicalRegister {FPRFixedClass, (uint8_t)Op->Reg};
|
||||
LOGMAN_THROW_A_FMT(Op->Class == RegClass::GPR || Op->Class == RegClass::FPR, "SRA classes");
|
||||
if (Op->Class == RegClass::FPR) {
|
||||
return PhysicalRegister {RegClass::FPRFixed, uint8_t(Op->Reg)};
|
||||
} else {
|
||||
return PhysicalRegister {GPRFixedClass, (uint8_t)Op->Reg};
|
||||
return PhysicalRegister {RegClass::GPRFixed, uint8_t(Op->Reg)};
|
||||
}
|
||||
}
|
||||
};
|
||||
@@ -267,7 +268,7 @@ private:
|
||||
SourceIndex = SourcesNextUses.size();
|
||||
}
|
||||
|
||||
void SpillReg(RegisterClass* Class, IROp_CodeBlock* Block, IROp_Header* Exclude) {
|
||||
void SpillReg(RegisterClassData* Class, IROp_CodeBlock* Block, IROp_Header* Exclude) {
|
||||
// We're about to use next-use information, so calculate it.
|
||||
if (!AnySpilled) {
|
||||
CalculateNextUses(Block, Exclude);
|
||||
@@ -318,7 +319,7 @@ private:
|
||||
// If we already spilled the Candidate, we don't need to spill again.
|
||||
// Similarly, if we can rematerialize the instruction, we don't spill it.
|
||||
if (!Spilled && Header->Op != OP_CONSTANT) {
|
||||
LOGMAN_THROW_A_FMT(Reg.Class == GetRegClassFromNode(IR, Header), "Consistent");
|
||||
LOGMAN_THROW_A_FMT(Reg.AsRegClass() == GetRegClassFromNode(IR, Header), "Consistent");
|
||||
|
||||
// SpillSlots allocation is deferred.
|
||||
if (SpillSlots.empty()) {
|
||||
@@ -329,7 +330,7 @@ private:
|
||||
uint32_t Slot = IR->GetHeader()->SpillSlots++;
|
||||
|
||||
// We must map here in case we're spilling something we shuffled.
|
||||
auto SpillOp = IREmit->_SpillRegister(OrderedNodeWrapper::FromImmediate(Reg.Raw), Slot, RegisterClassType {Reg.Class});
|
||||
auto SpillOp = IREmit->_SpillRegister(OrderedNodeWrapper::FromImmediate(Reg.Raw), Slot, Reg.AsRegClass());
|
||||
SpillOp.first->Header.Size = Header->Size;
|
||||
SpillOp.first->Header.ElementSize = Header->ElementSize;
|
||||
SpillSlots[Value] = Slot + 1;
|
||||
@@ -341,7 +342,7 @@ private:
|
||||
};
|
||||
|
||||
void RemapReg(Ref Node, PhysicalRegister Reg) {
|
||||
RegisterClass* Class = GetClass(Reg);
|
||||
RegisterClassData* Class = GetClass(Reg);
|
||||
Class->RegToSSA[Reg.Reg] = Node;
|
||||
|
||||
uint32_t Index = IR->GetID(Node).Value;
|
||||
@@ -352,7 +353,7 @@ private:
|
||||
|
||||
// Record a given assignment of register Reg to Node.
|
||||
void SetReg(Ref Node, PhysicalRegister Reg) {
|
||||
RegisterClass* Class = GetClass(Reg);
|
||||
RegisterClassData* Class = GetClass(Reg);
|
||||
uint32_t RegBits = GetRegBits(Reg);
|
||||
|
||||
LOGMAN_THROW_A_FMT((Class->Available & RegBits) == RegBits, "Precondition");
|
||||
@@ -370,7 +371,7 @@ private:
|
||||
// Prioritize preferred registers.
|
||||
if (Node < PreferredReg.size()) {
|
||||
if (PhysicalRegister Reg = PreferredReg[Node]; !Reg.IsInvalid()) {
|
||||
RegisterClass* Class = GetClass(Reg);
|
||||
RegisterClassData* Class = GetClass(Reg);
|
||||
uint32_t RegBits = GetRegBits(Reg);
|
||||
|
||||
if ((Class->Available & RegBits) == RegBits) {
|
||||
@@ -383,10 +384,10 @@ private:
|
||||
// Try to handle tied registers. This can fail, the JIT will insert moves.
|
||||
if (int TiedIdx = IR::TiedSource(IROp->Op); TiedIdx >= 0) {
|
||||
auto Reg = PhysicalRegister(IROp->Args[TiedIdx]);
|
||||
RegisterClass* Class = GetClass(Reg);
|
||||
RegisterClassData* Class = GetClass(Reg);
|
||||
uint32_t RegBits = GetRegBits(Reg);
|
||||
|
||||
if (Reg.Class != GPRFixedClass && Reg.Class != FPRFixedClass && (Class->Available & RegBits) == RegBits) {
|
||||
if (Reg.AsRegClass() != RegClass::GPRFixed && Reg.AsRegClass() != RegClass::FPRFixed && (Class->Available & RegBits) == RegBits) {
|
||||
SetReg(CodeNode, Reg);
|
||||
return;
|
||||
}
|
||||
@@ -394,7 +395,7 @@ private:
|
||||
|
||||
// Try to coalesce reserved pairs. Just a heuristic to remove some moves.
|
||||
if (IROp->Op == OP_ALLOCATEGPR && IROp->C<IROp_AllocateGPR>()->ForPair) {
|
||||
uint32_t Available = Classes[GPRClass].Available;
|
||||
uint32_t Available = Classes[FEXCore::ToUnderlying(RegClass::GPR)].Available;
|
||||
|
||||
// Only choose base register R if R and R + 1 are both free
|
||||
Available &= (Available >> 1);
|
||||
@@ -405,20 +406,20 @@ private:
|
||||
|
||||
if (Available) {
|
||||
unsigned Reg = std::countr_zero(Available);
|
||||
SetReg(CodeNode, PhysicalRegister(GPRClass, Reg));
|
||||
SetReg(CodeNode, PhysicalRegister(RegClass::GPR, Reg));
|
||||
return;
|
||||
}
|
||||
} else if (IROp->Op == OP_ALLOCATEGPRAFTER) {
|
||||
uint32_t Available = Classes[GPRClass].Available;
|
||||
uint32_t Available = Classes[FEXCore::ToUnderlying(RegClass::GPR)].Available;
|
||||
auto After = PhysicalRegister(IROp->Args[0]);
|
||||
if ((After.Reg & 1) == 0 && Available & (1ull << (After.Reg + 1))) {
|
||||
SetReg(CodeNode, PhysicalRegister(GPRClass, After.Reg + 1));
|
||||
SetReg(CodeNode, PhysicalRegister(RegClass::GPR, After.Reg + 1));
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
RegisterClassType ClassType = GetRegClassFromNode(IR, IROp);
|
||||
RegisterClass* Class = &Classes[ClassType];
|
||||
RegClass ClassType = GetRegClassFromNode(IR, IROp);
|
||||
RegisterClassData* Class = &Classes[FEXCore::ToUnderlying(ClassType)];
|
||||
|
||||
// Spill to make room in the register file.
|
||||
if (!Class->Available) {
|
||||
@@ -433,10 +434,10 @@ private:
|
||||
};
|
||||
};
|
||||
|
||||
void ConstrainedRAPass::AddRegisters(IR::RegisterClassType Class, uint32_t RegisterCount) {
|
||||
void ConstrainedRAPass::AddRegisters(IR::RegClass Class, uint32_t RegisterCount) {
|
||||
LOGMAN_THROW_A_FMT(RegisterCount <= 31, "Up to 31 regs supported");
|
||||
|
||||
Classes[Class].Count = RegisterCount;
|
||||
Classes[FEXCore::ToUnderlying(Class)].Count = RegisterCount;
|
||||
}
|
||||
|
||||
inline bool KillMove(IROp_Header* LastOp, IROp_Header* IROp, Ref LastNode, Ref CodeNode) {
|
||||
@@ -530,7 +531,7 @@ bool ConstrainedRAPass::TryPostRAMerge(Ref LastNode, Ref CodeNode, IROp_Header*
|
||||
const auto Result = CPUID->RunFunction(ConstantFunction, 0 /* leaf */);
|
||||
|
||||
IREmit->SetWriteCursorBefore(CodeNode);
|
||||
IREmit->_Fence({FEXCore::IR::Fence_Inst});
|
||||
IREmit->_Fence(IR::FenceType::Inst);
|
||||
IREmit->_Constant(Result.eax).Node->Reg = PhysicalRegister(Op->OutEAX).Raw;
|
||||
IREmit->_Constant(Result.ebx).Node->Reg = PhysicalRegister(Op->OutEBX).Raw;
|
||||
IREmit->_Constant(Result.ecx).Node->Reg = PhysicalRegister(Op->OutECX).Raw;
|
||||
@@ -663,7 +664,7 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
// Static registers must be consistent at SRA load/store. Evict to ensure.
|
||||
if (auto Node = DecodeSRANode(IROp, CodeNode); Node != nullptr) {
|
||||
auto Reg = DecodeSRAReg(IROp, CodeNode);
|
||||
RegisterClass* Class = &Classes[Reg.Class];
|
||||
RegisterClassData* Class = &Classes[Reg.Class];
|
||||
|
||||
if (!(Class->Available & (1u << Reg.Reg))) {
|
||||
Ref Old = Class->RegToSSA[Reg.Reg];
|
||||
@@ -678,7 +679,7 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) {
|
||||
|
||||
Ref Copy;
|
||||
|
||||
if (Reg.Class == FPRFixedClass) {
|
||||
if (Reg.AsRegClass() == RegClass::FPRFixed) {
|
||||
IROp_Header* Header = IR->GetOp<IROp_Header>(Old);
|
||||
Copy = IREmit->_VMov(Header->Size, OrderedNodeWrapper::FromImmediate(Reg.Raw));
|
||||
} else {
|
||||
|
||||
@@ -12,14 +12,14 @@ $end_info$
|
||||
#include <stdint.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
struct RegisterClassType;
|
||||
enum class RegClass : uint32_t;
|
||||
|
||||
class RegisterAllocationPass : public FEXCore::IR::Pass {
|
||||
public:
|
||||
virtual void AddRegisters(FEXCore::IR::RegisterClassType Class, uint32_t RegisterCount) = 0;
|
||||
virtual void AddRegisters(RegClass Class, uint32_t RegisterCount) = 0;
|
||||
|
||||
// Number of GPRs usable for pairs at start of GPR set. Must be even.
|
||||
uint32_t PairRegs;
|
||||
uint32_t PairRegs {};
|
||||
};
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -6,7 +6,6 @@
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "FEXCore/IR/IR.h"
|
||||
#include "FEXCore/Utils/Profiler.h"
|
||||
#include "FEXCore/Utils/MathUtils.h"
|
||||
#include "FEXCore/Core/HostFeatures.h"
|
||||
#include "Interface/Core/Addressing.h"
|
||||
|
||||
@@ -66,7 +65,7 @@ public:
|
||||
int8_t TopOffset = 0;
|
||||
|
||||
FixedSizeStack()
|
||||
: buffer(FixedSizeStack::size, {StackSlot::UNUSED, T()}) {}
|
||||
: buffer(FixedSizeStack::size, {StackSlot::UNUSED, T::Invalid}) {}
|
||||
|
||||
void push(const T& Value) {
|
||||
rotate();
|
||||
@@ -85,7 +84,7 @@ public:
|
||||
}
|
||||
|
||||
void pop() {
|
||||
buffer.front() = {StackSlot::INVALID, T()};
|
||||
buffer.front() = {StackSlot::INVALID, T::Invalid};
|
||||
rotate(false);
|
||||
}
|
||||
|
||||
@@ -103,7 +102,7 @@ public:
|
||||
|
||||
void clear() {
|
||||
for (auto& Elem : buffer) {
|
||||
Elem = {StackSlot::UNUSED, T()};
|
||||
Elem = {StackSlot::UNUSED, T::Invalid};
|
||||
}
|
||||
TopOffset = 0;
|
||||
}
|
||||
@@ -171,28 +170,40 @@ private:
|
||||
// Helpers
|
||||
Ref RotateRight8(uint32_t V, Ref Amount);
|
||||
|
||||
void F80SplitStore_Helper(const IROp_StoreStackMem* Op, Ref StackNode) {
|
||||
Ref AddrNode = IR->GetNode(Op->Addr);
|
||||
Ref Offset = IR->GetNode(Op->Offset);
|
||||
OpSize Align = Op->Align;
|
||||
MemOffsetType OffsetType = Op->OffsetType;
|
||||
uint8_t OffsetScale = Op->OffsetScale;
|
||||
|
||||
IREmit->_StoreMem(FPRClass, OpSize::i64Bit, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
void F80SplitStore_Helper(const IROp_StoreStackMem* Op, Ref StackNode, Ref AddrNode, Ref Offset, OpSize Align, MemOffsetType OffsetType,
|
||||
uint8_t OffsetScale) {
|
||||
IREmit->_StoreMemFPR(OpSize::i64Bit, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
auto Upper = IREmit->_VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, StackNode, 1);
|
||||
|
||||
// Store the Upper part of the register (the remaining 2 bytes) into memory.
|
||||
AddressMode A {.Base = AddrNode,
|
||||
.Index = Op->Offset.IsInvalid() ? nullptr : Offset,
|
||||
.IndexType = MEM_OFFSET_SXTX,
|
||||
.IndexScale = OffsetScale,
|
||||
.Offset = 8,
|
||||
.IndexType = MemOffsetType::SXTX,
|
||||
.IndexScale = OffsetScale,
|
||||
.AddrSize = OpSize::i64Bit};
|
||||
A = SelectAddressMode(IREmit, A, GPROpSize, Features.SupportsTSOImm9, false, false, OpSize::i16Bit);
|
||||
IREmit->_StoreMem(GPRClass, OpSize::i16Bit, Upper, A.Base, A.Index, OpSize::i64Bit, MEM_OFFSET_SXTX, A.IndexScale);
|
||||
IREmit->_StoreMemGPR(OpSize::i16Bit, Upper, A.Base, A.Index, OpSize::i64Bit, MemOffsetType::SXTX, A.IndexScale);
|
||||
}
|
||||
|
||||
void Store80BitToMem(const IROp_StoreStackMem* Op, Ref StackNode, Ref AddrNode, Ref Offset, OpSize Align, MemOffsetType OffsetType,
|
||||
uint8_t OffsetScale) {
|
||||
if (Features.SupportsSVE128 || Features.SupportsSVE256) {
|
||||
AddressMode A {.Base = AddrNode,
|
||||
.Index = Op->Offset.IsInvalid() ? nullptr : Offset,
|
||||
.IndexType = MemOffsetType::SXTX,
|
||||
.IndexScale = OffsetScale,
|
||||
.AddrSize = OpSize::i64Bit};
|
||||
AddrNode = LoadEffectiveAddress(IREmit, A, GPROpSize, false);
|
||||
IREmit->_StoreMemX87SVEOptPredicate(OpSize::i128Bit, OpSize::i16Bit, StackNode, AddrNode);
|
||||
} else {
|
||||
F80SplitStore_Helper(Op, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
}
|
||||
}
|
||||
|
||||
void StoreStackMem_Helper(const IROp_StoreStackMem* Op, Ref StackNode) {
|
||||
LOGMAN_THROW_A_FMT(!ReducedPrecisionMode, "Full precision mode expected.");
|
||||
|
||||
Ref AddrNode = IR->GetNode(Op->Addr);
|
||||
Ref Offset = IR->GetNode(Op->Offset);
|
||||
OpSize Align = Op->Align;
|
||||
@@ -204,22 +215,12 @@ private:
|
||||
case OpSize::i32Bit:
|
||||
case OpSize::i64Bit: {
|
||||
StackNode = IREmit->_F80CVT(Op->StoreSize, StackNode);
|
||||
IREmit->_StoreMem(FPRClass, Op->StoreSize, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
IREmit->_StoreMemFPR(Op->StoreSize, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
break;
|
||||
}
|
||||
|
||||
case OpSize::f80Bit: {
|
||||
if (Features.SupportsSVE128 || Features.SupportsSVE256) {
|
||||
AddressMode A {.Base = AddrNode,
|
||||
.Index = Op->Offset.IsInvalid() ? nullptr : Offset,
|
||||
.IndexType = MEM_OFFSET_SXTX,
|
||||
.IndexScale = OffsetScale,
|
||||
.AddrSize = OpSize::i64Bit};
|
||||
AddrNode = LoadEffectiveAddress(IREmit, A, GPROpSize, false);
|
||||
IREmit->_StoreMemX87SVEOptPredicate(OpSize::i128Bit, OpSize::i16Bit, StackNode, AddrNode);
|
||||
} else { // 80bit requires split-store
|
||||
F80SplitStore_Helper(Op, StackNode);
|
||||
}
|
||||
Store80BitToMem(Op, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
break;
|
||||
}
|
||||
default: ERROR_AND_DIE_FMT("Unsupported x87 size");
|
||||
@@ -229,6 +230,8 @@ private:
|
||||
// Performs a store to memory from a value the stack passed in as StackNode.
|
||||
// This is the version dealing with the reduced precision case.
|
||||
void StoreStackMem_Reduced_Helper(const IROp_StoreStackMem* Op, Ref StackNode) {
|
||||
LOGMAN_THROW_A_FMT(ReducedPrecisionMode, "Reduced precision mode expected.");
|
||||
|
||||
Ref AddrNode = IR->GetNode(Op->Addr);
|
||||
Ref Offset = IR->GetNode(Op->Offset);
|
||||
OpSize Align = Op->Align;
|
||||
@@ -241,14 +244,13 @@ private:
|
||||
[[fallthrough]];
|
||||
}
|
||||
case OpSize::i64Bit: {
|
||||
IREmit->_StoreMem(FPRClass, Op->StoreSize, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
IREmit->_StoreMemFPR(Op->StoreSize, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
break;
|
||||
}
|
||||
|
||||
// 80bit requires split-store
|
||||
case OpSize::f80Bit: {
|
||||
StackNode = IREmit->_F80CVTTo(StackNode, OpSize::i64Bit);
|
||||
F80SplitStore_Helper(Op, StackNode);
|
||||
Store80BitToMem(Op, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
break;
|
||||
}
|
||||
default: ERROR_AND_DIE_FMT("Unsupported x87 size");
|
||||
@@ -290,23 +292,24 @@ private:
|
||||
void Reset();
|
||||
|
||||
struct StackMemberInfo {
|
||||
StackMemberInfo() {}
|
||||
StackMemberInfo() = delete;
|
||||
StackMemberInfo(Ref Data)
|
||||
: StackDataNode(Data) {}
|
||||
StackMemberInfo(Ref Data, Ref Source, OpSize Size, bool Float)
|
||||
StackMemberInfo(Ref Data, Ref Source, OpSize Size)
|
||||
: StackDataNode(Data)
|
||||
, Source({Size, Source})
|
||||
, InterpretAsFloat(Float) {}
|
||||
, Source({Size, Source}) {}
|
||||
Ref StackDataNode {}; // Reference to the data in the Stack.
|
||||
// This is the source data node in the stack format, possibly converted to 64/80 bits.
|
||||
struct StackMemberData final {
|
||||
OpSize Size;
|
||||
Ref Node;
|
||||
};
|
||||
|
||||
static const StackMemberInfo Invalid;
|
||||
|
||||
// Tuple is only valid if we have information about the Source of the Stack Data Node.
|
||||
// In it's valid then OpSize is the original source size and Ref is the original source node.
|
||||
std::optional<StackMemberData> Source {};
|
||||
bool InterpretAsFloat {false}; // True if this is a floating point value, false if integer
|
||||
};
|
||||
|
||||
// StackData, TopCache need to be always properly set to ensure
|
||||
@@ -359,6 +362,8 @@ private:
|
||||
IRListView* IR = nullptr;
|
||||
};
|
||||
|
||||
inline const X87StackOptimization::StackMemberInfo X87StackOptimization::StackMemberInfo::Invalid {nullptr};
|
||||
|
||||
inline void X87StackOptimization::InvalidateCaches() {
|
||||
InvalidateCachedRegs();
|
||||
ConstantPool.fill(nullptr);
|
||||
@@ -401,8 +406,7 @@ inline void X87StackOptimization::MigrateToSlowPathIf(bool ShouldMigrate) {
|
||||
|
||||
inline Ref X87StackOptimization::GetTopWithCache_Slow() {
|
||||
if (!TopOffsetCache[0]) {
|
||||
TopOffsetCache[0] =
|
||||
IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
TopOffsetCache[0] = IREmit->_LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
}
|
||||
return TopOffsetCache[0];
|
||||
}
|
||||
@@ -447,7 +451,7 @@ inline void X87StackOptimization::SetTopWithCache_Slow(Ref Value) {
|
||||
|
||||
inline Ref X87StackOptimization::GetFTW() {
|
||||
if (!FTWCached) {
|
||||
FTWCached = IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
FTWCached = IREmit->_LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
}
|
||||
return FTWCached;
|
||||
}
|
||||
@@ -470,7 +474,7 @@ inline Ref X87StackOptimization::LoadStackValueAtOffset_Slow(uint8_t Offset) {
|
||||
OrderedNode* TopOffsetAddress = GetOffsetTopAddressWithCache_Slow(Offset);
|
||||
auto Size = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
if (!TopValueCache[Offset]) {
|
||||
TopValueCache[Offset] = IREmit->_LoadMem(FPRClass, Size, TopOffsetAddress, IREmit->_InlineConstant(MMBaseOffset()), Size, MEM_OFFSET_SXTX, 1);
|
||||
TopValueCache[Offset] = IREmit->_LoadMemFPR(Size, TopOffsetAddress, IREmit->_InlineConstant(MMBaseOffset()), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
return TopValueCache[Offset];
|
||||
}
|
||||
@@ -595,7 +599,7 @@ inline void X87StackOptimization::UpdateTopForPush_Slow() {
|
||||
|
||||
void X87StackOptimization::FlushCachedRegs() {
|
||||
if (FlushTopPending) {
|
||||
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, TopOffsetCache[0], offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
IREmit->_StoreContextGPR(OpSize::i8Bit, TopOffsetCache[0], offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
FlushTopPending = false;
|
||||
}
|
||||
|
||||
@@ -603,7 +607,7 @@ void X87StackOptimization::FlushCachedRegs() {
|
||||
for (size_t i = 0; i < FlushValuesPending.size(); i++) {
|
||||
if (FlushValuesPending[i]) {
|
||||
OrderedNode* TopOffsetAddress = GetOffsetTopAddressWithCache_Slow(i);
|
||||
IREmit->_StoreMem(FPRClass, Size, TopValueCache[i], TopOffsetAddress, IREmit->_InlineConstant(MMBaseOffset()), Size, MEM_OFFSET_SXTX, 1);
|
||||
IREmit->_StoreMemFPR(Size, TopValueCache[i], TopOffsetAddress, IREmit->_InlineConstant(MMBaseOffset()), Size, MemOffsetType::SXTX, 1);
|
||||
// store
|
||||
FlushValuesPending[i] = false;
|
||||
}
|
||||
@@ -654,7 +658,7 @@ void X87StackOptimization::FlushCachedRegs() {
|
||||
}
|
||||
}();
|
||||
|
||||
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, NewFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
IREmit->_StoreContextGPR(OpSize::i8Bit, NewFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
FTWCached = NewFTW;
|
||||
}
|
||||
|
||||
@@ -729,6 +733,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
// The optimization should run per-block
|
||||
Reset();
|
||||
|
||||
IREmit->SetCurrentCodeBlock(BlockNode);
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
if (!LoweredX87(IROp->Op)) {
|
||||
continue;
|
||||
@@ -928,8 +933,13 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
StoreStackValueAtOffset_Slow(SourceNode);
|
||||
} else {
|
||||
auto* SourceNode = CurrentIR.GetNode(Op->X80Src);
|
||||
auto* OriginalNode = CurrentIR.GetNode(Op->OriginalValue);
|
||||
StackData.push(StackMemberInfo {SourceNode, OriginalNode, Op->LoadSize, Op->Float});
|
||||
if (Op->OriginalValue.IsInvalid()) {
|
||||
// No original value to track - just push the converted data
|
||||
StackData.push(StackMemberInfo {SourceNode});
|
||||
} else {
|
||||
auto* OriginalNode = CurrentIR.GetNode(Op->OriginalValue);
|
||||
StackData.push(StackMemberInfo {SourceNode, OriginalNode, Op->LoadSize});
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -994,9 +1004,16 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
// str w2, [x1]
|
||||
// or similar. As long as the source size and dest size are one and the same.
|
||||
// This will avoid any conversions between source and stack element size and conversion back.
|
||||
if (!SlowPath && Value->Source && Value->Source->Size == Op->StoreSize && Value->InterpretAsFloat) {
|
||||
IREmit->_StoreMem(Value->InterpretAsFloat ? FPRClass : GPRClass, Op->StoreSize, Value->Source->Node, AddrNode, Offset, Align,
|
||||
OffsetType, OffsetScale);
|
||||
OpSize StoreSize = Op->StoreSize;
|
||||
LOGMAN_THROW_A_FMT(Op->StoreSize == OpSize::i32Bit || Op->StoreSize == OpSize::i64Bit || Op->StoreSize == OpSize::f80Bit,
|
||||
"Invalid store size in x87 store stack mem");
|
||||
if (!SlowPath && Value->Source && Value->Source->Size == StoreSize) {
|
||||
Ref SourceValue = Value->Source->Node;
|
||||
if (Op->StoreSize == OpSize::f80Bit) {
|
||||
Store80BitToMem(Op, SourceValue, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
} else {
|
||||
IREmit->_StoreMemFPR(StoreSize, SourceValue, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -1036,11 +1053,26 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
case OP_F80STACKXCHANGE: {
|
||||
const auto* Op = IROp->C<IROp_F80StackXchange>();
|
||||
auto Offset = Op->SrcStack;
|
||||
Ref ValueTop = LoadStackValue();
|
||||
Ref ValueOffset = LoadStackValue(Offset);
|
||||
|
||||
StoreStackValue(ValueOffset);
|
||||
StoreStackValue(ValueTop, Offset);
|
||||
if (Offset == 0) {
|
||||
// No-op
|
||||
break;
|
||||
}
|
||||
|
||||
const auto [ValidTop, StackMemberTop] = StackData.top(0);
|
||||
const auto [ValidOffset, StackMemberOffset] = StackData.top(Offset);
|
||||
|
||||
if (ValidTop != StackSlot::VALID || ValidOffset != StackSlot::VALID) {
|
||||
// Slow path: do actual memory operations
|
||||
Ref ValueTop = LoadStackValue();
|
||||
Ref ValueOffset = LoadStackValue(Offset);
|
||||
StoreStackValue(ValueOffset);
|
||||
StoreStackValue(ValueTop, Offset);
|
||||
} else {
|
||||
// Fast path: swap complete StackMemberInfo preserving Source metadata
|
||||
StackData.setTop(StackMemberOffset, 0);
|
||||
StackData.setTop(StackMemberTop, Offset);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -1160,7 +1192,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
|
||||
Ref Value {};
|
||||
if (ReducedPrecisionMode) {
|
||||
Value = IREmit->_Vector_FToI(OpSize::i64Bit, OpSize::i64Bit, St0, Round_Host);
|
||||
Value = IREmit->_Vector_FToI(OpSize::i64Bit, OpSize::i64Bit, St0, RoundMode::Host);
|
||||
} else {
|
||||
Value = IREmit->_F80Round(St0);
|
||||
}
|
||||
|
||||
@@ -19,9 +19,9 @@ union PhysicalRegister {
|
||||
return Raw == Other.Raw;
|
||||
}
|
||||
|
||||
PhysicalRegister(RegisterClassType Class, uint8_t Reg)
|
||||
PhysicalRegister(RegClass Class, uint8_t Reg)
|
||||
: Reg(Reg)
|
||||
, Class(Class.Val) {}
|
||||
, Class(uint8_t(Class)) {}
|
||||
|
||||
PhysicalRegister(OrderedNodeWrapper Arg)
|
||||
: Raw(Arg.GetImmediate()) {}
|
||||
@@ -29,12 +29,16 @@ union PhysicalRegister {
|
||||
PhysicalRegister(Ref Node)
|
||||
: Raw(Node->Reg) {}
|
||||
|
||||
RegClass AsRegClass() const {
|
||||
return RegClass {Class};
|
||||
}
|
||||
|
||||
static const PhysicalRegister Invalid() {
|
||||
return PhysicalRegister(InvalidClass, 0);
|
||||
return PhysicalRegister(RegClass::Invalid, 0);
|
||||
}
|
||||
|
||||
bool IsInvalid() const {
|
||||
static_assert(InvalidClass == 0);
|
||||
static_assert(uint8_t(RegClass::Invalid) == 0);
|
||||
return Raw == 0;
|
||||
}
|
||||
};
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/Utils/PrctlUtils.h>
|
||||
#include <FEXCore/Utils/TypeDefines.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
@@ -50,8 +51,17 @@ void* FEX_mmap(void* addr, size_t length, int prot, int flags, int fd, off_t off
|
||||
errno = -(uint64_t)Result;
|
||||
return (void*)-1;
|
||||
}
|
||||
|
||||
if (flags & MAP_ANONYMOUS) {
|
||||
VirtualName("FEXMem", Result, length);
|
||||
}
|
||||
return Result;
|
||||
}
|
||||
|
||||
void VirtualName(const char* Name, void* Ptr, size_t Size) {
|
||||
prctl(PR_SET_VMA, PR_SET_VMA_ANON_NAME, Ptr, Size, Name);
|
||||
}
|
||||
|
||||
int FEX_munmap(void* addr, size_t length) {
|
||||
int Result = Alloc64->Munmap(addr, length);
|
||||
|
||||
@@ -83,7 +93,7 @@ void* DisableSBRKAllocations() {
|
||||
// calls won't allocate any memory through that.
|
||||
void* AlignedBRK = reinterpret_cast<void*>(FEXCore::AlignUp(reinterpret_cast<uintptr_t>(StartingSBRK), FEXCore::Utils::FEX_PAGE_SIZE));
|
||||
void* AfterBRK =
|
||||
mmap(AlignedBRK, FEXCore::Utils::FEX_PAGE_SIZE, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED_NOREPLACE | MAP_NORESERVE, -1, 0);
|
||||
::mmap(AlignedBRK, FEXCore::Utils::FEX_PAGE_SIZE, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED_NOREPLACE | MAP_NORESERVE, -1, 0);
|
||||
if (AfterBRK == INVALID_PTR) {
|
||||
// Couldn't allocate the page after the aligned brk? This should never happen.
|
||||
// FEXCore::LogMan isn't configured yet so we just need to print the message.
|
||||
@@ -130,10 +140,7 @@ void ClearHooks() {
|
||||
FEXCore::Allocator::mmap = ::mmap;
|
||||
FEXCore::Allocator::munmap = ::munmap;
|
||||
|
||||
// XXX: This is currently a leak.
|
||||
// We can't work around this yet until static initializers that allocate memory are completely removed from our codebase
|
||||
// Luckily we only remove this on process shutdown, so the kernel will do the cleanup for us
|
||||
Alloc64.release();
|
||||
Alloc::OSAllocator::ReleaseAllocatorWorkaround(std::move(Alloc64));
|
||||
}
|
||||
#pragma GCC diagnostic pop
|
||||
|
||||
@@ -284,7 +291,7 @@ fextl::vector<MemoryRegion> StealMemoryRegion(uintptr_t Begin, uintptr_t End) {
|
||||
--StackRegionIt;
|
||||
|
||||
auto Alloc =
|
||||
mmap(StackRegionIt->Ptr, StackRegionIt->Size, PROT_READ | PROT_WRITE, MAP_ANONYMOUS | MAP_NORESERVE | MAP_PRIVATE | MAP_FIXED, -1, 0);
|
||||
::mmap(StackRegionIt->Ptr, StackRegionIt->Size, PROT_READ | PROT_WRITE, MAP_ANONYMOUS | MAP_NORESERVE | MAP_PRIVATE | MAP_FIXED, -1, 0);
|
||||
|
||||
LogMan::Throw::AFmt(Alloc != MAP_FAILED, "mmap({},{:x}) failed", fmt::ptr(StackRegionIt->Ptr), StackRegionIt->Size);
|
||||
LogMan::Throw::AFmt(Alloc == StackRegionIt->Ptr, "mmap returned {} instead of {}", Alloc, fmt::ptr(StackRegionIt->Ptr));
|
||||
@@ -295,7 +302,7 @@ fextl::vector<MemoryRegion> StealMemoryRegion(uintptr_t Begin, uintptr_t End) {
|
||||
|
||||
// Block remaining memory gaps
|
||||
for (auto RegionIt = Regions.begin(); RegionIt != Regions.end(); ++RegionIt) {
|
||||
auto Alloc = mmap(RegionIt->Ptr, RegionIt->Size, PROT_NONE, MAP_ANONYMOUS | MAP_NORESERVE | MAP_PRIVATE | MAP_FIXED_NOREPLACE, -1, 0);
|
||||
auto Alloc = ::mmap(RegionIt->Ptr, RegionIt->Size, PROT_NONE, MAP_ANONYMOUS | MAP_NORESERVE | MAP_PRIVATE | MAP_FIXED_NOREPLACE, -1, 0);
|
||||
|
||||
LogMan::Throw::AFmt(Alloc != MAP_FAILED, "mmap({},{:x}) failed", fmt::ptr(RegionIt->Ptr), RegionIt->Size);
|
||||
LogMan::Throw::AFmt(Alloc == RegionIt->Ptr, "mmap returned {} instead of {}", Alloc, fmt::ptr(RegionIt->Ptr));
|
||||
|
||||
@@ -98,7 +98,7 @@ private:
|
||||
// Align UsedPages so it pads to the next page.
|
||||
// Necessary to take advantage of madvise zero page pooling.
|
||||
using FlexBitElementType = uint64_t;
|
||||
alignas(4096) FEXCore::FlexBitSet<FlexBitElementType> UsedPages;
|
||||
alignas(FEXCore::Utils::FEX_PAGE_SIZE) FEXCore::FlexBitSet<FlexBitElementType> UsedPages;
|
||||
|
||||
// This returns the size of the LiveVMARegion in addition to the flex set that tracks the used data
|
||||
// The LiveVMARegion lives at the start of the VMA region which means on initialization we need to set that
|
||||
@@ -140,7 +140,7 @@ private:
|
||||
}
|
||||
};
|
||||
|
||||
static_assert(sizeof(LiveVMARegion) == 4096, "Needs to be the size of a page");
|
||||
static_assert(sizeof(LiveVMARegion) == FEXCore::Utils::FEX_PAGE_SIZE, "Needs to be the size of a page");
|
||||
|
||||
static_assert(std::is_trivially_copyable<LiveVMARegion>::value, "Needs to be trivially copyable");
|
||||
static_assert(offsetof(LiveVMARegion, UsedPages) == sizeof(LiveVMARegion), "FlexBitSet needs to be at the end");
|
||||
@@ -168,6 +168,7 @@ private:
|
||||
LOGMAN_THROW_A_FMT(Res != -1, "Couldn't mprotect region: {} '{}' Likely occurs when running out of memory or Maximum VMAs", errno,
|
||||
strerror(errno));
|
||||
|
||||
FEXCore::Allocator::VirtualName("FEXMem_Misc", reinterpret_cast<void*>(ReservedRegion->Base), SizePlusManagedData);
|
||||
LiveVMARegion* LiveRange = new (reinterpret_cast<void*>(ReservedRegion->Base)) LiveVMARegion();
|
||||
|
||||
// Copy over the reserved data
|
||||
@@ -206,7 +207,7 @@ OSAllocator_64Bit::LiveVMARegion* OSAllocator_64Bit::FindLiveRegionForAddress(ui
|
||||
uintptr_t RegionBegin = (*it)->SlabInfo->Base;
|
||||
uintptr_t RegionEnd = RegionBegin + (*it)->SlabInfo->RegionSize;
|
||||
|
||||
if (Addr >= RegionBegin && Addr < RegionEnd) {
|
||||
if (Addr >= RegionBegin && AddrEnd < RegionEnd) {
|
||||
LiveRegion = *it;
|
||||
// Leave our loop
|
||||
break;
|
||||
@@ -404,14 +405,18 @@ again:
|
||||
// Mark the pages as used
|
||||
uintptr_t RegionBegin = LiveRegion->SlabInfo->Base;
|
||||
uintptr_t MappedBegin = (AllocatedOffset - RegionBegin) >> FEXCore::Utils::FEX_PAGE_SHIFT;
|
||||
size_t PagesSet {};
|
||||
|
||||
for (size_t i = 0; i < NumberOfPages; ++i) {
|
||||
LiveRegion->UsedPages.Set(MappedBegin + i);
|
||||
PagesSet += LiveRegion->UsedPages.TestAndSet(MappedBegin + i) == false;
|
||||
}
|
||||
|
||||
// Change our last allocation region
|
||||
LiveRegion->LastPageAllocation = MappedBegin + NumberOfPages;
|
||||
LiveRegion->FreeSpace -= length;
|
||||
LiveRegion->FreeSpace -= PagesSet * FEXCore::Utils::FEX_PAGE_SIZE;
|
||||
LOGMAN_THROW_A_FMT(LiveRegion->FreeSpace <= LiveRegion->SlabInfo->RegionSize,
|
||||
"Corrupt LiveRegion free space! 0x{:x} > 0x{:x}. After allocating 0x{:x} (0x{:x} overlapped)", LiveRegion->FreeSpace,
|
||||
LiveRegion->SlabInfo->RegionSize, length, PagesSet);
|
||||
}
|
||||
|
||||
if (!AllocatedOffset) {
|
||||
@@ -473,7 +478,7 @@ int OSAllocator_64Bit::Munmap(void* addr, size_t length) {
|
||||
::mmap(addr, length, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, 0);
|
||||
}
|
||||
|
||||
(*it)->FreeSpace += FreedPages * 4096;
|
||||
(*it)->FreeSpace += FreedPages * FEXCore::Utils::FEX_PAGE_SIZE;
|
||||
|
||||
// Set the last allocated page to the minimum of last page allocation or this slab
|
||||
// This will let us more quickly fill holes
|
||||
@@ -505,6 +510,8 @@ void OSAllocator_64Bit::AllocateMemoryRegions(fextl::vector<FEXCore::Allocator::
|
||||
// This enables the kernel to use transparent large pages in the allocator which can reduce memory pressure
|
||||
::madvise(it.Ptr, ObjectAllocSize, MADV_HUGEPAGE);
|
||||
|
||||
FEXCore::Allocator::VirtualName("FEXMem_Misc", reinterpret_cast<void*>(it.Ptr), ObjectAllocSize);
|
||||
|
||||
ObjectAlloc = new (it.Ptr) Alloc::ForwardOnlyIntrusiveArenaAllocator(it.Ptr, ObjectAllocSize);
|
||||
ReservedRegions = ObjectAlloc->new_construct(ReservedRegions, ObjectAlloc);
|
||||
LiveRegions = ObjectAlloc->new_construct(LiveRegions, ObjectAlloc);
|
||||
@@ -602,6 +609,8 @@ fextl::unique_ptr<T> make_alloc_unique(FEXCore::Allocator::MemoryRegion& Base, A
|
||||
ERROR_AND_DIE_FMT("Couldn't allocate memory region");
|
||||
}
|
||||
|
||||
FEXCore::Allocator::VirtualName("FEXMem_Misc", reinterpret_cast<void*>(ptr), MinPage);
|
||||
|
||||
// Remove the page from the base region.
|
||||
// Could be zero after this.
|
||||
Base.Size -= MinPage;
|
||||
|
||||
@@ -27,6 +27,11 @@ struct FlexBitSet final {
|
||||
Memory[Element / MinimumSizeBits] &= ~(1ULL << (Element % MinimumSizeBits));
|
||||
return Value;
|
||||
}
|
||||
bool TestAndSet(size_t Element) {
|
||||
bool Value = Get(Element);
|
||||
Memory[Element / MinimumSizeBits] |= (1ULL << (Element % MinimumSizeBits));
|
||||
return Value;
|
||||
}
|
||||
void Set(size_t Element) {
|
||||
Memory[Element / MinimumSizeBits] |= (1ULL << (Element % MinimumSizeBits));
|
||||
}
|
||||
@@ -70,12 +75,17 @@ struct FlexBitSet final {
|
||||
template<bool WantUnset>
|
||||
BitsetScanResults BackwardScanForRange(size_t BeginningElement, size_t ElementCount, size_t MinimumElement) {
|
||||
bool FoundHole {};
|
||||
for (size_t CurrentPage = BeginningElement; CurrentPage >= (MinimumElement + ElementCount);) {
|
||||
|
||||
// Final element to iterate to.
|
||||
const size_t FinalElement = MinimumElement + ElementCount - 1;
|
||||
|
||||
for (size_t CurrentPage = BeginningElement; CurrentPage >= FinalElement;) {
|
||||
size_t Remaining = ElementCount;
|
||||
LOGMAN_THROW_A_FMT(Remaining <= CurrentPage, "Scanning less than available range");
|
||||
LOGMAN_THROW_A_FMT(CurrentPage <= BeginningElement && CurrentPage >= FinalElement, "BackwardScanForRange: Scanning less than "
|
||||
"available range");
|
||||
|
||||
while (Remaining) {
|
||||
if (this->Get(CurrentPage - Remaining) == WantUnset) {
|
||||
if (this->Get(CurrentPage - Remaining + 1) == WantUnset) {
|
||||
// Has an intersecting range
|
||||
break;
|
||||
}
|
||||
@@ -92,7 +102,7 @@ struct FlexBitSet final {
|
||||
CurrentPage -= Remaining;
|
||||
} else {
|
||||
// We have a slab range
|
||||
return BitsetScanResults {CurrentPage - ElementCount, FoundHole};
|
||||
return BitsetScanResults {CurrentPage - ElementCount + 1, FoundHole};
|
||||
}
|
||||
}
|
||||
|
||||
@@ -108,11 +118,15 @@ struct FlexBitSet final {
|
||||
BitsetScanResults ForwardScanForRange(size_t BeginningElement, size_t ElementCount, size_t ElementsInSet) {
|
||||
bool FoundHole {};
|
||||
|
||||
for (size_t CurrentElement = BeginningElement; CurrentElement < (ElementsInSet - ElementCount);) {
|
||||
// Final element to iterate to.
|
||||
const size_t FinalElement = ElementsInSet - ElementCount + 1;
|
||||
|
||||
for (size_t CurrentElement = BeginningElement; CurrentElement <= FinalElement;) {
|
||||
// If we have enough free space, check if we have enough free pages that are contiguous
|
||||
size_t Remaining = ElementCount;
|
||||
|
||||
LOGMAN_THROW_A_FMT((CurrentElement + Remaining - 1) < ElementsInSet, "Scanning less than available range");
|
||||
LOGMAN_THROW_A_FMT(CurrentElement >= BeginningElement && CurrentElement <= FinalElement, "ForwardScanForRange: Scanning less than "
|
||||
"available range");
|
||||
|
||||
while (Remaining) {
|
||||
if (this->Get(CurrentElement + Remaining - 1) == WantUnset) {
|
||||
|
||||
@@ -53,4 +53,12 @@ public:
|
||||
namespace Alloc::OSAllocator {
|
||||
fextl::unique_ptr<Alloc::HostAllocator> Create64BitAllocator();
|
||||
fextl::unique_ptr<Alloc::HostAllocator> Create64BitAllocatorWithRegions(fextl::vector<FEXCore::Allocator::MemoryRegion>& Regions);
|
||||
static inline void ReleaseAllocatorWorkaround(fextl::unique_ptr<Alloc::HostAllocator> Allocator) {
|
||||
// XXX: This is currently a leak.
|
||||
// We can't work around this yet until static initializers that allocate memory are completely removed from our codebase
|
||||
// The allocator is also intrusively allocated, so the unique_ptr tries to double free the HostAllocator object.
|
||||
// Luckily we only remove this on process shutdown, so the kernel will do the cleanup for us
|
||||
Allocator.release();
|
||||
}
|
||||
|
||||
} // namespace Alloc::OSAllocator
|
||||
@@ -144,33 +144,33 @@ static __uint128_t LoadAcquire128(uint64_t Addr) {
|
||||
}
|
||||
|
||||
static uint64_t LoadAcquire64(uint64_t Addr) {
|
||||
std::atomic<uint64_t>* Atom = reinterpret_cast<std::atomic<uint64_t>*>(Addr);
|
||||
return Atom->load(std::memory_order_acquire);
|
||||
auto Atom = std::atomic_ref<uint64_t>(*reinterpret_cast<uint64_t*>(Addr));
|
||||
return Atom.load(std::memory_order_acquire);
|
||||
}
|
||||
|
||||
static bool StoreCAS64(uint64_t& Expected, uint64_t Val, uint64_t Addr) {
|
||||
std::atomic<uint64_t>* Atom = reinterpret_cast<std::atomic<uint64_t>*>(Addr);
|
||||
return Atom->compare_exchange_strong(Expected, Val);
|
||||
auto Atom = std::atomic_ref<uint64_t>(*reinterpret_cast<uint64_t*>(Addr));
|
||||
return Atom.compare_exchange_strong(Expected, Val);
|
||||
}
|
||||
|
||||
static uint32_t LoadAcquire32(uint64_t Addr) {
|
||||
std::atomic<uint32_t>* Atom = reinterpret_cast<std::atomic<uint32_t>*>(Addr);
|
||||
return Atom->load(std::memory_order_acquire);
|
||||
auto Atom = std::atomic_ref<uint32_t>(*reinterpret_cast<uint32_t*>(Addr));
|
||||
return Atom.load(std::memory_order_acquire);
|
||||
}
|
||||
|
||||
static bool StoreCAS32(uint32_t& Expected, uint32_t Val, uint64_t Addr) {
|
||||
std::atomic<uint32_t>* Atom = reinterpret_cast<std::atomic<uint32_t>*>(Addr);
|
||||
return Atom->compare_exchange_strong(Expected, Val);
|
||||
auto Atom = std::atomic_ref<uint32_t>(*reinterpret_cast<uint32_t*>(Addr));
|
||||
return Atom.compare_exchange_strong(Expected, Val);
|
||||
}
|
||||
|
||||
static uint8_t LoadAcquire8(uint64_t Addr) {
|
||||
std::atomic<uint8_t>* Atom = reinterpret_cast<std::atomic<uint8_t>*>(Addr);
|
||||
return Atom->load(std::memory_order_acquire);
|
||||
auto Atom = std::atomic_ref<uint8_t>(*reinterpret_cast<uint8_t*>(Addr));
|
||||
return Atom.load(std::memory_order_acquire);
|
||||
}
|
||||
|
||||
static bool StoreCAS8(uint8_t& Expected, uint8_t Val, uint64_t Addr) {
|
||||
std::atomic<uint8_t>* Atom = reinterpret_cast<std::atomic<uint8_t>*>(Addr);
|
||||
return Atom->compare_exchange_strong(Expected, Val);
|
||||
auto Atom = std::atomic_ref<uint8_t>(*reinterpret_cast<uint8_t*>(Addr));
|
||||
return Atom.compare_exchange_strong(Expected, Val);
|
||||
}
|
||||
|
||||
static uint16_t DoLoad16(uint64_t Addr) {
|
||||
@@ -211,8 +211,8 @@ static uint16_t DoLoad16(uint64_t Addr) {
|
||||
uint64_t Alignment = Addr & AlignmentMask;
|
||||
Addr &= ~AlignmentMask;
|
||||
|
||||
std::atomic<uint64_t>* Atomic = reinterpret_cast<std::atomic<uint64_t>*>(Addr);
|
||||
uint64_t TmpResult = Atomic->load();
|
||||
auto Atomic = std::atomic_ref<uint64_t>(*reinterpret_cast<uint64_t*>(Addr));
|
||||
uint64_t TmpResult = Atomic.load();
|
||||
|
||||
// Zexts the result
|
||||
uint16_t Result = TmpResult >> (Alignment * 8);
|
||||
@@ -224,8 +224,8 @@ static uint16_t DoLoad16(uint64_t Addr) {
|
||||
uint64_t Alignment = Addr & AlignmentMask;
|
||||
Addr &= ~AlignmentMask;
|
||||
|
||||
std::atomic<uint32_t>* Atomic = reinterpret_cast<std::atomic<uint32_t>*>(Addr);
|
||||
uint32_t TmpResult = Atomic->load();
|
||||
auto Atomic = std::atomic_ref<uint32_t>(*reinterpret_cast<uint32_t*>(Addr));
|
||||
uint32_t TmpResult = Atomic.load();
|
||||
|
||||
// Zexts the result
|
||||
uint16_t Result = TmpResult >> (Alignment * 8);
|
||||
@@ -272,8 +272,8 @@ static uint32_t DoLoad32(uint64_t Addr) {
|
||||
uint64_t Alignment = Addr & AlignmentMask;
|
||||
Addr &= ~AlignmentMask;
|
||||
|
||||
std::atomic<uint64_t>* Atomic = reinterpret_cast<std::atomic<uint64_t>*>(Addr);
|
||||
uint64_t TmpResult = Atomic->load();
|
||||
auto Atomic = std::atomic_ref<uint64_t>(*reinterpret_cast<uint64_t*>(Addr));
|
||||
uint64_t TmpResult = Atomic.load();
|
||||
|
||||
return TmpResult >> (Alignment * 8);
|
||||
}
|
||||
@@ -465,7 +465,7 @@ static bool RunCASPAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg1, uint3
|
||||
// Fits within a 16byte region
|
||||
uint64_t Alignment = Addr & 0b1111;
|
||||
Addr &= ~0b1111ULL;
|
||||
std::atomic<__uint128_t>* Atomic128 = reinterpret_cast<std::atomic<__uint128_t>*>(Addr);
|
||||
auto Atomic128 = std::atomic_ref<__uint128_t>(*reinterpret_cast<__uint128_t*>(Addr));
|
||||
|
||||
__uint128_t Mask = ~0ULL;
|
||||
Mask <<= Alignment * 8;
|
||||
@@ -480,7 +480,7 @@ static bool RunCASPAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg1, uint3
|
||||
Expected <<= Alignment * 8;
|
||||
|
||||
while (1) {
|
||||
TmpExpected = Atomic128->load();
|
||||
TmpExpected = Atomic128.load();
|
||||
|
||||
// Set up expected
|
||||
TmpExpected &= NegMask;
|
||||
@@ -491,7 +491,7 @@ static bool RunCASPAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg1, uint3
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired;
|
||||
|
||||
bool CASResult = Atomic128->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
bool CASResult = Atomic128.compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Successful, so we are done
|
||||
return true;
|
||||
@@ -617,36 +617,6 @@ static uint64_t HandleCASPAL_ARMv8(uint32_t Instr, uintptr_t ProgramCounter, uin
|
||||
}
|
||||
}
|
||||
|
||||
static bool HandleAtomicVectorStore(uint32_t Instr, uintptr_t ProgramCounter) {
|
||||
uint32_t* PC = (uint32_t*)ProgramCounter;
|
||||
|
||||
uint32_t Size = (Instr >> 30) & 1;
|
||||
uint32_t DataReg = Instr & 0x1F;
|
||||
|
||||
if (Size == 1) {
|
||||
// 64-bit pair happens on paranoid vector stores
|
||||
// [0] ldaxp(xzr, TMP3, MemSrc); // <- Can hit SIGBUS. Overwritten with DMB
|
||||
// [1] stlxp(TMP3, TMP1, TMP2, MemSrc); // <- Can also hit SIGBUS
|
||||
// [2] cbnz(TMP3, &B); // < Overwritten with DMB
|
||||
if (DataReg == 31) {
|
||||
uint32_t NextInstr = PC[1];
|
||||
uint32_t AddrReg = (NextInstr >> 5) & 0x1F;
|
||||
DataReg = NextInstr & 0x1F;
|
||||
uint32_t DataReg2 = (NextInstr >> 10) & 0x1F;
|
||||
uint32_t STP = (0b10 << 30) | (0b101001000000000 << 15) | (DataReg2 << 10) | (AddrReg << 5) | DataReg;
|
||||
|
||||
PC[0] = DMB;
|
||||
PC[1] = STP;
|
||||
PC[2] = DMB;
|
||||
// Back up one instruction and have another go
|
||||
ClearICache(&PC[0], 12);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
using CASExpectedFn = T (*)(T Src, T Expected);
|
||||
template<typename T>
|
||||
@@ -740,7 +710,7 @@ static uint16_t DoCAS16(uint16_t DesiredSrc, uint16_t ExpectedSrc, uint64_t Addr
|
||||
// Fits within a 16byte region
|
||||
uint64_t Alignment = Addr & 0b1111;
|
||||
Addr &= ~0b1111ULL;
|
||||
std::atomic<__uint128_t>* Atomic128 = reinterpret_cast<std::atomic<__uint128_t>*>(Addr);
|
||||
auto Atomic128 = std::atomic_ref<__uint128_t>(*reinterpret_cast<__uint128_t*>(Addr));
|
||||
|
||||
__uint128_t Mask = 0xFFFF;
|
||||
Mask <<= Alignment * 8;
|
||||
@@ -749,7 +719,7 @@ static uint16_t DoCAS16(uint16_t DesiredSrc, uint16_t ExpectedSrc, uint64_t Addr
|
||||
__uint128_t TmpDesired {};
|
||||
|
||||
while (1) {
|
||||
TmpExpected = Atomic128->load();
|
||||
TmpExpected = Atomic128.load();
|
||||
|
||||
__uint128_t Desired = DesiredFunction(TmpExpected >> (Alignment * 8), DesiredSrc);
|
||||
Desired <<= Alignment * 8;
|
||||
@@ -766,7 +736,7 @@ static uint16_t DoCAS16(uint16_t DesiredSrc, uint16_t ExpectedSrc, uint64_t Addr
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired;
|
||||
|
||||
bool CASResult = Atomic128->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
bool CASResult = Atomic128.compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Successful, so we are done
|
||||
return Expected >> (Alignment * 8);
|
||||
@@ -810,9 +780,9 @@ static uint16_t DoCAS16(uint16_t DesiredSrc, uint16_t ExpectedSrc, uint64_t Addr
|
||||
uint64_t TmpExpected {};
|
||||
uint64_t TmpDesired {};
|
||||
|
||||
std::atomic<uint64_t>* Atomic = reinterpret_cast<std::atomic<uint64_t>*>(Addr);
|
||||
auto Atomic = std::atomic_ref<uint64_t>(*reinterpret_cast<uint64_t*>(Addr));
|
||||
while (1) {
|
||||
TmpExpected = Atomic->load();
|
||||
TmpExpected = Atomic.load();
|
||||
|
||||
uint64_t Desired = DesiredFunction(TmpExpected >> (Alignment * 8), DesiredSrc);
|
||||
Desired <<= Alignment * 8;
|
||||
@@ -829,7 +799,7 @@ static uint16_t DoCAS16(uint16_t DesiredSrc, uint16_t ExpectedSrc, uint64_t Addr
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired;
|
||||
|
||||
bool CASResult = Atomic->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
bool CASResult = Atomic.compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Successful, so we are done
|
||||
return Expected >> (Alignment * 8);
|
||||
@@ -873,9 +843,9 @@ static uint16_t DoCAS16(uint16_t DesiredSrc, uint16_t ExpectedSrc, uint64_t Addr
|
||||
uint32_t TmpExpected {};
|
||||
uint32_t TmpDesired {};
|
||||
|
||||
std::atomic<uint32_t>* Atomic = reinterpret_cast<std::atomic<uint32_t>*>(Addr);
|
||||
auto Atomic = std::atomic_ref<uint32_t>(*reinterpret_cast<uint32_t*>(Addr));
|
||||
while (1) {
|
||||
TmpExpected = Atomic->load();
|
||||
TmpExpected = Atomic.load();
|
||||
|
||||
|
||||
uint32_t Desired = DesiredFunction(TmpExpected >> (Alignment * 8), DesiredSrc);
|
||||
@@ -893,7 +863,7 @@ static uint16_t DoCAS16(uint16_t DesiredSrc, uint16_t ExpectedSrc, uint64_t Addr
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired;
|
||||
|
||||
bool CASResult = Atomic->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
bool CASResult = Atomic.compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Successful, so we are done
|
||||
return Expected >> (Alignment * 8);
|
||||
@@ -1040,7 +1010,7 @@ static uint32_t DoCAS32(uint32_t DesiredSrc, uint32_t ExpectedSrc, uint64_t Addr
|
||||
// Fits within a 16byte region
|
||||
uint64_t Alignment = Addr & 0b1111;
|
||||
Addr &= ~0b1111ULL;
|
||||
std::atomic<__uint128_t>* Atomic128 = reinterpret_cast<std::atomic<__uint128_t>*>(Addr);
|
||||
auto Atomic128 = std::atomic_ref<__uint128_t>(*reinterpret_cast<__uint128_t*>(Addr));
|
||||
|
||||
__uint128_t Mask = ~0U;
|
||||
Mask <<= Alignment * 8;
|
||||
@@ -1049,7 +1019,7 @@ static uint32_t DoCAS32(uint32_t DesiredSrc, uint32_t ExpectedSrc, uint64_t Addr
|
||||
__uint128_t TmpDesired {};
|
||||
|
||||
while (1) {
|
||||
__uint128_t TmpActual = Atomic128->load();
|
||||
__uint128_t TmpActual = Atomic128.load();
|
||||
|
||||
__uint128_t Desired = DesiredFunction(TmpActual >> (Alignment * 8), DesiredSrc);
|
||||
__uint128_t Expected = ExpectedFunction(TmpActual >> (Alignment * 8), ExpectedSrc);
|
||||
@@ -1064,7 +1034,7 @@ static uint32_t DoCAS32(uint32_t DesiredSrc, uint32_t ExpectedSrc, uint64_t Addr
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired << (Alignment * 8);
|
||||
|
||||
bool CASResult = Atomic128->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
bool CASResult = Atomic128.compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Stored successfully
|
||||
return Expected;
|
||||
@@ -1108,9 +1078,9 @@ static uint32_t DoCAS32(uint32_t DesiredSrc, uint32_t ExpectedSrc, uint64_t Addr
|
||||
uint64_t TmpExpected {};
|
||||
uint64_t TmpDesired {};
|
||||
|
||||
std::atomic<uint64_t>* Atomic = reinterpret_cast<std::atomic<uint64_t>*>(Addr);
|
||||
auto Atomic = std::atomic_ref<uint64_t>(*reinterpret_cast<uint64_t*>(Addr));
|
||||
while (1) {
|
||||
uint64_t TmpActual = Atomic->load();
|
||||
uint64_t TmpActual = Atomic.load();
|
||||
|
||||
uint64_t Desired = DesiredFunction(TmpActual >> (Alignment * 8), DesiredSrc);
|
||||
uint64_t Expected = ExpectedFunction(TmpActual >> (Alignment * 8), ExpectedSrc);
|
||||
@@ -1125,7 +1095,7 @@ static uint32_t DoCAS32(uint32_t DesiredSrc, uint32_t ExpectedSrc, uint64_t Addr
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired << (Alignment * 8);
|
||||
|
||||
bool CASResult = Atomic->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
bool CASResult = Atomic.compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Stored successfully
|
||||
return Expected;
|
||||
@@ -1270,7 +1240,7 @@ static uint64_t DoCAS64(uint64_t DesiredSrc, uint64_t ExpectedSrc, uint64_t Addr
|
||||
// Fits within a 16byte region
|
||||
uint64_t Alignment = Addr & AlignmentMask;
|
||||
Addr &= ~AlignmentMask;
|
||||
std::atomic<__uint128_t>* Atomic128 = reinterpret_cast<std::atomic<__uint128_t>*>(Addr);
|
||||
auto Atomic128 = std::atomic_ref<__uint128_t>(*reinterpret_cast<__uint128_t*>(Addr));
|
||||
|
||||
__uint128_t Mask = ~0ULL;
|
||||
Mask <<= Alignment * 8;
|
||||
@@ -1279,7 +1249,7 @@ static uint64_t DoCAS64(uint64_t DesiredSrc, uint64_t ExpectedSrc, uint64_t Addr
|
||||
__uint128_t TmpDesired {};
|
||||
|
||||
while (1) {
|
||||
__uint128_t TmpActual = Atomic128->load();
|
||||
__uint128_t TmpActual = Atomic128.load();
|
||||
|
||||
__uint128_t Desired = DesiredFunction(TmpActual >> (Alignment * 8), DesiredSrc);
|
||||
__uint128_t Expected = ExpectedFunction(TmpActual >> (Alignment * 8), ExpectedSrc);
|
||||
@@ -1294,7 +1264,7 @@ static uint64_t DoCAS64(uint64_t DesiredSrc, uint64_t ExpectedSrc, uint64_t Addr
|
||||
TmpDesired &= NegMask;
|
||||
TmpDesired |= Desired << (Alignment * 8);
|
||||
|
||||
bool CASResult = Atomic128->compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
bool CASResult = Atomic128.compare_exchange_strong(TmpExpected, TmpDesired);
|
||||
if (CASResult) {
|
||||
// Stored successfully
|
||||
return Expected;
|
||||
@@ -1326,9 +1296,7 @@ static uint64_t DoCAS64(uint64_t DesiredSrc, uint64_t ExpectedSrc, uint64_t Addr
|
||||
}
|
||||
}
|
||||
|
||||
static bool RunCASAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg, uint32_t ExpectedReg, uint32_t AddressReg, uint32_t* StrictSplitLockMutex) {
|
||||
uint64_t Addr = GPRs[AddressReg];
|
||||
|
||||
static std::optional<uint64_t> DoCAS(uint32_t Size, uint64_t Desired, uint64_t Expected, uint64_t Addr, uint32_t* StrictSplitLockMutex) {
|
||||
// Cross-cacheline CAS doesn't work on ARM
|
||||
// It isn't even guaranteed to work on x86
|
||||
// Intel will do a "split lock" which locks the full bus
|
||||
@@ -1341,7 +1309,7 @@ static bool RunCASAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg, uint32_
|
||||
// Only need to handle 16, 32, 64
|
||||
if (Size == 2) {
|
||||
auto Res = DoCAS16<false>(
|
||||
GPRs[DesiredReg], GPRs[ExpectedReg], Addr,
|
||||
Desired, Expected, Addr,
|
||||
[](uint16_t, uint16_t Expected) -> uint16_t {
|
||||
// Expected is just Expected
|
||||
return Expected;
|
||||
@@ -1351,16 +1319,10 @@ static bool RunCASAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg, uint32_
|
||||
return Desired;
|
||||
},
|
||||
StrictSplitLockMutex);
|
||||
|
||||
// Regardless of pass or fail
|
||||
// We set the result register if it isn't a zero register
|
||||
if (ExpectedReg != 31) {
|
||||
GPRs[ExpectedReg] = Res;
|
||||
}
|
||||
return true;
|
||||
return Res;
|
||||
} else if (Size == 4) {
|
||||
auto Res = DoCAS32<false>(
|
||||
GPRs[DesiredReg], GPRs[ExpectedReg], Addr,
|
||||
Desired, Expected, Addr,
|
||||
[](uint32_t, uint32_t Expected) -> uint32_t {
|
||||
// Expected is just Expected
|
||||
return Expected;
|
||||
@@ -1370,16 +1332,10 @@ static bool RunCASAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg, uint32_
|
||||
return Desired;
|
||||
},
|
||||
StrictSplitLockMutex);
|
||||
|
||||
// Regardless of pass or fail
|
||||
// We set the result register if it isn't a zero register
|
||||
if (ExpectedReg != 31) {
|
||||
GPRs[ExpectedReg] = Res;
|
||||
}
|
||||
return true;
|
||||
return Res;
|
||||
} else if (Size == 8) {
|
||||
auto Res = DoCAS64<false>(
|
||||
GPRs[DesiredReg], GPRs[ExpectedReg], Addr,
|
||||
Desired, Expected, Addr,
|
||||
[](uint64_t, uint64_t Expected) -> uint64_t {
|
||||
// Expected is just Expected
|
||||
return Expected;
|
||||
@@ -1389,16 +1345,24 @@ static bool RunCASAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg, uint32_
|
||||
return Desired;
|
||||
},
|
||||
StrictSplitLockMutex);
|
||||
|
||||
// Regardless of pass or fail
|
||||
// We set the result register if it isn't a zero register
|
||||
if (ExpectedReg != 31) {
|
||||
GPRs[ExpectedReg] = Res;
|
||||
}
|
||||
return true;
|
||||
return Res;
|
||||
}
|
||||
|
||||
return false;
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
static bool RunCASAL(uint64_t* GPRs, uint32_t Size, uint32_t DesiredReg, uint32_t ExpectedReg, uint32_t AddressReg, uint32_t* StrictSplitLockMutex) {
|
||||
std::optional<uint64_t> Res = DoCAS(Size, GPRs[DesiredReg], GPRs[ExpectedReg], GPRs[AddressReg], StrictSplitLockMutex);
|
||||
if (!Res.has_value()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Regardless of pass or fail
|
||||
// We set the result register if it isn't a zero register
|
||||
if (ExpectedReg != 31) {
|
||||
GPRs[ExpectedReg] = *Res;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool HandleCASAL(uint64_t* GPRs, uint32_t Instr, uint32_t* StrictSplitLockMutex) {
|
||||
@@ -1560,38 +1524,43 @@ static bool HandleAtomicMemOp(uint32_t Instr, uint64_t* GPRs, uint32_t* StrictSp
|
||||
return false;
|
||||
}
|
||||
|
||||
static bool HandleAtomicLoad(uint32_t Instr, uint64_t* GPRs, int64_t Offset) {
|
||||
static bool HandleAtomicLoad(uint32_t Instr, uint64_t* GPRs, int64_t Offset, Core::UnalignedExclusiveStore* Store = nullptr) {
|
||||
uint32_t Size = 1 << (Instr >> 30);
|
||||
|
||||
uint32_t ResultReg = Instr & 0b11111;
|
||||
uint32_t AddressReg = (Instr >> 5) & 0b11111;
|
||||
|
||||
uint64_t Addr = GPRs[AddressReg] + Offset;
|
||||
uint64_t Res;
|
||||
|
||||
if (Size == 2) {
|
||||
auto Res = DoLoad16(Addr);
|
||||
Res = DoLoad16(Addr);
|
||||
// We set the result register if it isn't a zero register
|
||||
if (ResultReg != 31) {
|
||||
GPRs[ResultReg] = Res;
|
||||
}
|
||||
return true;
|
||||
} else if (Size == 4) {
|
||||
auto Res = DoLoad32(Addr);
|
||||
Res = DoLoad32(Addr);
|
||||
// We set the result register if it isn't a zero register
|
||||
if (ResultReg != 31) {
|
||||
GPRs[ResultReg] = Res;
|
||||
}
|
||||
return true;
|
||||
} else if (Size == 8) {
|
||||
auto Res = DoLoad64(Addr);
|
||||
Res = DoLoad64(Addr);
|
||||
// We set the result register if it isn't a zero register
|
||||
if (ResultReg != 31) {
|
||||
GPRs[ResultReg] = Res;
|
||||
}
|
||||
return true;
|
||||
} else {
|
||||
return false;
|
||||
}
|
||||
|
||||
return false;
|
||||
if (Store) {
|
||||
Store->Addr = Addr;
|
||||
Store->Store = Res;
|
||||
Store->Size = Size;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool HandleAtomicStore(uint32_t Instr, uint64_t* GPRs, int64_t Offset, uint32_t* StrictSplitLockMutex) {
|
||||
@@ -1952,8 +1921,8 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
std::optional<int32_t>
|
||||
HandleUnalignedAccess(FEXCore::Core::InternalThreadState* Thread, UnalignedHandlerType HandleType, uintptr_t ProgramCounter, uint64_t* GPRs) {
|
||||
std::optional<int32_t> HandleUnalignedAccess(FEXCore::Core::InternalThreadState* Thread, UnalignedHandlerType HandleType,
|
||||
uintptr_t ProgramCounter, uint64_t* GPRs, bool IsJIT) {
|
||||
#ifdef _M_ARM_64
|
||||
constexpr bool is_arm64 = true;
|
||||
#else
|
||||
@@ -1977,8 +1946,7 @@ HandleUnalignedAccess(FEXCore::Core::InternalThreadState* Thread, UnalignedHandl
|
||||
auto CTX = static_cast<Context::ContextImpl*>(Thread->CTX);
|
||||
uint32_t* StrictSplitLockMutex {CTX->Config.StrictInProcessSplitLocks ? &CTX->StrictSplitLockMutex : nullptr};
|
||||
|
||||
// ParanoidTSO path doesn't modify any code.
|
||||
if (HandleType == UnalignedHandlerType::Paranoid) [[unlikely]] {
|
||||
if (!IsJIT) [[unlikely]] {
|
||||
if ((Instr & LDAXR_MASK) == LDAR_INST || // LDAR*
|
||||
(Instr & LDAXR_MASK) == LDAPR_INST) { // LDAPR*
|
||||
if (ArchHelpers::Arm64::HandleAtomicLoad(Instr, GPRs, 0)) {
|
||||
@@ -2016,7 +1984,29 @@ HandleUnalignedAccess(FEXCore::Core::InternalThreadState* Thread, UnalignedHandl
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDLUR*: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return std::nullopt;
|
||||
}
|
||||
} else if ((Instr & ArchHelpers::Arm64::LDAXR_MASK) == ArchHelpers::Arm64::LDAXR_INST) { // LDAXR*
|
||||
if (ArchHelpers::Arm64::HandleAtomicLoad(Instr, GPRs, 0, &Thread->ExclusiveStore)) {
|
||||
return 4;
|
||||
}
|
||||
} else if ((Instr & ArchHelpers::Arm64::STLXR_MASK) == ArchHelpers::Arm64::STLXR_INST) { // STLXR*
|
||||
uint32_t StatusReg = Instr << 11 >> 27;
|
||||
// // Emulate exclusive store by validating the address and value against the last unaligned LDAXR*.
|
||||
if (GPRs[AddrReg] != Thread->ExclusiveStore.Addr || Size > Thread->ExclusiveStore.Size) {
|
||||
if (StatusReg != 31) {
|
||||
GPRs[StatusReg] = 1;
|
||||
}
|
||||
return 4;
|
||||
}
|
||||
if (std::optional<uint64_t> Prev =
|
||||
DoCAS(Size, DataReg == 31 ? 0 : GPRs[DataReg], Thread->ExclusiveStore.Store, GPRs[AddrReg], StrictSplitLockMutex)) {
|
||||
if (StatusReg != 31) {
|
||||
GPRs[StatusReg] = !!memcmp(&Thread->ExclusiveStore.Store, &*Prev, Size);
|
||||
}
|
||||
Thread->ExclusiveStore.Size = 0;
|
||||
return 4;
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
const auto Frame = Thread->CurrentFrame;
|
||||
@@ -2068,6 +2058,9 @@ HandleUnalignedAccess(FEXCore::Core::InternalThreadState* Thread, UnalignedHandl
|
||||
if (BytesToSkip) {
|
||||
// Skip this instruction now
|
||||
return BytesToSkip;
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS CASPAL: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return std::nullopt;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2131,19 +2124,6 @@ HandleUnalignedAccess(FEXCore::Core::InternalThreadState* Thread, UnalignedHandl
|
||||
ClearICache(&PC[-1], 8);
|
||||
// Back up one instruction and have another go
|
||||
return -4;
|
||||
} else if ((Instr & ArchHelpers::Arm64::LDAXP_MASK) == ArchHelpers::Arm64::LDAXP_INST) { // LDAXP
|
||||
/// This is handling the case of paranoid ARMv8.0-a atomic stores.
|
||||
/// This backpatches the ldaxp+stlxp+cbnz if the previous `HandleCASPAL_ARMv8` didn't handle the case.
|
||||
if (ArchHelpers::Arm64::HandleAtomicVectorStore(Instr, ProgramCounter)) {
|
||||
return 0;
|
||||
} else {
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAXP: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return std::nullopt;
|
||||
}
|
||||
} else if ((Instr & ArchHelpers::Arm64::STLXP_MASK) == ArchHelpers::Arm64::STLXP_INST) { // STLXP
|
||||
// Should not trigger - middle of an LDAXP/STAXP pair.
|
||||
LogMan::Msg::EFmt("Unhandled JIT SIGBUS STLXP: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]);
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
// Check if another thread backpatched this instruction before this thread got here
|
||||
|
||||
@@ -1,7 +1,10 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include <FEXCore/Utils/LongJump.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
namespace FEXCore::LongJump {
|
||||
#include <cstring>
|
||||
|
||||
namespace FEXCore::UncheckedLongJump {
|
||||
#if defined(_M_ARM_64)
|
||||
[[nodiscard]]
|
||||
FEX_DEFAULT_VISIBILITY FEX_NAKED uint64_t SetJump(JumpBuf& Buffer) {
|
||||
@@ -32,7 +35,7 @@ FEX_DEFAULT_VISIBILITY FEX_NAKED uint64_t SetJump(JumpBuf& Buffer) {
|
||||
}
|
||||
|
||||
[[noreturn]]
|
||||
FEX_DEFAULT_VISIBILITY FEX_NAKED void LongJump(JumpBuf& Buffer, uint64_t Value) {
|
||||
FEX_DEFAULT_VISIBILITY FEX_NAKED void LongJump(const JumpBuf& Buffer, uint64_t Value) {
|
||||
__asm volatile(R"(
|
||||
// x0 contains the jumpbuffer
|
||||
ldp x19, x20, [x0, #( 0 * 8)];
|
||||
@@ -58,6 +61,27 @@ FEX_DEFAULT_VISIBILITY FEX_NAKED void LongJump(JumpBuf& Buffer, uint64_t Value)
|
||||
)" ::
|
||||
: "memory");
|
||||
}
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void ManuallyLoadJumpBuf(const JumpBuf& Buffer, uint64_t Value, uint64_t* GPRs, __uint128_t* FPRs, uint64_t* PC) {
|
||||
// First 12 values are registers [x19,x30].
|
||||
memcpy(&GPRs[19], &Buffer.Registers[0], sizeof(uint64_t) * 12);
|
||||
|
||||
// Next 8 values are [D8,D15]
|
||||
// Retain upper 64-bits of the register, only modifying lower 64-bits.
|
||||
for (size_t i = 0; i < 8; ++i) {
|
||||
memcpy(&FPRs[8 + i], &Buffer.Registers[12 + i], sizeof(uint64_t));
|
||||
}
|
||||
|
||||
// Last value is stack pointer
|
||||
memcpy(&GPRs[31], &Buffer.Registers[20], sizeof(uint64_t));
|
||||
|
||||
// Load the expected value in to X0
|
||||
GPRs[0] = Value;
|
||||
|
||||
// Load the PC with the current LR.
|
||||
*PC = GPRs[30];
|
||||
}
|
||||
|
||||
#else
|
||||
[[nodiscard]]
|
||||
FEX_DEFAULT_VISIBILITY FEX_NAKED uint64_t SetJump(JumpBuf& Buffer) {
|
||||
@@ -86,7 +110,7 @@ FEX_DEFAULT_VISIBILITY FEX_NAKED uint64_t SetJump(JumpBuf& Buffer) {
|
||||
}
|
||||
|
||||
[[noreturn]]
|
||||
FEX_DEFAULT_VISIBILITY FEX_NAKED void LongJump(JumpBuf& Buffer, uint64_t Value) {
|
||||
FEX_DEFAULT_VISIBILITY FEX_NAKED void LongJump(const JumpBuf& Buffer, uint64_t Value) {
|
||||
__asm volatile(R"(
|
||||
.intel_syntax noprefix;
|
||||
// rdi contains the jumpbuffer
|
||||
@@ -115,5 +139,9 @@ FEX_DEFAULT_VISIBILITY FEX_NAKED void LongJump(JumpBuf& Buffer, uint64_t Value)
|
||||
: "memory");
|
||||
}
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void ManuallyLoadJumpBuf(JumpBuf& Buffer, uint64_t Value, uint64_t* GPRs, __uint128_t* FPRs, uint64_t* PC) {
|
||||
LOGMAN_MSG_A_FMT("This is unimplemented on x86-64");
|
||||
}
|
||||
|
||||
#endif
|
||||
} // namespace FEXCore::LongJump
|
||||
} // namespace FEXCore::UncheckedLongJump
|
||||
@@ -9,6 +9,7 @@
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
|
||||
@@ -70,6 +71,10 @@ static std::array<const char*, 2> TraceFSDirectories {
|
||||
};
|
||||
|
||||
void Init() {
|
||||
FEX_CONFIG_OPT(EnableGpuvisProfiling, ENABLEGPUVISPROFILING);
|
||||
if (!EnableGpuvisProfiling()) {
|
||||
return;
|
||||
}
|
||||
for (auto Path : TraceFSDirectories) {
|
||||
#ifdef _WIN32
|
||||
constexpr auto flags = O_WRONLY;
|
||||
|
||||
@@ -1,9 +1,14 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <atomic>
|
||||
#include <chrono>
|
||||
#include <mutex>
|
||||
#include <type_traits>
|
||||
|
||||
#include <FEXCore/fextl/functional.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
|
||||
namespace FEXCore::Utils::SpinWaitLock {
|
||||
/**
|
||||
* @brief This provides routines to implement implement an "efficient spin-loop" using ARM's WFE and exclusive monitor interfaces.
|
||||
@@ -123,35 +128,26 @@ static inline uint64_t WFELoadAtomic(uint64_t* Futex) {
|
||||
return Result;
|
||||
}
|
||||
|
||||
template<typename T, typename TT = T>
|
||||
static inline void Wait(T* Futex, TT ExpectedValue) {
|
||||
std::atomic<T>* AtomicFutex = reinterpret_cast<std::atomic<T>*>(Futex);
|
||||
T Result = AtomicFutex->load();
|
||||
template<typename Pred, typename T>
|
||||
static inline void WaitPred(T* Futex, T ComparisonValue) {
|
||||
auto AtomicFutex = std::atomic_ref<T>(*Futex);
|
||||
T Result = AtomicFutex.load();
|
||||
|
||||
// Early exit if possible.
|
||||
if (Result == ExpectedValue) {
|
||||
return;
|
||||
}
|
||||
|
||||
do {
|
||||
while (!Pred {}(Result, ComparisonValue)) {
|
||||
Result = LoadExclusive(Futex);
|
||||
if (Result == ExpectedValue) {
|
||||
if (Pred {}(Result, ComparisonValue)) {
|
||||
return;
|
||||
}
|
||||
Result = WFELoadAtomic(Futex);
|
||||
} while (Result != ExpectedValue);
|
||||
}
|
||||
|
||||
template void Wait<uint8_t>(uint8_t*, uint8_t);
|
||||
template void Wait<uint16_t>(uint16_t*, uint16_t);
|
||||
template void Wait<uint32_t>(uint32_t*, uint32_t);
|
||||
template void Wait<uint64_t>(uint64_t*, uint64_t);
|
||||
Result = WFELoadAtomic(Futex);
|
||||
}
|
||||
}
|
||||
|
||||
template<typename T, typename TT>
|
||||
static inline bool Wait(T* Futex, TT ExpectedValue, const std::chrono::nanoseconds& Timeout) {
|
||||
std::atomic<T>* AtomicFutex = reinterpret_cast<std::atomic<T>*>(Futex);
|
||||
auto AtomicFutex = std::atomic_ref<T>(*Futex);
|
||||
|
||||
T Result = AtomicFutex->load();
|
||||
T Result = AtomicFutex.load();
|
||||
|
||||
// Early exit if possible.
|
||||
if (Result == ExpectedValue) {
|
||||
@@ -184,27 +180,43 @@ template bool Wait<uint16_t>(uint16_t*, uint16_t, const std::chrono::nanoseconds
|
||||
template bool Wait<uint32_t>(uint32_t*, uint32_t, const std::chrono::nanoseconds&);
|
||||
template bool Wait<uint64_t>(uint64_t*, uint64_t, const std::chrono::nanoseconds&);
|
||||
|
||||
#else
|
||||
template<typename T, typename TT>
|
||||
static inline void Wait(T* Futex, TT ExpectedValue) {
|
||||
std::atomic<T>* AtomicFutex = reinterpret_cast<std::atomic<T>*>(Futex);
|
||||
T Result = AtomicFutex->load();
|
||||
template<typename T>
|
||||
static inline T OneShotWFEBitComparison(T* Futex, T Mask, T Comp) {
|
||||
auto AtomicFutex = std::atomic_ref<T>(*Futex);
|
||||
T Result = AtomicFutex.load();
|
||||
|
||||
// Early exit if possible.
|
||||
if (Result == ExpectedValue) {
|
||||
return;
|
||||
if ((Result & Mask) == Comp) {
|
||||
return Result;
|
||||
}
|
||||
|
||||
do {
|
||||
Result = AtomicFutex->load();
|
||||
} while (Result != ExpectedValue);
|
||||
Result = LoadExclusive(Futex);
|
||||
if ((Result & Mask) == Comp) {
|
||||
return Result;
|
||||
}
|
||||
|
||||
// Waits for write and returns result.
|
||||
Result = WFELoadAtomic(Futex);
|
||||
return Result;
|
||||
}
|
||||
|
||||
#else
|
||||
|
||||
template<typename Pred, typename T>
|
||||
static inline void WaitPred(T* Futex, T ComparisonValue) {
|
||||
auto AtomicFutex = std::atomic_ref<T>(*Futex);
|
||||
T Result = AtomicFutex.load();
|
||||
|
||||
while (!Pred {}(Result, ComparisonValue)) {
|
||||
Result = AtomicFutex.load();
|
||||
}
|
||||
}
|
||||
|
||||
template<typename T, typename TT>
|
||||
static inline bool Wait(T* Futex, TT ExpectedValue, const std::chrono::nanoseconds& Timeout) {
|
||||
std::atomic<T>* AtomicFutex = reinterpret_cast<std::atomic<T>*>(Futex);
|
||||
auto AtomicFutex = std::atomic_ref<T>(*Futex);
|
||||
|
||||
T Result = AtomicFutex->load();
|
||||
T Result = AtomicFutex.load();
|
||||
|
||||
// Early exit if possible.
|
||||
if (Result == ExpectedValue) {
|
||||
@@ -214,7 +226,7 @@ static inline bool Wait(T* Futex, TT ExpectedValue, const std::chrono::nanosecon
|
||||
const auto Begin = std::chrono::high_resolution_clock::now();
|
||||
|
||||
do {
|
||||
Result = AtomicFutex->load();
|
||||
Result = AtomicFutex.load();
|
||||
|
||||
const auto CurrentCycleCounter = std::chrono::high_resolution_clock::now();
|
||||
if ((CurrentCycleCounter - Begin) >= Timeout) {
|
||||
@@ -228,14 +240,24 @@ static inline bool Wait(T* Futex, TT ExpectedValue, const std::chrono::nanosecon
|
||||
}
|
||||
#endif
|
||||
|
||||
template<typename T, typename TT = T>
|
||||
static inline void Wait(T* Futex, TT ExpectedValue) {
|
||||
WaitPred<std::equal_to<>, T>(Futex, ExpectedValue);
|
||||
}
|
||||
|
||||
template void Wait<uint8_t>(uint8_t*, uint8_t);
|
||||
template void Wait<uint16_t>(uint16_t*, uint16_t);
|
||||
template void Wait<uint32_t>(uint32_t*, uint32_t);
|
||||
template void Wait<uint64_t>(uint64_t*, uint64_t);
|
||||
|
||||
template<typename T>
|
||||
static inline void lock(T* Futex) {
|
||||
std::atomic<T>* AtomicFutex = reinterpret_cast<std::atomic<T>*>(Futex);
|
||||
auto AtomicFutex = std::atomic_ref<T>(*Futex);
|
||||
T Expected {};
|
||||
T Desired {1};
|
||||
|
||||
// Try to CAS immediately.
|
||||
if (AtomicFutex->compare_exchange_strong(Expected, Desired)) {
|
||||
if (AtomicFutex.compare_exchange_strong(Expected, Desired)) {
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -243,17 +265,17 @@ static inline void lock(T* Futex) {
|
||||
// Wait until the futex is unlocked.
|
||||
Wait(Futex, 0);
|
||||
Expected = 0;
|
||||
} while (!AtomicFutex->compare_exchange_strong(Expected, Desired));
|
||||
} while (!AtomicFutex.compare_exchange_strong(Expected, Desired));
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
static inline bool try_lock(T* Futex) {
|
||||
std::atomic<T>* AtomicFutex = reinterpret_cast<std::atomic<T>*>(Futex);
|
||||
auto AtomicFutex = std::atomic_ref<T>(*Futex);
|
||||
T Expected {};
|
||||
T Desired {1};
|
||||
|
||||
// Try to CAS immediately.
|
||||
if (AtomicFutex->compare_exchange_strong(Expected, Desired)) {
|
||||
if (AtomicFutex.compare_exchange_strong(Expected, Desired)) {
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -262,8 +284,8 @@ static inline bool try_lock(T* Futex) {
|
||||
|
||||
template<typename T>
|
||||
static inline void unlock(T* Futex) {
|
||||
std::atomic<T>* AtomicFutex = reinterpret_cast<std::atomic<T>*>(Futex);
|
||||
AtomicFutex->store(0);
|
||||
auto AtomicFutex = std::atomic_ref<T>(*Futex);
|
||||
AtomicFutex.store(0);
|
||||
}
|
||||
|
||||
#undef SPINLOOP_8BIT
|
||||
|
||||
@@ -0,0 +1,381 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <atomic>
|
||||
#include <cstdint>
|
||||
|
||||
#if !defined(_WIN32)
|
||||
#include <linux/futex.h> /* Definition of FUTEX_* constants */
|
||||
#include <sys/syscall.h> /* Definition of SYS_* constants */
|
||||
#include <unistd.h>
|
||||
#else
|
||||
#include <synchapi.h>
|
||||
#endif
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include "Utils/SpinWaitLock.h"
|
||||
|
||||
namespace FEXCore::Utils::WritePriorityMutex {
|
||||
|
||||
// A custom mutex that prioritizes exclusive locks.
|
||||
// In highly contested scenarios, this can help minimize overall contention time.
|
||||
//
|
||||
// Features:
|
||||
// - Up to 32767 pending exclusive locks ("writers")
|
||||
// - Up to 32767 pending shared_locks ("readers")
|
||||
// - Low-overhead waiting via WFE with a fallback to futex on timeout
|
||||
// - Direct writer->reader hand-off and vice-versa to further reduce overhead
|
||||
//
|
||||
// Trade-offs:
|
||||
// - No guaranteed order of wake-ups besides prioritizing writers
|
||||
// - No support for recursive locking
|
||||
// - We can't use FUTEX_LOCK_PI to enable priority inheritance
|
||||
class Mutex final {
|
||||
public:
|
||||
Mutex() = default;
|
||||
|
||||
// Move-only type
|
||||
Mutex(const Mutex&) = delete;
|
||||
Mutex& operator=(const Mutex&) = delete;
|
||||
Mutex(Mutex&& rhs) = delete;
|
||||
Mutex& operator=(Mutex&&) = delete;
|
||||
|
||||
void lock() {
|
||||
// Try a non-blocking lock first.
|
||||
if (try_lock()) {
|
||||
return;
|
||||
}
|
||||
|
||||
// Try a quick WFE write-lock.
|
||||
if (Attempt_WFE_WriteLock()) {
|
||||
return;
|
||||
}
|
||||
|
||||
// Still couldn't get it. Start waiting.
|
||||
auto AtomicFutex = std::atomic_ref<uint32_t>(Futex);
|
||||
|
||||
uint32_t Expected {};
|
||||
uint32_t Desired {};
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
Expected = AtomicFutex.load(std::memory_order_relaxed);
|
||||
do {
|
||||
// Increment the number of write waiters.
|
||||
Desired = Expected + WRITE_WAITER_INCREMENT;
|
||||
|
||||
LOGMAN_THROW_A_FMT((Desired & WRITE_WAITER_COUNT_MASK) != 0, "Overflow in write-waiters!");
|
||||
} while (AtomicFutex.compare_exchange_strong(Expected, Desired, std::memory_order_acq_rel, std::memory_order_acquire) == false);
|
||||
#else
|
||||
Expected = AtomicFutex.fetch_add(WRITE_WAITER_INCREMENT);
|
||||
Desired = Expected + WRITE_WAITER_INCREMENT;
|
||||
#endif
|
||||
|
||||
// Thread added to waiter list.
|
||||
Expected = Desired;
|
||||
|
||||
while (true) {
|
||||
bool Sleep = false;
|
||||
|
||||
do {
|
||||
if ((Expected & WRITE_OWNED_BIT) == 0 && (Expected & READ_OWNER_COUNT_MASK) == 0) {
|
||||
// If not write-owned, and no read-owners, try to acquire.
|
||||
LOGMAN_THROW_A_FMT((Expected & WRITE_WAITER_COUNT_MASK) != 0, "Underflow in write-waiters!");
|
||||
|
||||
// Add write-owned bit.
|
||||
Desired = Expected | WRITE_OWNED_BIT;
|
||||
|
||||
// Remove ourselves from the wait list.
|
||||
Desired -= WRITE_WAITER_INCREMENT;
|
||||
|
||||
Sleep = false;
|
||||
} else {
|
||||
// Already write-owned or read-locked. Go to sleep.
|
||||
Desired = Expected;
|
||||
Sleep = true;
|
||||
break;
|
||||
}
|
||||
|
||||
} while (AtomicFutex.compare_exchange_strong(Expected, Desired, std::memory_order_acq_rel, std::memory_order_acquire) == false);
|
||||
|
||||
if (!Sleep) {
|
||||
// Acquired early.
|
||||
LOGMAN_THROW_A_FMT((Desired & WRITE_OWNED_BIT) == WRITE_OWNED_BIT, "Somehow acquired a write-lock without it being set!");
|
||||
return;
|
||||
}
|
||||
FutexWaitForWriteAvailable(Desired);
|
||||
|
||||
Expected = AtomicFutex.load(std::memory_order_relaxed);
|
||||
}
|
||||
}
|
||||
|
||||
void lock_shared() {
|
||||
// Try an uncontended lock first.
|
||||
if (try_lock_shared()) {
|
||||
return;
|
||||
}
|
||||
|
||||
// Try a quick WFE read-lock.
|
||||
if (Attempt_WFE_ReadLock()) {
|
||||
return;
|
||||
}
|
||||
|
||||
auto AtomicFutex = std::atomic_ref<uint32_t>(Futex);
|
||||
|
||||
uint32_t Expected = AtomicFutex.load(std::memory_order_relaxed);
|
||||
uint32_t Desired {};
|
||||
|
||||
while (true) {
|
||||
bool Sleep = false;
|
||||
do {
|
||||
if ((Expected & WRITE_OWNED_BIT) == 0 && (Expected & WRITE_WAITER_COUNT_MASK) == 0) {
|
||||
// If no write-owner and no write-waiting, try and acquire.
|
||||
|
||||
Desired = Expected + READ_OWNER_INCREMENT;
|
||||
LOGMAN_THROW_A_FMT((Desired & READ_OWNER_COUNT_MASK) != 0, "Overflow in read-owners!");
|
||||
Sleep = false;
|
||||
} else {
|
||||
// Waiting for lock to become available. Add to waiters.
|
||||
Desired = Expected | READ_WAITER_BIT;
|
||||
Sleep = true;
|
||||
}
|
||||
} while (AtomicFutex.compare_exchange_strong(Expected, Desired, std::memory_order_acq_rel, std::memory_order_acquire) == false);
|
||||
|
||||
if (!Sleep) {
|
||||
// Acquired early.
|
||||
LOGMAN_THROW_A_FMT((Desired & WRITE_OWNED_BIT) != WRITE_OWNED_BIT, "Somehow read-locked and got a write lock!");
|
||||
return;
|
||||
}
|
||||
|
||||
FutexWaitForReadAvailable(Desired);
|
||||
|
||||
Expected = AtomicFutex.load(std::memory_order_relaxed);
|
||||
}
|
||||
}
|
||||
|
||||
void unlock() {
|
||||
auto AtomicFutex = std::atomic_ref<uint32_t>(Futex);
|
||||
|
||||
uint32_t Expected = AtomicFutex.load(std::memory_order_relaxed);
|
||||
uint32_t Desired {};
|
||||
do {
|
||||
LOGMAN_THROW_A_FMT((Expected & WRITE_OWNED_BIT) == WRITE_OWNED_BIT, "Trying to write-unlock something not write-locked!");
|
||||
// Remove the exclusive lock bit.
|
||||
Desired = Expected & ~WRITE_OWNED_BIT;
|
||||
|
||||
// If no more writers, then make sure to clear the read-waiters bit as well.
|
||||
if ((Desired & WRITE_WAITER_COUNT_MASK) == 0) {
|
||||
Desired &= ~READ_WAITER_BIT;
|
||||
}
|
||||
} while (AtomicFutex.compare_exchange_strong(Expected, Desired, std::memory_order_acq_rel, std::memory_order_acquire) == false);
|
||||
|
||||
// If success, then `Expected` has old value. Containing `READ_WAITER_BIT` which was just masked off, and also `WRITE_WAITER_COUNT_MASK`.
|
||||
if ((Expected & WRITE_WAITER_COUNT_MASK)) {
|
||||
// Handle write-write handoff.
|
||||
FutexWakeWriter();
|
||||
} else if ((Expected & READ_WAITER_BIT)) {
|
||||
// Handle write-reader handoff.
|
||||
FutexWakeReaders();
|
||||
}
|
||||
}
|
||||
|
||||
void unlock_shared() {
|
||||
auto AtomicFutex = std::atomic_ref<uint32_t>(Futex);
|
||||
|
||||
uint32_t Desired {};
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
uint32_t Expected = AtomicFutex.load(std::memory_order_relaxed);
|
||||
do {
|
||||
LOGMAN_THROW_A_FMT((Expected & WRITE_OWNED_BIT) != WRITE_OWNED_BIT, "Trying to read-unlock something write-locked!");
|
||||
LOGMAN_THROW_A_FMT((Expected & READ_OWNER_COUNT_MASK) != 0, "Trying to read-unlock something not read-locked!");
|
||||
|
||||
// Decrement the shared counter.
|
||||
Desired = Expected - READ_OWNER_INCREMENT;
|
||||
} while (AtomicFutex.compare_exchange_strong(Expected, Desired, std::memory_order_acq_rel, std::memory_order_acquire) == false);
|
||||
#else
|
||||
Desired = AtomicFutex.fetch_sub(READ_OWNER_INCREMENT) - READ_OWNER_INCREMENT;
|
||||
#endif
|
||||
|
||||
// Handle read->write handoff if there are any waiting writers, and no readers left.
|
||||
if ((Desired & WRITE_WAITER_COUNT_MASK) && (Desired & READ_OWNER_COUNT_MASK) == 0) {
|
||||
FutexWakeWriter();
|
||||
}
|
||||
}
|
||||
|
||||
bool try_lock() {
|
||||
auto AtomicFutex = std::atomic_ref<uint32_t>(Futex);
|
||||
|
||||
uint32_t Expected = 0;
|
||||
|
||||
// Try and grab the owned bit.
|
||||
uint32_t Desired = WRITE_OWNED_BIT;
|
||||
|
||||
// try to CAS immediately.
|
||||
return AtomicFutex.compare_exchange_strong(Expected, Desired, std::memory_order_acq_rel, std::memory_order_acquire);
|
||||
}
|
||||
|
||||
// Can race with other threads trying to lock shared!
|
||||
bool try_lock_shared() {
|
||||
auto AtomicFutex = std::atomic_ref<uint32_t>(Futex);
|
||||
uint32_t Expected = AtomicFutex.load(std::memory_order_relaxed);
|
||||
|
||||
// Exclusively owned or has a list of waiting owners. Can't pass.
|
||||
if ((Expected & WRITE_OWNED_BIT) || (Expected & WRITE_WAITER_COUNT_MASK)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Try to add reader.
|
||||
uint32_t Desired = Expected + READ_OWNER_INCREMENT;
|
||||
LOGMAN_THROW_A_FMT((Desired & READ_OWNER_COUNT_MASK) != 0, "Overflow in read-owners!");
|
||||
|
||||
// Uncontended mutex check
|
||||
return AtomicFutex.compare_exchange_strong(Expected, Desired, std::memory_order_acq_rel, std::memory_order_acquire);
|
||||
}
|
||||
|
||||
#if !defined(_WIN32)
|
||||
// Initialize the internal mutex object to its default initializer state.
|
||||
// Should only ever be used in the child process when a Linux fork() has occured.
|
||||
void StealAndDropActiveLocks() {
|
||||
Futex = 0;
|
||||
}
|
||||
#endif
|
||||
|
||||
private:
|
||||
|
||||
#if !defined(_WIN32)
|
||||
void FutexWaitForWriteAvailable(uint32_t Expected) {
|
||||
::syscall(SYS_futex, &Futex, FUTEX_PRIVATE_FLAG | FUTEX_WAIT_BITSET, Expected, nullptr, nullptr, FUTEX_BITSET_WAIT_WRITERS);
|
||||
}
|
||||
|
||||
// Read-lock waiting for writers to drain out.
|
||||
void FutexWaitForReadAvailable(uint32_t Expected) {
|
||||
::syscall(SYS_futex, &Futex, FUTEX_PRIVATE_FLAG | FUTEX_WAIT_BITSET, Expected, nullptr, nullptr, FUTEX_BITSET_WAIT_READERS);
|
||||
}
|
||||
|
||||
// Read-Lock or Write-lock unlocked, wake one writer.
|
||||
// - Read->Write handoff.
|
||||
// - Write->Write handoff.
|
||||
void FutexWakeWriter() {
|
||||
::syscall(SYS_futex, &Futex, FUTEX_PRIVATE_FLAG | FUTEX_WAKE_BITSET, 1, nullptr, nullptr, FUTEX_BITSET_WAIT_WRITERS);
|
||||
}
|
||||
|
||||
// Write-lock unlocked, wake read-locks waiting.
|
||||
void FutexWakeReaders() {
|
||||
// Wake all readers.
|
||||
::syscall(SYS_futex, &Futex, FUTEX_PRIVATE_FLAG | FUTEX_WAKE_BITSET, INT_MAX, nullptr, nullptr, FUTEX_BITSET_WAIT_READERS);
|
||||
}
|
||||
#else
|
||||
// Writers wait for the full 32-bit futex.
|
||||
void FutexWaitForWriteAvailable(uint32_t Expected) {
|
||||
WaitOnAddress(&Futex, &Expected, sizeof(Futex), INFINITE);
|
||||
}
|
||||
|
||||
// Readers wait for Futex bits [31:16] to be zero.
|
||||
void FutexWaitForReadAvailable(uint32_t Expected) {
|
||||
auto ReadWaiterAddress = reinterpret_cast<uint8_t*>(&Futex) + 2;
|
||||
uint16_t smol_Expected = Expected >> 16;
|
||||
WaitOnAddress(ReadWaiterAddress, &smol_Expected, sizeof(smol_Expected), INFINITE);
|
||||
}
|
||||
|
||||
void FutexWakeWriter() {
|
||||
WakeByAddressSingle(&Futex);
|
||||
}
|
||||
|
||||
void FutexWakeReaders() {
|
||||
auto ReadWaiterAddress = reinterpret_cast<uint8_t*>(&Futex) + 2;
|
||||
WakeByAddressAll(ReadWaiterAddress);
|
||||
}
|
||||
#endif
|
||||
|
||||
// Reuse the SpinWaitLock WFE implementations for read/write lock acquiring with WFE.
|
||||
// Can't reuse the spin-lock directly as some bit-representations are different.
|
||||
// WFE-write-lock is less likely to occur the more read-lock threads are participating. Can still occur so good to try.
|
||||
// WFE-read-lock is actually quite likely to succeed.
|
||||
// Return: true if the lock was acquired.
|
||||
bool Attempt_WFE_WriteLock() {
|
||||
#ifdef _M_ARM_64
|
||||
const auto Begin = FEXCore::Utils::SpinWaitLock::GetCycleCounter();
|
||||
auto Now = Begin;
|
||||
const auto Duration = FEXCore::Utils::SpinWaitLock::CycleCounterFrequency / CYCLECOUNT_DIVISOR;
|
||||
|
||||
auto AtomicFutex = std::atomic_ref<uint32_t>(Futex);
|
||||
uint32_t Expected = AtomicFutex.load(std::memory_order_relaxed);
|
||||
|
||||
while ((Now - Begin) < Duration) {
|
||||
if (Expected == 0) {
|
||||
// Try and grab the owned bit.
|
||||
uint32_t Desired = WRITE_OWNED_BIT;
|
||||
|
||||
if (AtomicFutex.compare_exchange_strong(Expected, Desired, std::memory_order_acq_rel, std::memory_order_acquire)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
// One-shot attempt to wait for mask to be zero.
|
||||
Expected = FEXCore::Utils::SpinWaitLock::OneShotWFEBitComparison(&Futex, ~0U, 0U);
|
||||
Now = FEXCore::Utils::SpinWaitLock::GetCycleCounter();
|
||||
}
|
||||
#endif
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
// Return: true if the lock was acquired.
|
||||
bool Attempt_WFE_ReadLock() {
|
||||
#ifdef _M_ARM_64
|
||||
// Spin on a WFE for a short-amount of time, waiting for write-owned and writer-count to be zero.
|
||||
// - Attempt to acquire read-lock at that point.
|
||||
// - Don't add read-waiters bit on failure, return false.
|
||||
const auto Begin = FEXCore::Utils::SpinWaitLock::GetCycleCounter();
|
||||
auto Now = Begin;
|
||||
const auto Duration = FEXCore::Utils::SpinWaitLock::CycleCounterFrequency / CYCLECOUNT_DIVISOR;
|
||||
|
||||
auto AtomicFutex = std::atomic_ref<uint32_t>(Futex);
|
||||
uint32_t Expected = AtomicFutex.load(std::memory_order_relaxed);
|
||||
uint32_t Desired {};
|
||||
|
||||
while ((Now - Begin) < Duration) {
|
||||
if ((Expected & WRITE_OWNED_BIT) == 0 && (Expected & WRITE_WAITER_COUNT_MASK) == 0) {
|
||||
// If no write-owner and no write-waiting, try and acquire.
|
||||
|
||||
Desired = Expected + READ_OWNER_INCREMENT;
|
||||
LOGMAN_THROW_A_FMT((Desired & READ_OWNER_COUNT_MASK) != 0, "Overflow in read-owners!");
|
||||
if (AtomicFutex.compare_exchange_strong(Expected, Desired, std::memory_order_acq_rel, std::memory_order_acquire)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
// One-shot attempt to wait for mask to be zero.
|
||||
Expected = FEXCore::Utils::SpinWaitLock::OneShotWFEBitComparison(&Futex, WRITE_OWNED_BIT | WRITE_WAITER_COUNT_MASK, 0U);
|
||||
Now = FEXCore::Utils::SpinWaitLock::GetCycleCounter();
|
||||
}
|
||||
#endif
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
constexpr static uint32_t WRITE_OWNED_BIT = 1U << 31;
|
||||
constexpr static uint32_t READ_WAITER_BIT = 1U << 15;
|
||||
constexpr static uint32_t WRITE_WAITER_OFFSET = 16;
|
||||
constexpr static uint32_t WRITE_WAITER_INCREMENT = 1U << WRITE_WAITER_OFFSET;
|
||||
constexpr static uint32_t READ_OWNER_INCREMENT = 1;
|
||||
|
||||
// Count masks
|
||||
constexpr static uint32_t WRITE_WAITER_COUNT_MASK = 0x7FFFU << WRITE_WAITER_OFFSET;
|
||||
constexpr static uint32_t READ_OWNER_COUNT_MASK = 0x7FFFU;
|
||||
|
||||
// Independent futex bit-set masks.
|
||||
// Wait for readers to drain.
|
||||
constexpr static uint32_t FUTEX_BITSET_WAIT_READERS = 1U << 0;
|
||||
// Wait for writers to drain.
|
||||
constexpr static uint32_t FUTEX_BITSET_WAIT_WRITERS = 1U << 1;
|
||||
|
||||
// Only spin on WFE for 0.01ms (10k ns).
|
||||
constexpr static uint64_t CYCLECOUNT_DIVISOR = 1'000'000'000ULL / 10'000U;
|
||||
|
||||
// Layout:
|
||||
// Bits[31]: Write-lock bit.
|
||||
// Bits[30:16]: Write-waiter count.
|
||||
// Bits[15]: Read-waiter bit.
|
||||
// Bits[14:0]: Read-owner count.
|
||||
uint32_t Futex {};
|
||||
};
|
||||
} // namespace FEXCore::Utils::WritePriorityMutex
|
||||
@@ -103,28 +103,25 @@ static inline std::optional<fextl::string> EnumParser(const ArrayPairType& EnumP
|
||||
return fextl::fmt::format("{}", EnumMask);
|
||||
}
|
||||
|
||||
namespace DefaultValues {
|
||||
#define P(x) x
|
||||
#define OPT_BASE(type, group, enum, json, default) extern const P(type) P(enum);
|
||||
#define OPT_STR(group, enum, json, default) extern const std::string_view P(enum);
|
||||
#define OPT_STRARRAY(group, enum, json, default) OPT_STR(group, enum, json, default)
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
using StringArrayType = fextl::list<fextl::string>;
|
||||
|
||||
namespace Type {
|
||||
using StringArrayType = fextl::list<fextl::string>;
|
||||
#define OPT_BASE(type, group, enum, json, default) using P(enum) = P(type);
|
||||
#define OPT_STR(group, enum, json, default) using P(enum) = fextl::string;
|
||||
#define OPT_STRARRAY(group, enum, json, default) using P(enum) = StringArrayType;
|
||||
namespace detail {
|
||||
template<ConfigOption Option>
|
||||
struct ConfigOptionInfo;
|
||||
#define DEFINE_METAINFO(type, enum, default) \
|
||||
template<> \
|
||||
struct ConfigOptionInfo<ConfigOption::CONFIG_##enum> { \
|
||||
using Type = type; \
|
||||
static auto Default() { \
|
||||
extern default; \
|
||||
return enum; \
|
||||
} \
|
||||
};
|
||||
#define OPT_BASE(type, group, enum, json, default) DEFINE_METAINFO(type, enum, const type enum)
|
||||
#define OPT_STR(group, enum, json, default) DEFINE_METAINFO(fextl::string, enum, const std::string_view enum)
|
||||
#define OPT_STRARRAY(group, enum, json, default) DEFINE_METAINFO(StringArrayType, enum, const std::string_view enum)
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
} // namespace Type
|
||||
#define FEX_CONFIG_OPT(name, enum) \
|
||||
FEXCore::Config::Value<FEXCore::Config::DefaultValues::Type::enum> name { \
|
||||
FEXCore::Config::CONFIG_##enum, \
|
||||
FEXCore::Config::DefaultValues::enum \
|
||||
}
|
||||
|
||||
#undef P
|
||||
} // namespace DefaultValues
|
||||
} // namespace detail
|
||||
|
||||
FEX_DEFAULT_VISIBILITY void SetDataDirectory(std::string_view Path, bool Global);
|
||||
FEX_DEFAULT_VISIBILITY void SetConfigDirectory(const std::string_view Path, bool Global);
|
||||
@@ -135,8 +132,7 @@ FEX_DEFAULT_VISIBILITY const fextl::string& GetConfigDirectory(bool Global);
|
||||
FEX_DEFAULT_VISIBILITY const fextl::string& GetConfigFileLocation(bool Global = false);
|
||||
FEX_DEFAULT_VISIBILITY fextl::string GetApplicationConfig(const std::string_view Program, bool Global);
|
||||
|
||||
using LayerValue =
|
||||
std::variant< fextl::string, DefaultValues::Type::StringArrayType, uint8_t, int8_t, uint16_t, int16_t, uint32_t, int32_t, uint64_t, int64_t, bool >;
|
||||
using LayerValue = std::variant< fextl::string, StringArrayType, uint8_t, int8_t, uint16_t, int16_t, uint32_t, int32_t, uint64_t, int64_t, bool >;
|
||||
|
||||
using LayerOptions = fextl::unordered_map<ConfigOption, LayerValue>;
|
||||
|
||||
@@ -151,16 +147,16 @@ public:
|
||||
return OptionMap.find(Option) != OptionMap.end();
|
||||
}
|
||||
|
||||
std::optional<DefaultValues::Type::StringArrayType*> All(ConfigOption Option) {
|
||||
std::optional<StringArrayType*> All(ConfigOption Option) {
|
||||
const auto it = OptionMap.find(Option);
|
||||
if (it == OptionMap.end()) {
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
auto& Value = it->second;
|
||||
LOGMAN_THROW_A_FMT(std::holds_alternative<DefaultValues::Type::StringArrayType>(Value), "Tried to get config of invalid type!");
|
||||
LOGMAN_THROW_A_FMT(std::holds_alternative<StringArrayType>(Value), "Tried to get config of invalid type!");
|
||||
|
||||
return &std::get<DefaultValues::Type::StringArrayType>(Value);
|
||||
return &std::get<StringArrayType>(Value);
|
||||
}
|
||||
|
||||
std::optional<fextl::string*> Get(ConfigOption Option) {
|
||||
@@ -201,12 +197,12 @@ public:
|
||||
auto it = OptionMap.find(Option);
|
||||
if (it == OptionMap.end()) {
|
||||
// If the option didn't exist as a StringArrayType yet, emplace it.
|
||||
it = OptionMap.emplace(Option, DefaultValues::Type::StringArrayType {}).first;
|
||||
it = OptionMap.emplace(Option, StringArrayType {}).first;
|
||||
}
|
||||
|
||||
auto& Value = it->second;
|
||||
LOGMAN_THROW_A_FMT(std::holds_alternative<DefaultValues::Type::StringArrayType>(Value), "Tried to get config of invalid type!");
|
||||
std::get<DefaultValues::Type::StringArrayType>(Value).emplace_back(Data);
|
||||
LOGMAN_THROW_A_FMT(std::holds_alternative<StringArrayType>(Value), "Tried to get config of invalid type!");
|
||||
std::get<StringArrayType>(Value).emplace_back(Data);
|
||||
}
|
||||
|
||||
void Erase(ConfigOption Option) {
|
||||
@@ -236,7 +232,9 @@ FEX_DEFAULT_VISIBILITY fextl::string FindContainerPrefix();
|
||||
FEX_DEFAULT_VISIBILITY void AddLayer(fextl::unique_ptr<FEXCore::Config::Layer> _Layer);
|
||||
|
||||
FEX_DEFAULT_VISIBILITY bool Exists(ConfigOption Option);
|
||||
FEX_DEFAULT_VISIBILITY std::optional<DefaultValues::Type::StringArrayType*> All(ConfigOption Option);
|
||||
FEX_DEFAULT_VISIBILITY std::optional<StringArrayType*> All(ConfigOption Option);
|
||||
template<typename T>
|
||||
FEX_DEFAULT_VISIBILITY std::optional<T> GetConv(ConfigOption Option);
|
||||
FEX_DEFAULT_VISIBILITY std::optional<fextl::string*> Get(ConfigOption Option);
|
||||
FEX_DEFAULT_VISIBILITY void Set(ConfigOption Option, std::string_view Data);
|
||||
FEX_DEFAULT_VISIBILITY void Erase(ConfigOption Option);
|
||||
@@ -271,18 +269,18 @@ public:
|
||||
return ValueData;
|
||||
}
|
||||
|
||||
Value(T Value) requires (!std::is_same_v<T, DefaultValues::Type::StringArrayType>)
|
||||
Value(T Value) requires (!std::is_same_v<T, StringArrayType>)
|
||||
{
|
||||
ValueData = std::move(Value);
|
||||
}
|
||||
|
||||
// Array value types.
|
||||
Value(FEXCore::Config::ConfigOption Option, std::string_view) requires (std::is_same_v<T, DefaultValues::Type::StringArrayType>)
|
||||
Value(FEXCore::Config::ConfigOption Option, std::string_view) requires (std::is_same_v<T, StringArrayType>)
|
||||
{
|
||||
GetListIfExists(Option, &ValueData);
|
||||
}
|
||||
|
||||
DefaultValues::Type::StringArrayType& All() requires (std::is_same_v<T, DefaultValues::Type::StringArrayType>)
|
||||
StringArrayType& All() requires (std::is_same_v<T, StringArrayType>)
|
||||
{
|
||||
return ValueData;
|
||||
}
|
||||
@@ -293,6 +291,38 @@ private:
|
||||
static T GetIfExists(FEXCore::Config::ConfigOption Option, T Default);
|
||||
static T GetIfExists(FEXCore::Config::ConfigOption Option, std::string_view Default);
|
||||
|
||||
static void GetListIfExists(FEXCore::Config::ConfigOption Option, DefaultValues::Type::StringArrayType* List);
|
||||
static void GetListIfExists(FEXCore::Config::ConfigOption Option, StringArrayType* List);
|
||||
};
|
||||
|
||||
/**
|
||||
* Wrapper around Value that automatically picks the default for the given ConfigOption
|
||||
*/
|
||||
template<ConfigOption Option>
|
||||
struct FEX_DEFAULT_VISIBILITY Getter : public Value<typename detail::ConfigOptionInfo<Option>::Type> {
|
||||
using OptionInfo = detail::ConfigOptionInfo<Option>;
|
||||
Getter()
|
||||
: Value<typename OptionInfo::Type> {Option, OptionInfo::Default()} {}
|
||||
};
|
||||
|
||||
/**
|
||||
* Helper for reading a config value with caching.
|
||||
*
|
||||
* Typically this is used to declare class members so that the value is read
|
||||
* on construction of the parent.
|
||||
*/
|
||||
#define FEX_CONFIG_OPT(name, enum) FEXCore::Config::Getter<FEXCore::Config::ConfigOption::CONFIG_##enum> name {}
|
||||
|
||||
#define OPT_BASE(type, group, enum, json, default) \
|
||||
/** \
|
||||
* Helper for reading a config value. \
|
||||
* \
|
||||
* In contrast to FEX_CONFIG_OPT, this can be used in arbitrary expressions, \
|
||||
* at the expense of not caching the value. Use Getter instead if the value \
|
||||
* is read frequently. \
|
||||
*/ \
|
||||
inline auto Get_##enum() { \
|
||||
return Getter<FEXCore::Config::ConfigOption::CONFIG_##enum> {}; \
|
||||
}
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
|
||||
} // namespace FEXCore::Config
|
||||
@@ -1,10 +1,20 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/fextl/functional.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/set.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cstdint>
|
||||
#include <mutex>
|
||||
#include <optional>
|
||||
#include <shared_mutex>
|
||||
#include <span>
|
||||
#include <unistd.h>
|
||||
|
||||
namespace FEXCore {
|
||||
|
||||
@@ -20,8 +30,14 @@ namespace HLE {
|
||||
struct ExecutableFileInfo {
|
||||
~ExecutableFileInfo();
|
||||
|
||||
#if __clang_major__ < 16
|
||||
// Workaround for broken aggregate-initialization with std::piecewise_construct
|
||||
ExecutableFileInfo(fextl::unique_ptr<HLE::SourcecodeMap>, uint64_t, fextl::string);
|
||||
ExecutableFileInfo() = default;
|
||||
#endif
|
||||
|
||||
fextl::unique_ptr<HLE::SourcecodeMap> SourcecodeMap;
|
||||
fextl::string FileId;
|
||||
uint64_t FileId = 0;
|
||||
fextl::string Filename;
|
||||
};
|
||||
|
||||
@@ -33,10 +49,129 @@ struct ExecutableFileSectionInfo {
|
||||
uintptr_t FileStartVA;
|
||||
};
|
||||
|
||||
using CodeMapFileId = uint64_t;
|
||||
|
||||
/**
|
||||
* Code maps capture information required for offline code cache generation
|
||||
* and are written to disk during execution of FEX.
|
||||
*
|
||||
* Almost all CodeMap data will be an Entry that indicates blocks to be
|
||||
* compiled for cache generation. The reserved value `LoadExternalLibrary`
|
||||
* indicates that an instance of ExternalLibraryInfo follows (the entry data
|
||||
* itself should be skipped in that case).
|
||||
*/
|
||||
struct CodeMap {
|
||||
// Describes the location of an entry block compiled during execution
|
||||
struct FEX_PACKED Entry {
|
||||
CodeMapFileId FileId;
|
||||
uint32_t BlockOffset;
|
||||
};
|
||||
|
||||
// Describes an external library referenced during execution
|
||||
struct ExternalLibraryInfo {
|
||||
CodeMapFileId ExternalFileId;
|
||||
|
||||
// null-terminated file path; EITHER relative to the main executable OR an absolute path OR starting with a magic identifier:
|
||||
// - WINE/: Path to Wine/Proton installation
|
||||
// - WINEPREFIX/: Path to Wine/Proton prefix
|
||||
// - SLR/: Path to Steam Linux Runtime
|
||||
// At runtime, FEX will always dump absolute paths
|
||||
char Path[];
|
||||
// Followed by padding to a 4 byte boundary
|
||||
};
|
||||
|
||||
// Followed by ExternalLibraryInfo
|
||||
static constexpr Entry LoadExternalLibrary = {0xffff'ffff'ffff'ffff, 0xffff'ffff};
|
||||
|
||||
struct FEX_PACKED SetExecutableFileId {
|
||||
Entry Marker = {0xffff'ffff'ffff'ffff, 0xffff'fffe};
|
||||
CodeMapFileId ExecutableFileId;
|
||||
};
|
||||
|
||||
struct ParsedContents {
|
||||
fextl::string Filename;
|
||||
fextl::set<uint64_t> Blocks;
|
||||
bool IsExecutable = false;
|
||||
};
|
||||
|
||||
// Follows scheme fileid[-nomb]
|
||||
// The nomb ("no multiblock") suffix signifies that the code map is for use without multiblock, only.
|
||||
static fextl::string GetBaseFilename(const ExecutableFileInfo& MainExecutable, bool AddNombSuffix);
|
||||
|
||||
static fextl::map<CodeMapFileId, ParsedContents> ParseCodeMap(std::ifstream& File);
|
||||
};
|
||||
|
||||
struct CodeMapOpener {
|
||||
virtual ~CodeMapOpener() = default;
|
||||
virtual int OpenCodeMapFile() = 0;
|
||||
};
|
||||
|
||||
class CodeMapWriter {
|
||||
public:
|
||||
CodeMapWriter(CodeMapOpener&, bool OpenEagerly = false);
|
||||
~CodeMapWriter();
|
||||
|
||||
// Checks if writing is enabled. Calls to this functions may also be interpreted as signals that writes are about to happen
|
||||
bool IsWriteEnabled(const ExecutableFileSectionInfo&);
|
||||
|
||||
void ResetAfterFork() {
|
||||
if (CodeMapFD.value_or(-1) != -1) {
|
||||
close(CodeMapFD.value());
|
||||
CodeMapFD.reset();
|
||||
}
|
||||
BufferOffset = 0;
|
||||
KnownFileIds.clear();
|
||||
}
|
||||
|
||||
bool IsBackingFD(int FD) const {
|
||||
if (FD == CodeMapFD) {
|
||||
LogMan::Msg::DFmt("Hiding directory entry for code map FD");
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
void AppendBlock(const FEXCore::ExecutableFileSectionInfo&, uint64_t Entry);
|
||||
void AppendLibraryLoad(const FEXCore::ExecutableFileInfo&);
|
||||
void AppendSetMainExecutable(const FEXCore::ExecutableFileInfo&);
|
||||
|
||||
// Thread-safely commit any pending data to disk
|
||||
void Flush(size_t Offset);
|
||||
|
||||
private:
|
||||
// Queues data into an internal ring buffer.
|
||||
// Call Flush() to commit the data to disk.
|
||||
void AppendData(std::span<const std::byte> Data);
|
||||
|
||||
// Commit given data range to disk
|
||||
void Flush(size_t Offset, std::unique_lock<std::shared_mutex>&);
|
||||
|
||||
std::shared_mutex Mutex;
|
||||
fextl::vector<std::byte> Buffer;
|
||||
std::atomic<size_t> BufferOffset {0};
|
||||
|
||||
fextl::set<CodeMapFileId> KnownFileIds;
|
||||
|
||||
// std::nullopt: We haven't requested a CodeMapFD yet
|
||||
// value is -1: We requested a CodeMapFD but FEXServer told us not to write any data
|
||||
// other values: Code map writing is active
|
||||
std::optional<int> CodeMapFD;
|
||||
|
||||
CodeMapOpener& FileOpener;
|
||||
};
|
||||
|
||||
class AbstractCodeCache {
|
||||
public:
|
||||
virtual ~AbstractCodeCache() = default;
|
||||
|
||||
/**
|
||||
* Computes a unique identifier for the referenced binary file to be used for
|
||||
* generating the code map.
|
||||
* This identifier is independent of FEX build/runtime configuration and
|
||||
* stable across FEX updates.
|
||||
*/
|
||||
virtual uint64_t ComputeCodeMapId(std::string_view Filename, int FD) = 0;
|
||||
|
||||
/**
|
||||
* Loads a code cache from mapped memory and appends it to the current Core state.
|
||||
* TODO: Optionally recompiles all contained code blocks at runtime for validation.
|
||||
|
||||
@@ -42,9 +42,6 @@ enum OperatingMode {
|
||||
|
||||
using CodeRangeInvalidationFn = std::function<void(uint64_t start, uint64_t Length)>;
|
||||
|
||||
// Nested vector of guest block entrypoints
|
||||
using InvalidatedEntryAccumulator = fextl::vector<fextl::vector<uint64_t>>;
|
||||
|
||||
using CustomIREntrypointHandler = std::function<void(uintptr_t Entrypoint, IR::IREmitter*)>;
|
||||
|
||||
using ExitHandler = std::function<void(Core::InternalThreadState* Thread)>;
|
||||
@@ -139,10 +136,13 @@ public:
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) = 0;
|
||||
|
||||
virtual AbstractCodeCache& GetCodeCache() = 0;
|
||||
virtual void SetCodeMapWriter(fextl::unique_ptr<CodeMapWriter>) = 0;
|
||||
virtual void FlushAndCloseCodeMap() = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer = true) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void InvalidateGuestCodeRange(
|
||||
FEXCore::Core::InternalThreadState* Thread, InvalidatedEntryAccumulator& Accumulator, uint64_t Start, uint64_t Length) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void InvalidateCodeBuffersCodeRange(uint64_t Start, uint64_t Length) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void
|
||||
InvalidateThreadCachedCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::ForkableSharedMutex& GetCodeInvalidationMutex() = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void
|
||||
|
||||
@@ -104,6 +104,10 @@ struct CPUState {
|
||||
uint64_t avx_high[16][2];
|
||||
|
||||
uint64_t gregs[16] {};
|
||||
uint64_t L1Pointer {};
|
||||
uint64_t L1Mask {};
|
||||
uint64_t callret_sp {};
|
||||
uint64_t _pad1 {};
|
||||
XMMRegs xmm {};
|
||||
|
||||
// Raw segment register indexes
|
||||
@@ -116,8 +120,6 @@ struct CPUState {
|
||||
uint64_t gs_cached {};
|
||||
uint64_t fs_cached {};
|
||||
uint8_t flags[48] {};
|
||||
uint64_t callret_sp {};
|
||||
uint64_t _pad1 {};
|
||||
uint64_t mm[8][2] {};
|
||||
|
||||
// 32bit x86 state
|
||||
@@ -247,6 +249,8 @@ static_assert(offsetof(CPUState, xmm) % 32 == 0, "xmm needs to be 256-bit aligne
|
||||
static_assert(offsetof(CPUState, mm) % 16 == 0, "mm needs to be 128-bit aligned!");
|
||||
static_assert(offsetof(CPUState, gregs[15]) <= 504, "gregs maximum offset must be <= 504 for ldp/stp to work");
|
||||
static_assert(offsetof(CPUState, DeferredSignalRefCount) % 8 == 0, "Needs to be 8-byte aligned");
|
||||
static_assert(offsetof(CPUState, L1Pointer) <= 504, "This needs to be <= 504 for ldp");
|
||||
static_assert(offsetof(CPUState, L1Mask) == (offsetof(CPUState, L1Pointer) + 8), "These two variables are paired");
|
||||
|
||||
struct InternalThreadState;
|
||||
|
||||
@@ -349,13 +353,11 @@ struct JITPointers {
|
||||
uint64_t ExitFunctionLinker {};
|
||||
uint64_t ThreadStopHandlerSpillSRA {};
|
||||
uint64_t ThreadPauseHandlerSpillSRA {};
|
||||
uint64_t UnimplementedInstructionHandler {};
|
||||
uint64_t GuestSignal_SIGILL {};
|
||||
uint64_t GuestSignal_SIGTRAP {};
|
||||
uint64_t GuestSignal_SIGSEGV {};
|
||||
uint64_t SignalReturnHandler {};
|
||||
uint64_t SignalReturnHandlerRT {};
|
||||
uint64_t L1Pointer {};
|
||||
uint64_t L2Pointer {};
|
||||
/** @} */
|
||||
|
||||
@@ -368,8 +370,6 @@ struct JITPointers {
|
||||
// Process specific
|
||||
uint64_t LUDIV {};
|
||||
uint64_t LDIV {};
|
||||
uint64_t LUREM {};
|
||||
uint64_t LREM {};
|
||||
|
||||
// Thread Specific
|
||||
|
||||
@@ -378,8 +378,6 @@ struct JITPointers {
|
||||
* @{ */
|
||||
uint64_t LUDIVHandler {};
|
||||
uint64_t LDIVHandler {};
|
||||
uint64_t LUREMHandler {};
|
||||
uint64_t LREMHandler {};
|
||||
/** @} */
|
||||
} AArch64;
|
||||
|
||||
|
||||
@@ -40,6 +40,7 @@ struct HostFeatures {
|
||||
bool SupportsECV {};
|
||||
bool SupportsWFXT {};
|
||||
bool Supports3DNow {};
|
||||
bool SupportsSSE4a {};
|
||||
|
||||
// Float exception behaviour
|
||||
bool SupportsAFP {};
|
||||
|
||||
@@ -67,6 +67,10 @@ public:
|
||||
return Config;
|
||||
}
|
||||
|
||||
virtual uintptr_t GetThunkCallbackRET() const {
|
||||
return 0;
|
||||
}
|
||||
|
||||
protected:
|
||||
SignalDelegatorConfig Config;
|
||||
};
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
#include <FEXCore/Core/CoreState.h>
|
||||
#include <FEXCore/Utils/AllocatorHooks.h>
|
||||
#include <FEXCore/Utils/TypeDefines.h>
|
||||
#include <FEXCore/Utils/LongJump.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
@@ -80,7 +81,14 @@ private:
|
||||
static_assert(!std::is_move_constructible_v<NonMovableUniquePtr<int>>);
|
||||
static_assert(!std::is_move_assignable_v<NonMovableUniquePtr<int>>);
|
||||
|
||||
struct InternalThreadState : public FEXCore::Allocator::FEXAllocOperators {
|
||||
// Store used for unaligned LDAXR*/STLXR* emulation.
|
||||
struct UnalignedExclusiveStore {
|
||||
uint64_t Addr;
|
||||
uint64_t Store;
|
||||
uint8_t Size;
|
||||
};
|
||||
|
||||
struct alignas(FEXCore::Utils::FEX_PAGE_SIZE) InternalThreadState : public FEXCore::Allocator::FEXAllocOperators {
|
||||
FEXCore::Core::CpuStateFrame* const CurrentFrame = &BaseFrameState;
|
||||
|
||||
FEXCore::Context::Context* const CTX;
|
||||
@@ -101,6 +109,8 @@ struct InternalThreadState : public FEXCore::Allocator::FEXAllocOperators {
|
||||
// This pointer is owned by the frontend.
|
||||
FEXCore::SHMStats::ThreadStats* ThreadStats {};
|
||||
|
||||
UnalignedExclusiveStore ExclusiveStore;
|
||||
|
||||
///< Data pointer for exclusive use by the frontend
|
||||
void* FrontendPtr;
|
||||
|
||||
@@ -109,6 +119,10 @@ struct InternalThreadState : public FEXCore::Allocator::FEXAllocOperators {
|
||||
// The low address of the call-ret stack allocation (not including guard pages)
|
||||
void* CallRetStackBase {};
|
||||
|
||||
uintptr_t JITGuardPage {};
|
||||
uint64_t JITGuardOverflowArgument {};
|
||||
FEXCore::UncheckedLongJump::JumpBuf RestartJump;
|
||||
|
||||
// BaseFrameState should always be at the end, directly before the interrupt fault page
|
||||
alignas(16) FEXCore::Core::CpuStateFrame BaseFrameState {};
|
||||
|
||||
@@ -116,8 +130,9 @@ struct InternalThreadState : public FEXCore::Allocator::FEXAllocOperators {
|
||||
alignas(FEXCore::Utils::FEX_PAGE_SIZE) uint8_t InterruptFaultPage[FEXCore::Utils::FEX_PAGE_SIZE];
|
||||
};
|
||||
static_assert(std::is_standard_layout_v<FEXCore::Core::InternalThreadState>);
|
||||
static_assert(
|
||||
(offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState)) < 4096,
|
||||
"Fault page is outside of immediate range from CPU state");
|
||||
static_assert((offsetof(FEXCore::Core::InternalThreadState, InterruptFaultPage) - offsetof(FEXCore::Core::InternalThreadState, BaseFrameState)) <
|
||||
FEXCore::Utils::FEX_PAGE_SIZE,
|
||||
"Fault page is outside of immediate range from CPU state");
|
||||
static_assert(sizeof(FEXCore::Core::InternalThreadState) == (FEXCore::Utils::FEX_PAGE_SIZE * 2));
|
||||
|
||||
} // namespace FEXCore::Core
|
||||
@@ -84,6 +84,6 @@ private:
|
||||
|
||||
class SourcecodeResolver {
|
||||
public:
|
||||
virtual fextl::unique_ptr<SourcecodeMap> GenerateMap(const std::string_view& GuestBinaryFile, const std::string_view& GuestBinaryFileId) = 0;
|
||||
virtual fextl::unique_ptr<SourcecodeMap> GenerateMap(std::string_view GuestBinaryFile, std::string_view GuestBinaryFileId) = 0;
|
||||
};
|
||||
} // namespace FEXCore::HLE
|
||||
@@ -54,10 +54,6 @@ public:
|
||||
virtual ~SyscallHandler() = default;
|
||||
|
||||
virtual uint64_t HandleSyscall(FEXCore::Core::CpuStateFrame* Frame, FEXCore::HLE::SyscallArguments* Args) = 0;
|
||||
virtual SyscallABI GetSyscallABI(uint64_t Syscall) = 0;
|
||||
virtual FEXCore::IR::SyscallFlags GetSyscallFlags(uint64_t Syscall) const {
|
||||
return FEXCore::IR::SyscallFlags::DEFAULT;
|
||||
}
|
||||
|
||||
SyscallOSABI GetOSABI() const {
|
||||
return OSABI;
|
||||
|
||||
@@ -3,31 +3,12 @@
|
||||
|
||||
#include <FEXCore/Utils/EnumOperators.h>
|
||||
|
||||
#include <compare>
|
||||
#include <cstdint>
|
||||
#include <cstring>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
enum class SyscallFlags : uint8_t {
|
||||
DEFAULT = 0,
|
||||
// Syscalldoesn't care about CPUState being serialized up to the syscall instruction.
|
||||
// Means dead code elimination can optimize through a syscall operation.
|
||||
OPTIMIZETHROUGH = 1 << 0,
|
||||
// Syscall only reads the passed in arguments. Doesn't read CPUState.
|
||||
NOSYNCSTATEONENTRY = 1 << 1,
|
||||
// Syscall doesn't return. Code generation after syscall return can be removed.
|
||||
NORETURN = 1 << 2,
|
||||
// Syscall doesn't have any side-effects, so if the result isn't used then it can be removed.
|
||||
NOSIDEEFFECTS = 1 << 3,
|
||||
// Syscall doesn't return a result.
|
||||
// Means the resulting register shouldn't be written (Usually RAX).
|
||||
// Usually used with !NOSYNCSTATEONENTRY, so the syscall can modify CPU state entirely.
|
||||
// Then on return FEXCore picks up the new state.
|
||||
NORETURNEDRESULT = 1 << 4,
|
||||
};
|
||||
|
||||
FEX_DEF_NUM_OPS(SyscallFlags)
|
||||
|
||||
// This enum of named vector constants are linked to an array in CPUBackend.cpp.
|
||||
// This is used with the IROp `LoadNamedVectorConstant` to load a vector constant
|
||||
// that would otherwise be costly to materialize.
|
||||
@@ -96,15 +77,7 @@ enum IndexNamedVectorConstant : uint8_t {
|
||||
|
||||
struct SHA256Sum final {
|
||||
uint8_t data[32];
|
||||
[[nodiscard]]
|
||||
bool operator<(const SHA256Sum& rhs) const {
|
||||
return memcmp(data, rhs.data, sizeof(data)) < 0;
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
bool operator==(const SHA256Sum& rhs) const {
|
||||
return memcmp(data, rhs.data, sizeof(data)) == 0;
|
||||
}
|
||||
[[nodiscard]] auto operator<=>(const SHA256Sum&) const noexcept = default;
|
||||
};
|
||||
|
||||
typedef void ThunkedFunction(void* ArgsRv);
|
||||
|
||||
@@ -82,12 +82,15 @@ inline bool VirtualProtect(void* Ptr, size_t Size, ProtectOptions options) {
|
||||
return ::VirtualProtect(Ptr, Size, prot, nullptr) == 0;
|
||||
}
|
||||
|
||||
inline void VirtualName(const char*, void*, size_t) {}
|
||||
|
||||
#else
|
||||
using MMAP_Hook = void* (*)(void*, size_t, int, int, int, off_t);
|
||||
using MUNMAP_Hook = int (*)(void*, size_t);
|
||||
|
||||
FEX_DEFAULT_VISIBILITY extern MMAP_Hook mmap;
|
||||
FEX_DEFAULT_VISIBILITY extern MUNMAP_Hook munmap;
|
||||
FEX_DEFAULT_VISIBILITY extern void VirtualName(const char* Name, void* Ptr, size_t Size);
|
||||
|
||||
// All commit parameters are ignored here, they are unnecessary as Linux supports overcommit
|
||||
|
||||
|
||||
@@ -12,8 +12,6 @@ struct InternalThreadState;
|
||||
|
||||
namespace FEXCore::ArchHelpers::Arm64 {
|
||||
enum class UnalignedHandlerType {
|
||||
///< Don't backpatch code, instead handle inside SIGBUS handler.
|
||||
Paranoid,
|
||||
///< Backpatch unaligned access to half-barrier based atomic.
|
||||
HalfBarrier,
|
||||
///< Backpatch unaligned access to non-atomic.
|
||||
@@ -26,7 +24,7 @@ enum class UnalignedHandlerType {
|
||||
* This is an OS agnostic handler where the frontend must provide FEXCore with the information necessary to know if this is safe.
|
||||
* This does not check if the PC is within a JIT code buffer, the frontend must provide that safety with `CPUBackend::IsAddressInCodeBuffer`.
|
||||
*
|
||||
* @param ParanoidTSO If the unaligned fault needs to handled directly or can be backpatched.
|
||||
* @param HandleType Type of TSO handling to use.
|
||||
* @param ProgramCounter The location in memory for the instruction that did the access
|
||||
* @param GPRs The array of GPRs from the signal context. This will be modified and the host context needs to be updated on signal return.
|
||||
*
|
||||
@@ -34,6 +32,6 @@ enum class UnalignedHandlerType {
|
||||
* by. FEXCore will return a positive or negative offset depending on internal handling.
|
||||
*/
|
||||
[[nodiscard]]
|
||||
FEX_DEFAULT_VISIBILITY std::optional<int32_t>
|
||||
HandleUnalignedAccess(FEXCore::Core::InternalThreadState* Thread, UnalignedHandlerType HandleType, uintptr_t ProgramCounter, uint64_t* GPRs);
|
||||
FEX_DEFAULT_VISIBILITY std::optional<int32_t> HandleUnalignedAccess(
|
||||
FEXCore::Core::InternalThreadState* Thread, UnalignedHandlerType HandleType, uintptr_t ProgramCounter, uint64_t* GPRs, bool IsJIT = true);
|
||||
} // namespace FEXCore::ArchHelpers::Arm64
|
||||
@@ -54,6 +54,13 @@ namespace FEXCore {
|
||||
return static_cast<T>(key) == 0; \
|
||||
}
|
||||
|
||||
// Macro that defines a fmt formatter for a reasonable case where an enum
|
||||
// is formatted as a purely integral type based on its underlying type.
|
||||
#define FEX_DEFINE_ENUM_FMT_PASSTHROUGH(type) \
|
||||
constexpr auto format_as(type t) { \
|
||||
return FEXCore::ToUnderlying(t); \
|
||||
}
|
||||
|
||||
// Equivalent to C++23's std::to_underlying.
|
||||
template<typename Enum>
|
||||
[[nodiscard]]
|
||||
|
||||
@@ -4,8 +4,10 @@
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
// It's longjump without glibc fortification checks.
|
||||
namespace FEXCore::LongJump {
|
||||
// Reimplementation of longjmp without glibc fortification checks.
|
||||
// This is useful when false positives need to be avoided or when using
|
||||
// a libc implementation that does not implement std::longjmp.
|
||||
namespace FEXCore::UncheckedLongJump {
|
||||
// JumpBuf definition needs to be public because the frontend needs to understand it.
|
||||
#if defined(_M_ARM_64)
|
||||
struct JumpBuf {
|
||||
@@ -32,5 +34,6 @@ struct JumpBuf {
|
||||
#endif
|
||||
|
||||
[[nodiscard]] FEX_DEFAULT_VISIBILITY uint64_t SetJump(JumpBuf& Buffer);
|
||||
[[noreturn]] FEX_DEFAULT_VISIBILITY void LongJump(JumpBuf& Buffer, uint64_t Value);
|
||||
} // namespace FEXCore::LongJump
|
||||
[[noreturn]] FEX_DEFAULT_VISIBILITY void LongJump(const JumpBuf& Buffer, uint64_t Value);
|
||||
FEX_DEFAULT_VISIBILITY void ManuallyLoadJumpBuf(const JumpBuf& Buffer, uint64_t Value, uint64_t* GPRs, __uint128_t* FPRs, uint64_t* PC);
|
||||
} // namespace FEXCore::UncheckedLongJump
|
||||
@@ -0,0 +1,54 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#ifndef _WIN32
|
||||
#include <linux/prctl.h>
|
||||
#include <sys/mman.h>
|
||||
#include <sys/user.h>
|
||||
#include <sys/prctl.h>
|
||||
|
||||
#ifndef PR_SET_VMA
|
||||
#define PR_SET_VMA 0x53564d41
|
||||
#endif
|
||||
|
||||
#ifndef PR_SET_VMA_ANON_NAME
|
||||
#define PR_SET_VMA_ANON_NAME 0
|
||||
#endif
|
||||
|
||||
#ifndef PR_GET_MEM_MODEL
|
||||
#define PR_GET_MEM_MODEL 0x6d4d444c
|
||||
#endif
|
||||
#ifndef PR_SET_MEM_MODEL
|
||||
#define PR_SET_MEM_MODEL 0x4d4d444c
|
||||
#endif
|
||||
#ifndef PR_SET_MEM_MODEL_DEFAULT
|
||||
#define PR_SET_MEM_MODEL_DEFAULT 0
|
||||
#endif
|
||||
#ifndef PR_SET_MEM_MODEL_TSO
|
||||
#define PR_SET_MEM_MODEL_TSO 1
|
||||
#endif
|
||||
|
||||
#ifndef PR_GET_COMPAT_INPUT
|
||||
#define PR_GET_COMPAT_INPUT 0x63494e50
|
||||
#endif
|
||||
#ifndef PR_SET_COMPAT_INPUT
|
||||
#define PR_SET_COMPAT_INPUT 0x43494e50
|
||||
#endif
|
||||
#ifndef PR_SET_COMPAT_INPUT_DISABLE
|
||||
#define PR_SET_COMPAT_INPUT_DISABLE 0
|
||||
#endif
|
||||
#ifndef PR_SET_COMPAT_INPUT_ENABLE
|
||||
#define PR_SET_COMPAT_INPUT_ENABLE 1
|
||||
#endif
|
||||
|
||||
#ifndef PR_GET_SHADOW_STACK_STATUS
|
||||
#define PR_GET_SHADOW_STACK_STATUS 74
|
||||
#endif
|
||||
#ifndef PR_LOCK_SHADOW_STACK_STATUS
|
||||
#define PR_LOCK_SHADOW_STACK_STATUS 76
|
||||
#endif
|
||||
#ifndef PR_SHADOW_STACK_ENABLE
|
||||
#define PR_SHADOW_STACK_ENABLE (1ULL << 0)
|
||||
#endif
|
||||
|
||||
#endif // ifndef _WIN32
|
||||
Loaded 100 of 727 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user