mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-07 02:00:16 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e9e5b6fb0b | ||
|
|
68cb6e61d1 | ||
|
|
5d0b2060e2 | ||
|
|
b7d05a65c7 | ||
|
|
93fe2fe06c | ||
|
|
21eb6e03c7 | ||
|
|
1b53337925 | ||
|
|
52e5b8ccd9 | ||
|
|
8c8a8c84df | ||
|
|
0d6837f1a1 | ||
|
|
7ef3cb88f9 | ||
|
|
617977357a | ||
|
|
5a53c9231b | ||
|
|
a996e5300e | ||
|
|
25b2af14fd | ||
|
|
76059949ea | ||
|
|
6e15c9c213 | ||
|
|
01fcca884b | ||
|
|
91bd3aa62a | ||
|
|
7a0119b092 | ||
|
|
fa42c1616e | ||
|
|
18783948f7 | ||
|
|
969d2e4b6a | ||
|
|
7ceaf56407 | ||
|
|
7c52375267 | ||
|
|
a4ea792d03 | ||
|
|
cec98637c9 | ||
|
|
d5026f5815 | ||
|
|
1e4456ec40 | ||
|
|
e285c7c9a0 | ||
|
|
8244c7f2a6 | ||
|
|
747c5e17f8 | ||
|
|
52ce027e9b | ||
|
|
0fb3903889 | ||
|
|
2832afd0f7 | ||
|
|
4af42477a7 | ||
|
|
4786aa479c | ||
|
|
8a049aa0c3 | ||
|
|
64bac687b5 | ||
|
|
6e05494ad0 | ||
|
|
4d13f5d97d | ||
|
|
1ad928dc2c | ||
|
|
c235ab883d | ||
|
|
4bf3a0888b | ||
|
|
6834fe32e4 | ||
|
|
d196709162 | ||
|
|
475a6a38fa | ||
|
|
0f748bd724 | ||
|
|
47bd331239 | ||
|
|
173b70d191 | ||
|
|
96aa0a844e | ||
|
|
48121cd585 | ||
|
|
52d7efda10 | ||
|
|
30ab4d3f58 | ||
|
|
f9fa0f25e2 | ||
|
|
eebcbfda96 | ||
|
|
1483ddb538 | ||
|
|
d2bca9b997 | ||
|
|
8f48021a63 | ||
|
|
1da5d7f2e6 | ||
|
|
0a5db6c404 | ||
|
|
4681061011 | ||
|
|
4e9eeb1a16 | ||
|
|
95d728b634 | ||
|
|
187e551a0c | ||
|
|
7c4e4c4409 | ||
|
|
bdd8df2100 | ||
|
|
1d1bdfb96d | ||
|
|
2fc6542d15 | ||
|
|
5e88be3c99 | ||
|
|
be52844ce0 | ||
|
|
d629884147 | ||
|
|
64c72430bf | ||
|
|
1b13e8db7d | ||
|
|
1ce0ea8f3b | ||
|
|
8557656259 | ||
|
|
c87c04db59 | ||
|
|
f5262446a0 | ||
|
|
0a1820d444 | ||
|
|
a255813e99 | ||
|
|
4c6336b560 | ||
|
|
145c7799a5 | ||
|
|
91df503e55 | ||
|
|
3c417cee84 | ||
|
|
857780b46e | ||
|
|
cb34aa3dee | ||
|
|
0d74777954 | ||
|
|
092a023900 | ||
|
|
b57dd5c3aa | ||
|
|
0bea508935 | ||
|
|
9c175da0e2 | ||
|
|
cc359f09ad | ||
|
|
dedeba0575 | ||
|
|
dc3ccbcc43 | ||
|
|
ebd9b49dbe | ||
|
|
43e21ad749 | ||
|
|
1b3beec085 | ||
|
|
c88f2e0730 | ||
|
|
1645f6e582 | ||
|
|
af35e18979 | ||
|
|
012c0ae062 | ||
|
|
dc8f063a4b | ||
|
|
759747b3c5 | ||
|
|
4c6d26f167 | ||
|
|
7562ff2308 | ||
|
|
e2baa1106b | ||
|
|
68880fd66b | ||
|
|
62eef6c548 | ||
|
|
cfdc1bda47 | ||
|
|
f9a9645ef3 | ||
|
|
8fc903cc51 | ||
|
|
c81b1c3432 | ||
|
|
db9ba16a53 | ||
|
|
0be68a54d0 | ||
|
|
6e140e9ff2 | ||
|
|
d793f7295a | ||
|
|
8f6018a1b4 | ||
|
|
0a97d4456d | ||
|
|
c2dbe8309b | ||
|
|
06d66a0188 | ||
|
|
ef254da7ca | ||
|
|
162dcf3cd0 | ||
|
|
dcc47074d1 | ||
|
|
4dfee31012 | ||
|
|
3749fa6569 | ||
|
|
94273fbf4f | ||
|
|
162bbf2937 | ||
|
|
296adf1830 | ||
|
|
a86109280a | ||
|
|
d83960d4dd | ||
|
|
bba97823a8 | ||
|
|
abf5e8c6a5 | ||
|
|
8d288300c4 | ||
|
|
0caf263c77 | ||
|
|
29dc77c44b | ||
|
|
4c1f53c1ff | ||
|
|
5821175ddb | ||
|
|
003c88e537 | ||
|
|
27a1ebc2f5 | ||
|
|
e689c6fcfa | ||
|
|
61e905a339 | ||
|
|
10fdcaa109 | ||
|
|
ba672f868c | ||
|
|
d6697fce32 | ||
|
|
9c3a843df7 | ||
|
|
8defa2b55f | ||
|
|
77c88ffe53 | ||
|
|
5b261b0d2e | ||
|
|
2ce15ddc89 | ||
|
|
3d1b55383e | ||
|
|
69deaa0976 | ||
|
|
fc72fa9e5f | ||
|
|
6baee3b7c1 | ||
|
|
362a5a019a | ||
|
|
d529a58893 | ||
|
|
c77ea3b392 | ||
|
|
20aaad15e4 | ||
|
|
6f2452e2ea | ||
|
|
b34401bb33 | ||
|
|
dff9868e9a | ||
|
|
036a196984 | ||
|
|
a7ac4fa6e4 | ||
|
|
82295b2943 | ||
|
|
eca80b6046 | ||
|
|
02fc02964b | ||
|
|
e8fcb070b3 | ||
|
|
597da88035 | ||
|
|
1e829a47fa | ||
|
|
e633ef7cf5 | ||
|
|
ff3b40400c | ||
|
|
801106faf4 | ||
|
|
536b2ed495 | ||
|
|
072f027885 | ||
|
|
754bc18813 | ||
|
|
ef257418d3 | ||
|
|
842b71cf83 | ||
|
|
8d8b64d2b7 | ||
|
|
85c6ef8097 | ||
|
|
2be16d9054 | ||
|
|
41b3c52663 | ||
|
|
f1f50b7a98 | ||
|
|
d7200e2a1e | ||
|
|
3f884fe2d0 | ||
|
|
1d453f10a0 | ||
|
|
5ebd21ca6a | ||
|
|
fb34b507a1 | ||
|
|
2c91b5cff6 | ||
|
|
79f7dcbaa5 | ||
|
|
f7b7997c77 | ||
|
|
0674dfab0a | ||
|
|
6979dc9c4e | ||
|
|
fe5f17d92e | ||
|
|
be71886990 | ||
|
|
5043e5fbc0 | ||
|
|
2acfde3cad | ||
|
|
c1eeeaf688 | ||
|
|
f2b3229a87 | ||
|
|
e20bfc0701 | ||
|
|
4caee5c9be | ||
|
|
46d3f283b8 | ||
|
|
cee5512a56 | ||
|
|
8c53a373bf | ||
|
|
f73176d5dc | ||
|
|
80ae3e632d | ||
|
|
ed75c19324 | ||
|
|
4a7fa7f2bc | ||
|
|
98f51c47fa | ||
|
|
724a8e13bf | ||
|
|
d9b52fd67d | ||
|
|
d3a2795106 | ||
|
|
3cd6c2d91a | ||
|
|
b86abfbccf | ||
|
|
ee66985ae0 | ||
|
|
daeba0625f | ||
|
|
5e6af25194 | ||
|
|
7f4528a6b0 | ||
|
|
776b7674e4 | ||
|
|
24cb2610a2 | ||
|
|
54b7a43b95 | ||
|
|
1b1e9e0fb5 | ||
|
|
ead43c6a51 | ||
|
|
e49de77225 | ||
|
|
8f4fe39b7d | ||
|
|
f250509718 | ||
|
|
6179c5a13e | ||
|
|
9e14a83442 | ||
|
|
95bfd003d2 | ||
|
|
08ca43c3c4 | ||
|
|
96b428dcaa | ||
|
|
421214e723 | ||
|
|
1a18bbb966 | ||
|
|
699c3f5762 | ||
|
|
da0a1710c0 | ||
|
|
c1205eb809 | ||
|
|
31b7cd77e9 | ||
|
|
9def04c705 | ||
|
|
68a2441e65 | ||
|
|
4ac0dec568 | ||
|
|
1d7b4bb522 | ||
|
|
22f95e627d | ||
|
|
599b64e975 | ||
|
|
5c27febbc2 | ||
|
|
58c93568f6 | ||
|
|
491e4e2c23 | ||
|
|
1ed2e24fba | ||
|
|
1fbb8bd78f | ||
|
|
b953433404 | ||
|
|
eae950be16 | ||
|
|
7e6bb04db1 | ||
|
|
68555546bc | ||
|
|
716cac35a8 | ||
|
|
9722c4c5a4 | ||
|
|
e8c0e19afc | ||
|
|
c559fec959 | ||
|
|
8d2fabe705 | ||
|
|
6455c4817a | ||
|
|
5dbd1b8dc2 | ||
|
|
7765bbc7b8 | ||
|
|
2283c73fae | ||
|
|
5fef0c29aa | ||
|
|
d387c46aab | ||
|
|
3bb7f9d6b5 | ||
|
|
70d54122b2 | ||
|
|
4e266cf9fd | ||
|
|
ddd6dbfdcc | ||
|
|
7f2557e322 | ||
|
|
810c7d926c | ||
|
|
04c325661c | ||
|
|
0121e858f2 | ||
|
|
2c3361be9e | ||
|
|
2d800b2627 | ||
|
|
55ed3e0549 | ||
|
|
55d084ebb0 | ||
|
|
457dc5dd90 | ||
|
|
98eda5e163 | ||
|
|
592935790e | ||
|
|
92d0344d6a | ||
|
|
9327435f97 | ||
|
|
573f339647 | ||
|
|
c1f18951ab | ||
|
|
8a4bfba47c | ||
|
|
69ea03f0eb | ||
|
|
462feec2a6 | ||
|
|
15f5fe658b | ||
|
|
052aa4317b | ||
|
|
debcb0e047 | ||
|
|
baf04b6a41 | ||
|
|
cc85a6a722 | ||
|
|
e72fa02897 | ||
|
|
8a4c5bcc65 | ||
|
|
7a13a24c05 | ||
|
|
5a53931b92 | ||
|
|
f9b352a093 | ||
|
|
f444b03317 | ||
|
|
ed05846dd0 | ||
|
|
f609990f90 | ||
|
|
d2032da452 | ||
|
|
8047007a7a | ||
|
|
1f7e82ea09 | ||
|
|
35c52f20f9 | ||
|
|
17c82c22a6 | ||
|
|
df03a7b101 | ||
|
|
c859540d7e | ||
|
|
20794593e7 | ||
|
|
a80a2bf569 | ||
|
|
e86a792189 | ||
|
|
d6c9b549df | ||
|
|
1a4d5a1abb | ||
|
|
c3e123df25 | ||
|
|
cac798574a | ||
|
|
1506a19229 | ||
|
|
51861234bc | ||
|
|
677b72c9a5 | ||
|
|
71a8c66c95 | ||
|
|
3372e9bdbb | ||
|
|
df0723e14b | ||
|
|
7ee6fc0d7f | ||
|
|
bf773452ac | ||
|
|
95dbccc0ab | ||
|
|
e5189d63a2 | ||
|
|
7d5442357a | ||
|
|
9dcc1deec0 | ||
|
|
66d4206cd7 | ||
|
|
f39163b1e1 | ||
|
|
01837b3ad6 | ||
|
|
bdb68840e3 | ||
|
|
4e2dcf3298 | ||
|
|
628f825416 | ||
|
|
a80327f6df | ||
|
|
16f7002222 | ||
|
|
c9712e45cb | ||
|
|
a082161d72 | ||
|
|
9017325c95 | ||
|
|
ae536e44d7 | ||
|
|
7679485cc3 | ||
|
|
cb8bf1add6 | ||
|
|
a69c457715 | ||
|
|
7c4729678b | ||
|
|
537562fab7 | ||
|
|
f8721992c2 | ||
|
|
e652399fd3 | ||
|
|
e7c92c43a0 | ||
|
|
9b5e1c44c8 | ||
|
|
755600c371 | ||
|
|
bec8b70e5d | ||
|
|
bef8ddde48 | ||
|
|
fe06f1b151 | ||
|
|
92a15e00c7 | ||
|
|
2997257d6d | ||
|
|
7ceadc6b5b | ||
|
|
8c41e8f7d8 | ||
|
|
784b3064fc | ||
|
|
b3bc1e23cc | ||
|
|
1a9b6a89f4 | ||
|
|
21bf35d211 | ||
|
|
9473025b18 | ||
|
|
02f15f4099 | ||
|
|
e007789ced | ||
|
|
a2b165043c | ||
|
|
5b5808218b | ||
|
|
41ec987f3e | ||
|
|
0f4a5edf4f | ||
|
|
03f73531d3 | ||
|
|
69181d438c | ||
|
|
a2cbfccb3b | ||
|
|
4e01452a65 | ||
|
|
cc7a56b1a6 | ||
|
|
0b0dd3891e | ||
|
|
8bc33e95c1 | ||
|
|
96a0364a86 | ||
|
|
c0a783997d | ||
|
|
f78537109d | ||
|
|
0c156ed6f9 | ||
|
|
920913cf80 | ||
|
|
8840b2154c | ||
|
|
e02be8073e | ||
|
|
f75d3550b4 | ||
|
|
802c588695 | ||
|
|
fd962f40d7 | ||
|
|
a9b660af69 | ||
|
|
fd5c36ba9c | ||
|
|
5be798e9e6 | ||
|
|
09997cff9c | ||
|
|
c9d1f0d75a | ||
|
|
95b7592241 | ||
|
|
1dc4f8c429 | ||
|
|
1d7fcdb54a | ||
|
|
45d3b83143 | ||
|
|
d97fa9af14 | ||
|
|
c9101d3f68 | ||
|
|
52f64a0c7b | ||
|
|
737f917838 | ||
|
|
a6c6248bcb | ||
|
|
5646428640 | ||
|
|
0c8df2beaf | ||
|
|
de0f3984e9 | ||
|
|
ada226bbb4 | ||
|
|
4bc5a09e62 | ||
|
|
5704b5f23f | ||
|
|
6017a9135a | ||
|
|
6ef6d9c391 | ||
|
|
0ad6f98a8c | ||
|
|
1354f92cc5 | ||
|
|
8b90caad95 | ||
|
|
3a4a965347 | ||
|
|
45cdab2ac3 | ||
|
|
b89dc56ae1 | ||
|
|
d675b4af6f | ||
|
|
d75fb38344 | ||
|
|
4cb385a27b | ||
|
|
9a4fdd8059 | ||
|
|
363411f0c7 | ||
|
|
5bc418407c | ||
|
|
46e2dc7498 | ||
|
|
4a54197868 | ||
|
|
3f214dd244 | ||
|
|
61ca651fe1 | ||
|
|
cd0a340d29 | ||
|
|
77e8be1215 | ||
|
|
182010ca97 | ||
|
|
f7c663240e | ||
|
|
e9244680aa | ||
|
|
82b4aef30d | ||
|
|
22919a5b65 | ||
|
|
00dc373bb9 | ||
|
|
e593237670 | ||
|
|
0a4bf10ba5 | ||
|
|
f47caf48c6 | ||
|
|
5674d3a871 | ||
|
|
fde64aedf7 | ||
|
|
e03b859c20 | ||
|
|
ce4e991a6e | ||
|
|
7d822ba1c8 | ||
|
|
f90dcd2eb1 | ||
|
|
adbdd33ece | ||
|
|
06250d806d | ||
|
|
613ed559e7 | ||
|
|
8ac3841946 | ||
|
|
458259bf47 | ||
|
|
dc65a5ef8c | ||
|
|
2fc529d5b7 | ||
|
|
ea489567da | ||
|
|
ed69eb9f6f | ||
|
|
6eae064511 |
No files matched your search
@@ -25,7 +25,7 @@ jobs:
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
- uses: actions/checkout@v3
|
||||
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
@@ -172,6 +172,17 @@ jobs:
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_APITests.log || true
|
||||
|
||||
- name: FEXCore APITest tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target fexcore_apitests
|
||||
|
||||
- name: FEXCore APITest Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_FEXCoreAPITests.log || true
|
||||
|
||||
- name: ARMEmitter tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
@@ -241,13 +252,20 @@ jobs:
|
||||
# ASM tests get quite close to 10MB
|
||||
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
|
||||
|
||||
- name: Remove old SHM regions
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: cmake --build . --config $BUILD_TYPE --target remove_old_shm_regions
|
||||
|
||||
- name: Set runner name
|
||||
if: ${{ always() }}
|
||||
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v2'
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
|
||||
@@ -33,7 +33,7 @@ jobs:
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
- uses: actions/checkout@v3
|
||||
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
@@ -157,6 +157,17 @@ jobs:
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_APITests.log || true
|
||||
|
||||
- name: FEXCore APITest tests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
run: cmake --build . --config $BUILD_TYPE --target fexcore_apitests
|
||||
|
||||
- name: FEXCore APITest Test Results move
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: mv ${{runner.workspace}}/build/Testing/Temporary/LastTest.log ${{runner.workspace}}/build/Testing/Temporary/LastTest_FEXCoreAPITests.log || true
|
||||
|
||||
- name: FEXLinuxTests
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
@@ -176,13 +187,20 @@ jobs:
|
||||
# ASM tests get quite close to 10MB
|
||||
run: truncate --size=<20M ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log || true
|
||||
|
||||
- name: Remove old SHM regions
|
||||
if: ${{ always() }}
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
run: cmake --build . --config $BUILD_TYPE --target remove_old_shm_regions
|
||||
|
||||
- name: Set runner name
|
||||
if: ${{ always() }}
|
||||
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v2'
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
|
||||
@@ -0,0 +1,96 @@
|
||||
name: Mingw build
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
pull_request:
|
||||
branches:
|
||||
- main
|
||||
|
||||
env:
|
||||
BUILD_TYPE: Debug
|
||||
FEX_ENABLEAVX: 1
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ${{ matrix.arch }}
|
||||
strategy:
|
||||
matrix:
|
||||
arch: [[self-hosted, x64, mingw], [self-hosted, ARM64, mingw]]
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
|
||||
- name: Set CC x86
|
||||
if: matrix.arch[1] == 'x64'
|
||||
run: |
|
||||
echo "CC=$HOME/llvm-mingw/build/bin/x86_64-w64-mingw32-clang" >> $GITHUB_ENV
|
||||
echo "CXX=$HOME/llvm-mingw/build/bin/x86_64-w64-mingw32-clang++" >> $GITHUB_ENV
|
||||
|
||||
- name: Set CC Arm64
|
||||
if: matrix.arch[1] == 'ARM64'
|
||||
run: |
|
||||
echo "CC=$HOME/llvm-mingw/build/bin/aarch64-w64-mingw32-clang" >> $GITHUB_ENV
|
||||
echo "CXX=$HOME/llvm-mingw/build/bin/aarch64-w64-mingw32-clang++" >> $GITHUB_ENV
|
||||
|
||||
- name: Set rootfs paths
|
||||
run: |
|
||||
echo "FEX_ROOTFS_MOUNT=/mnt/AutoNFS/rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS_PATH=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "FEX_ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
echo "ROOTFS=$HOME/Rootfs/" >> $GITHUB_ENV
|
||||
|
||||
- name: Update RootFS cache
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
run: $GITHUB_WORKSPACE/Scripts/CI_FetchRootFS.py
|
||||
|
||||
- name : submodule checkout
|
||||
# Need to update submodules
|
||||
run: |
|
||||
git submodule sync --recursive
|
||||
git submodule update --init --depth 1
|
||||
|
||||
- name: Clean Build Environment
|
||||
run: rm -Rf ${{runner.workspace}}/build
|
||||
|
||||
- name: Create Build Environment
|
||||
# Some projects don't allow in-source building, so create a separate build directory
|
||||
# We'll use this as our working directory for all subsequent commands
|
||||
run: cmake -E make_directory ${{runner.workspace}}/build
|
||||
|
||||
- name: Configure CMake
|
||||
# Use a bash shell so we can use the same syntax for environment variable
|
||||
# access regardless of the host operating system
|
||||
shell: bash
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
# Note the current convention is to use the -S and -B options here to specify source
|
||||
# and build directories, but this is only available with CMake 3.13 and higher.
|
||||
# The CMake binaries on the Github Actions machines are (as of this writing) 3.12
|
||||
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE -G Ninja -DENABLE_LTO=False -DENABLE_ASSERTIONS=True -DENABLE_X86_HOST_DEBUG=True -DENABLE_INTERPRETER=False -DBUILD_TESTS=False -DENABLE_JEMALLOC=False -DENABLE_JEMALLOC_GLIBC_ALLOC=False -DCMAKE_INSTALL_PREFIX=${{runner.workspace}}/build/install
|
||||
|
||||
- name: Build
|
||||
working-directory: ${{runner.workspace}}/build
|
||||
shell: bash
|
||||
# Execute the build. You can specify a specific target with "--target <NAME>"
|
||||
run: cmake --build . --config $BUILD_TYPE
|
||||
|
||||
- name: Set runner name
|
||||
if: ${{ always() }}
|
||||
run: echo "runner_name=$(hostname)" >> $GITHUB_ENV
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
retention-days: 3
|
||||
|
||||
@@ -25,7 +25,7 @@ jobs:
|
||||
fail-fast: false
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
- uses: actions/checkout@v3
|
||||
|
||||
- name: Set runner label
|
||||
run: echo "runner_label=${{ matrix.arch[1] }}" >> $GITHUB_ENV
|
||||
@@ -111,7 +111,8 @@ jobs:
|
||||
|
||||
- name: Upload results
|
||||
if: ${{ always() }}
|
||||
uses: 'actions/upload-artifact@v2'
|
||||
uses: 'actions/upload-artifact@v3'
|
||||
timeout-minutes: 1
|
||||
with:
|
||||
name: Results-${{ env.runner_name }}
|
||||
path: ${{runner.workspace}}/build/Testing/Temporary/LastTest_*.log
|
||||
|
||||
@@ -0,0 +1,30 @@
|
||||
# Extracts a version from the passed in version string in the form of "<Major>.<Minor>.<Patch>".
|
||||
# If a part of the version is missing then it gets set as zero.
|
||||
# Version variables returned in:
|
||||
# ${Package}_VERSION_MAJOR
|
||||
# ${Package}_VERSION_MINOR
|
||||
# ${Package}_VERSION_PATCH
|
||||
function(version_to_variables VERSION _Package)
|
||||
string(REPLACE "." ";" VERSION_LIST "${VERSION}")
|
||||
list (LENGTH VERSION_LIST VERSION_LEN)
|
||||
if (${VERSION_LEN} GREATER 0)
|
||||
list(GET VERSION_LIST 0 VERSION_MAJOR)
|
||||
set(${_Package}_VERSION_MAJOR ${VERSION_MAJOR} PARENT_SCOPE)
|
||||
else()
|
||||
set(${_Package}_VERSION_MAJOR 0 PARENT_SCOPE)
|
||||
endif()
|
||||
|
||||
if (${VERSION_LEN} GREATER 1)
|
||||
list(GET VERSION_LIST 1 VERSION_MINOR)
|
||||
set(${_Package}_VERSION_MINOR ${VERSION_MINOR} PARENT_SCOPE)
|
||||
else()
|
||||
set(${_Package}_VERSION_MINOR 0 PARENT_SCOPE)
|
||||
endif()
|
||||
|
||||
if (${VERSION_LEN} GREATER 2)
|
||||
list(GET VERSION_LIST 2 VERSION_PATCH)
|
||||
set(${_Package}_VERSION_PATCH ${VERSION_PATCH} PARENT_SCOPE)
|
||||
else()
|
||||
set(${_Package}_VERSION_PATCH 0 PARENT_SCOPE)
|
||||
endif()
|
||||
endfunction()
|
||||
+12
-10
@@ -7,18 +7,17 @@ CHECK_INCLUDE_FILES ("gdb/jit-reader.h" HAVE_GDB_JIT_READER_H)
|
||||
option(BUILD_TESTS "Build unit tests to ensure sanity" TRUE)
|
||||
option(BUILD_FEX_LINUX_TESTS "Build FEXLinuxTests, requires x86 compiler" FALSE)
|
||||
option(BUILD_THUNKS "Build thunks" FALSE)
|
||||
option(BUILD_FEXCONFIG "Build FEXConfig, requires SDL2 and X11" TRUE)
|
||||
option(ENABLE_CLANG_THUNKS "Build thunks with clang" FALSE)
|
||||
option(ENABLE_CLANG_FORMAT "Run clang format over the source" FALSE)
|
||||
option(ENABLE_IWYU "Enables include what you use program" FALSE)
|
||||
option(ENABLE_LTO "Enable LTO with compilation" TRUE)
|
||||
option(ENABLE_XRAY "Enable building with LLVM X-Ray" FALSE)
|
||||
option(ENABLE_LLD "Enable linking with lld" FALSE)
|
||||
option(ENABLE_MOLD "Enable linking with mold" FALSE)
|
||||
set(USE_LINKER "" CACHE STRING "Allow overriding the linker path directly")
|
||||
option(ENABLE_ASAN "Enables Clang ASAN" FALSE)
|
||||
option(ENABLE_TSAN "Enables Clang TSAN" FALSE)
|
||||
option(ENABLE_ASSERTIONS "Enables assertions in build" FALSE)
|
||||
option(ENABLE_GDB_SYMBOLS "Enables GDBSymbols integration support" ${HAVE_GDB_JIT_READER_H})
|
||||
option(ENABLE_VISUAL_DEBUGGER "Enables the visual debugger for compiling" FALSE)
|
||||
option(ENABLE_STRICT_WERROR "Enables stricter -Werror for CI" FALSE)
|
||||
option(ENABLE_WERROR "Enables -Werror" FALSE)
|
||||
option(ENABLE_JEMALLOC "Enables jemalloc allocator" TRUE)
|
||||
@@ -157,13 +156,9 @@ endif()
|
||||
|
||||
set (PTHREAD_LIB pthread)
|
||||
|
||||
if (ENABLE_LLD AND ENABLE_MOLD)
|
||||
message (FATAL_ERROR "Cannot enable both lld and mold")
|
||||
elseif (ENABLE_LLD)
|
||||
set (LD_OVERRIDE "-fuse-ld=lld")
|
||||
add_link_options(${LD_OVERRIDE})
|
||||
elseif (ENABLE_MOLD)
|
||||
add_link_options("-fuse-ld=mold")
|
||||
if (USE_LINKER)
|
||||
message(STATUS "Overriding linker to: ${USE_LINKER}")
|
||||
add_link_options("-fuse-ld=${USE_LINKER}")
|
||||
endif()
|
||||
|
||||
if (ENABLE_LIBCXX)
|
||||
@@ -180,8 +175,13 @@ endif()
|
||||
if(DEFINED ENV{TERMUX_VERSION} OR ENABLE_TERMUX_BUILD)
|
||||
add_definitions(-DTERMUX_BUILD=1)
|
||||
set(TERMUX_BUILD 1)
|
||||
|
||||
# Termux doesn't support Jemalloc due to bad interactions between emutls, jemalloc, and scudo
|
||||
set(ENABLE_JEMALLOC FALSE)
|
||||
|
||||
# Termux builds can't rely on X11 packages
|
||||
# SDL2 isn't even compiled with GL support so our GUIs wouldn't even work
|
||||
set(BUILD_FEXCONFIG FALSE)
|
||||
endif()
|
||||
|
||||
if (ENABLE_ASAN)
|
||||
@@ -262,6 +262,8 @@ if (BUILD_TESTS)
|
||||
include(Catch)
|
||||
endif()
|
||||
|
||||
# Disable fmt install
|
||||
set(FMT_INSTALL OFF)
|
||||
add_subdirectory(External/fmt/)
|
||||
|
||||
add_subdirectory(External/imgui/)
|
||||
|
||||
+66
-63
@@ -6,10 +6,10 @@
|
||||
"X11"
|
||||
],
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGL.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGL.so.1",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGL.so.1.2.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGL.so.1.7.0"
|
||||
"@PREFIX_LIB@/libGL.so",
|
||||
"@PREFIX_LIB@/libGL.so.1",
|
||||
"@PREFIX_LIB@/libGL.so.1.2.0",
|
||||
"@PREFIX_LIB@/libGL.so.1.7.0"
|
||||
]
|
||||
},
|
||||
"GLESv2": {
|
||||
@@ -18,17 +18,17 @@
|
||||
"X11"
|
||||
],
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGLESv2.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGLESv2.so.2",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libGLESv2.so.2.0.0"
|
||||
"@PREFIX_LIB@/libGLESv2.so",
|
||||
"@PREFIX_LIB@/libGLESv2.so.2",
|
||||
"@PREFIX_LIB@/libGLESv2.so.2.0.0"
|
||||
]
|
||||
},
|
||||
"X11": {
|
||||
"Library": "libX11-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libX11.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libX11.so.6",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libX11.so.6.4.0"
|
||||
"@PREFIX_LIB@/libX11.so",
|
||||
"@PREFIX_LIB@/libX11.so.6",
|
||||
"@PREFIX_LIB@/libX11.so.6.4.0"
|
||||
]
|
||||
},
|
||||
"Vulkan": {
|
||||
@@ -37,8 +37,8 @@
|
||||
"xcb"
|
||||
],
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libvulkan.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libvulkan.so.1",
|
||||
"@PREFIX_LIB@/libvulkan.so",
|
||||
"@PREFIX_LIB@/libvulkan.so.1",
|
||||
"@HOME@/.local/share/Steam/ubuntu12_32/steam-runtime/pinned_libs_64/libvulkan.so.1"
|
||||
],
|
||||
"Comment": [
|
||||
@@ -46,139 +46,142 @@
|
||||
]
|
||||
},
|
||||
"xcb": {
|
||||
"Depends": [
|
||||
"X11"
|
||||
],
|
||||
"Library": "libxcb-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb.so.1",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb.so.1.1.0"
|
||||
"@PREFIX_LIB@/libxcb.so",
|
||||
"@PREFIX_LIB@/libxcb.so.1",
|
||||
"@PREFIX_LIB@/libxcb.so.1.1.0"
|
||||
]
|
||||
},
|
||||
"xcb-dri2": {
|
||||
"Library": "libxcb-dri2-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri2.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri2.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri2.so.0.0.0"
|
||||
"@PREFIX_LIB@/libxcb-dri2.so",
|
||||
"@PREFIX_LIB@/libxcb-dri2.so.0",
|
||||
"@PREFIX_LIB@/libxcb-dri2.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-dri3": {
|
||||
"Library": "libxcb-dri3-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri3.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri3.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-dri3.so.0.0.0"
|
||||
"@PREFIX_LIB@/libxcb-dri3.so",
|
||||
"@PREFIX_LIB@/libxcb-dri3.so.0",
|
||||
"@PREFIX_LIB@/libxcb-dri3.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-xfixes": {
|
||||
"Library": "libxcb-xfixes-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-xfixes.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-xfixes.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-xfixes.so.0.0.0"
|
||||
"@PREFIX_LIB@/libxcb-xfixes.so",
|
||||
"@PREFIX_LIB@/libxcb-xfixes.so.0",
|
||||
"@PREFIX_LIB@/libxcb-xfixes.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-shm": {
|
||||
"Library": "libxcb-shm-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-shm.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-shm.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-shm.so.0.0.0"
|
||||
"@PREFIX_LIB@/libxcb-shm.so",
|
||||
"@PREFIX_LIB@/libxcb-shm.so.0",
|
||||
"@PREFIX_LIB@/libxcb-shm.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-sync": {
|
||||
"Library": "libxcb-sync-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-sync.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-sync.so.1",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-sync.so.1.0.0"
|
||||
"@PREFIX_LIB@/libxcb-sync.so",
|
||||
"@PREFIX_LIB@/libxcb-sync.so.1",
|
||||
"@PREFIX_LIB@/libxcb-sync.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-randr": {
|
||||
"Library": "libxcb-randr-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-randr.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-randr.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-randr.so.0.1.0"
|
||||
"@PREFIX_LIB@/libxcb-randr.so",
|
||||
"@PREFIX_LIB@/libxcb-randr.so.0",
|
||||
"@PREFIX_LIB@/libxcb-randr.so.0.1.0"
|
||||
]
|
||||
},
|
||||
"xcb-present": {
|
||||
"Library": "libxcb-present-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-present.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-present.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-present.so.0.0.0"
|
||||
"@PREFIX_LIB@/libxcb-present.so",
|
||||
"@PREFIX_LIB@/libxcb-present.so.0",
|
||||
"@PREFIX_LIB@/libxcb-present.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xcb-glx": {
|
||||
"Library": "libxcb-glx-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-glx.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-glx.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxcb-glx.so.0.0.0"
|
||||
"@PREFIX_LIB@/libxcb-glx.so",
|
||||
"@PREFIX_LIB@/libxcb-glx.so.0",
|
||||
"@PREFIX_LIB@/libxcb-glx.so.0.0.0"
|
||||
]
|
||||
},
|
||||
"xshmfence": {
|
||||
"Library": "libxshmfence-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxshmfence.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxshmfence.so.1",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libxshmfence.so.1.0.0"
|
||||
"@PREFIX_LIB@/libxshmfence.so",
|
||||
"@PREFIX_LIB@/libxshmfence.so.1",
|
||||
"@PREFIX_LIB@/libxshmfence.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"drm": {
|
||||
"Library": "libdrm-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libdrm.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libdrm.so.2",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libdrm.so.2.4.0"
|
||||
"@PREFIX_LIB@/libdrm.so",
|
||||
"@PREFIX_LIB@/libdrm.so.2",
|
||||
"@PREFIX_LIB@/libdrm.so.2.4.0"
|
||||
]
|
||||
},
|
||||
"asound": {
|
||||
"Library": "libasound-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libasound.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libasound.so.2",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libasound.so.2.0.0"
|
||||
"@PREFIX_LIB@/libasound.so",
|
||||
"@PREFIX_LIB@/libasound.so.2",
|
||||
"@PREFIX_LIB@/libasound.so.2.0.0"
|
||||
]
|
||||
},
|
||||
"Xrender": {
|
||||
"Library": "libXrender-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXrender.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXrender.so.1",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXrender.so.1.3.0"
|
||||
"@PREFIX_LIB@/libXrender.so",
|
||||
"@PREFIX_LIB@/libXrender.so.1",
|
||||
"@PREFIX_LIB@/libXrender.so.1.3.0"
|
||||
]
|
||||
},
|
||||
"Xext": {
|
||||
"Library": "libXext-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXext.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXext.so.6",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXext.so.6.4.0"
|
||||
"@PREFIX_LIB@/libXext.so",
|
||||
"@PREFIX_LIB@/libXext.so.6",
|
||||
"@PREFIX_LIB@/libXext.so.6.4.0"
|
||||
]
|
||||
},
|
||||
"Xfixes": {
|
||||
"Library": "libXfixes-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXfixes.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXfixes.so.3",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libXfixes.so.3.1.0"
|
||||
"@PREFIX_LIB@/libXfixes.so",
|
||||
"@PREFIX_LIB@/libXfixes.so.3",
|
||||
"@PREFIX_LIB@/libXfixes.so.3.1.0"
|
||||
]
|
||||
},
|
||||
"OpenCL": {
|
||||
"Library" : "libOpenCL-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libOpenCL.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libOpenCL.so.1",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libOpenCL.so.1.0.0"
|
||||
"@PREFIX_LIB@/libOpenCL.so",
|
||||
"@PREFIX_LIB@/libOpenCL.so.1",
|
||||
"@PREFIX_LIB@/libOpenCL.so.1.0.0"
|
||||
]
|
||||
},
|
||||
"WaylandClient": {
|
||||
"Library" : "libwayland-client-guest.so",
|
||||
"Overlay": [
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libwayland-client.so",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libwayland-client.so.0",
|
||||
"@PREFIX_LIB@/@PREFIX_ARCH@-linux-gnu/libwayland-client.so.0.20.0"
|
||||
"@PREFIX_LIB@/libwayland-client.so",
|
||||
"@PREFIX_LIB@/libwayland-client.so.0",
|
||||
"@PREFIX_LIB@/libwayland-client.so.0.20.0"
|
||||
]
|
||||
},
|
||||
"":{}
|
||||
|
||||
+70
-5
@@ -27,7 +27,9 @@ def print_header():
|
||||
#ifndef OPT_STRARRAY
|
||||
#define OPT_STRARRAY(group, enum, json, default) OPT_BASE(fextl::string, group, enum, json, default)
|
||||
#endif
|
||||
|
||||
#ifndef OPT_STRENUM
|
||||
#define OPT_STRENUM(group, enum, json, default) OPT_BASE(uint64_t, group, enum, json, default)
|
||||
#endif
|
||||
'''
|
||||
output_file.write(header)
|
||||
|
||||
@@ -40,6 +42,7 @@ def print_tail():
|
||||
#undef OPT_UINT64
|
||||
#undef OPT_STR
|
||||
#undef OPT_STRARRAY
|
||||
#undef OPT_STRENUM
|
||||
'''
|
||||
output_file.write(tail)
|
||||
|
||||
@@ -127,12 +130,13 @@ def print_man_options(options):
|
||||
short = op_vals["ShortArg"]
|
||||
|
||||
default = op_vals["Default"]
|
||||
value_type = op_vals["Type"]
|
||||
|
||||
# Textual default rather than enum based
|
||||
if ("TextDefault" in op_vals):
|
||||
default = op_vals["TextDefault"]
|
||||
|
||||
if (op_vals["Type"] == "str" or op_vals["Type"] == "strarray"):
|
||||
if (value_type == "str" or value_type == "strarray" or value_type == "strenum"):
|
||||
# Wrap the string argument in quotes
|
||||
default = "'" + default + "'"
|
||||
print_man_option(
|
||||
@@ -141,6 +145,12 @@ def print_man_options(options):
|
||||
op_vals["Desc"],
|
||||
default
|
||||
)
|
||||
if (value_type == "strenum"):
|
||||
Enums = op_vals["Enums"]
|
||||
output_man.write("\\fBAvailable Options:\\fR\n")
|
||||
for enum_op_key, enum_op_vals in Enums.items():
|
||||
output_man.write("{}, ".format(enum_op_vals))
|
||||
output_man.write("\n")
|
||||
|
||||
output_man.write(".El\n")
|
||||
|
||||
@@ -150,12 +160,13 @@ def print_man_environment(options):
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
default = op_vals["Default"]
|
||||
value_type = op_vals["Type"]
|
||||
|
||||
# Textual default rather than enum based
|
||||
if ("TextDefault" in op_vals):
|
||||
default = op_vals["TextDefault"]
|
||||
|
||||
if (op_vals["Type"] == "str" or op_vals["Type"] == "strarray"):
|
||||
if (value_type == "str" or value_type == "strarray" or value_type == "strenum"):
|
||||
# Wrap the string argument in quotes
|
||||
default = "'" + default + "'"
|
||||
print_man_env_option(
|
||||
@@ -165,6 +176,13 @@ def print_man_environment(options):
|
||||
False
|
||||
)
|
||||
|
||||
if (value_type == "strenum"):
|
||||
Enums = op_vals["Enums"]
|
||||
output_man.write("\\fBAvailable Options:\\fR\n")
|
||||
for enum_op_key, enum_op_vals in Enums.items():
|
||||
output_man.write("{}, ".format(enum_op_vals))
|
||||
output_man.write("\n")
|
||||
|
||||
print_man_environment_tail()
|
||||
output_man.write(".El\n")
|
||||
|
||||
@@ -334,7 +352,7 @@ def print_argloader_options(options):
|
||||
for op_key, op_vals in group_vals.items():
|
||||
default = op_vals["Default"]
|
||||
|
||||
if (op_vals["Type"] == "str" or op_vals["Type"] == "strarray"):
|
||||
if (op_vals["Type"] == "str" or op_vals["Type"] == "strarray" or op_vals["Type"] == "strenum"):
|
||||
# Wrap the string argument in quotes
|
||||
default = "\"" + default + "\""
|
||||
|
||||
@@ -382,7 +400,10 @@ def print_parse_argloader_options(options):
|
||||
# boolean values need a decimal specifier. Otherwise fmt prints strings.
|
||||
conversion_func = "fextl::fmt::format(\"{:d}\", "
|
||||
|
||||
if (value_type == "strarray"):
|
||||
if (value_type == "strenum"):
|
||||
output_argloader.write("\tfextl::string UserValue = Options[\"{0}\"];\n".format(op_key))
|
||||
output_argloader.write("\tSet(FEXCore::Config::ConfigOption::CONFIG_{}, FEXCore::Config::EnumParser(FEXCore::Config::{}_EnumPairs, UserValue));\n".format(op_key.upper(), op_key, op_key))
|
||||
elif (value_type == "strarray"):
|
||||
# these need a bit more help
|
||||
output_argloader.write("\tauto Array = Options.all(\"{0}\");\n".format(op_key))
|
||||
output_argloader.write("\tfor (auto iter = Array.begin(); iter != Array.end(); ++iter) {\n")
|
||||
@@ -407,6 +428,12 @@ def print_parse_envloader_options(options):
|
||||
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
value_type = op_vals["Type"]
|
||||
if (value_type == "strenum"):
|
||||
output_argloader.write("else if (Key == \"FEX_{0}\") {{\n".format(op_key.upper()))
|
||||
output_argloader.write("Value = FEXCore::Config::EnumParser(FEXCore::Config::{}_EnumPairs, Value);\n".format(op_key, op_key))
|
||||
output_argloader.write("}\n")
|
||||
|
||||
if ("ArgumentHandler" in op_vals):
|
||||
conversion_func = "FEXCore::Config::Handler::{0}".format(op_vals["ArgumentHandler"])
|
||||
output_argloader.write("else if (Key == \"FEX_{0}\") {{\n".format(op_key.upper()))
|
||||
@@ -414,6 +441,41 @@ def print_parse_envloader_options(options):
|
||||
output_argloader.write("}\n")
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
def print_parse_enum_options(options):
|
||||
output_argloader.write("#ifdef ENUMDEFINES\n")
|
||||
output_argloader.write("#undef ENUMDEFINES\n")
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
if (op_vals["Type"] == "strenum"):
|
||||
output_argloader.write("enum {} : uint64_t {{\n".format(op_key))
|
||||
Enums = op_vals["Enums"]
|
||||
i = 0
|
||||
# Always have an OFF.
|
||||
output_argloader.write("\tOFF = 0,\n")
|
||||
for enum_op_key, enum_op_vals in Enums.items():
|
||||
output_argloader.write("\t{} = 1ULL << {},\n".format(enum_op_key.upper(), i))
|
||||
i += 1
|
||||
|
||||
output_argloader.write("};\n")
|
||||
|
||||
for op_group, group_vals in options.items():
|
||||
for op_key, op_vals in group_vals.items():
|
||||
if (op_vals["Type"] == "strenum"):
|
||||
Enums = op_vals["Enums"]
|
||||
|
||||
output_argloader.write("using {}ConfigPair = std::pair<std::string_view, FEXCore::Config::{}>;\n".format(op_key, op_key))
|
||||
output_argloader.write("constexpr static std::array<{}ConfigPair, {}> {}_EnumPairs = {{{{\n".format(op_key, len(Enums) + 1, op_key))
|
||||
i = 0
|
||||
# Always have an OFF.
|
||||
output_argloader.write("\t{{ \"off\", FEXCore::Config::{}::OFF }},\n".format(op_key))
|
||||
for enum_op_key, enum_op_vals in Enums.items():
|
||||
output_argloader.write("\t{{ \"{}\", FEXCore::Config::{}::{} }},\n".format(enum_op_vals, op_key, enum_op_key.upper()))
|
||||
i += 1
|
||||
|
||||
output_argloader.write("}};\n")
|
||||
|
||||
output_argloader.write("#endif\n")
|
||||
|
||||
def check_for_duplicate_options(options):
|
||||
short_map = []
|
||||
long_map = []
|
||||
@@ -492,4 +554,7 @@ print_parse_argloader_options(options);
|
||||
# Generate environment loader code
|
||||
print_parse_envloader_options(options);
|
||||
|
||||
# Generate enum variable options
|
||||
print_parse_enum_options(options);
|
||||
|
||||
output_argloader.close()
|
||||
+3
-5
@@ -251,11 +251,8 @@ def parse_ops(ops):
|
||||
|
||||
# Print out enum values
|
||||
def print_enums():
|
||||
if len(IROps) > 255:
|
||||
ExitError("We have more than uint8_t ops. We have {}. Time to upgrade to uint16_t".format(len(IROps)))
|
||||
|
||||
output_file.write("#ifdef IROP_ENUM\n")
|
||||
output_file.write("enum IROps : uint8_t {\n")
|
||||
output_file.write("enum IROps : uint16_t {\n")
|
||||
|
||||
for op in IROps:
|
||||
output_file.write("\tOP_{},\n" .format(op.Name.upper()))
|
||||
@@ -291,6 +288,7 @@ def print_ir_structs(defines):
|
||||
output_file.write("\tOrderedNodeWrapper Args[0];\n")
|
||||
|
||||
output_file.write("};\n\n");
|
||||
output_file.write("static_assert(sizeof(IROp_Header) == sizeof(uint32_t), \"IROp_Header should be 32-bits in size\");\n\n");
|
||||
|
||||
# Now the user defined types
|
||||
output_file.write("// User defined IR Op structs\n")
|
||||
@@ -677,7 +675,7 @@ def print_ir_allocator_helpers():
|
||||
output_file.write("\t\t#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED\n")
|
||||
|
||||
for Validation in op.EmitValidation:
|
||||
output_file.write("\t\tassert({});\n".format(Validation))
|
||||
output_file.write("\tLOGMAN_THROW_A_FMT({}, \"\");\n".format(Validation))
|
||||
output_file.write("\t\t#endif\n")
|
||||
|
||||
output_file.write("\t\treturn Op;\n")
|
||||
|
||||
+2
-2
@@ -1,4 +1,4 @@
|
||||
set (MAN_DIR ${CMAKE_INSTALL_PREFIX}/share/man CACHE PATH "MAN_DIR")
|
||||
set (MAN_DIR share/man CACHE PATH "MAN_DIR")
|
||||
|
||||
set (FEXCORE_BASE_SRCS
|
||||
Interface/Config/Config.cpp
|
||||
@@ -231,7 +231,7 @@ if (ENABLE_JIT_ARM64)
|
||||
)
|
||||
endif()
|
||||
|
||||
set (LIBS fmt::fmt vixl xxhash tiny-json FEXHeaderUtils)
|
||||
set (LIBS fmt::fmt vixl xxhash FEXHeaderUtils)
|
||||
|
||||
if (NOT MINGW_BUILD)
|
||||
list (APPEND LIBS dl)
|
||||
|
||||
+5
-1
@@ -17,8 +17,12 @@ namespace FEXCore {
|
||||
|
||||
void JITSymbols::InitFile() {
|
||||
// We can't use FILE here since we must be robust against forking processes closing our FD from under us.
|
||||
#ifdef __ANDROID__
|
||||
// Android simpleperf looks in /data/local/tmp instead of /tmp
|
||||
const auto PerfMap = fextl::fmt::format("/data/local/tmp/perf-{}.map", getpid());
|
||||
#else
|
||||
const auto PerfMap = fextl::fmt::format("/tmp/perf-{}.map", getpid());
|
||||
|
||||
#endif
|
||||
fd = open(PerfMap.c_str(), O_CREAT | O_TRUNC | O_WRONLY | O_APPEND, 0644);
|
||||
}
|
||||
|
||||
|
||||
+3
-218
@@ -1,5 +1,6 @@
|
||||
#include "Common/StringConv.h"
|
||||
#include "Common/StringUtils.h"
|
||||
#include "FEXCore/Utils/EnumUtils.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
@@ -16,7 +17,6 @@
|
||||
#include <FEXHeaderUtils/Filesystem.h>
|
||||
|
||||
#include <array>
|
||||
#include <assert.h>
|
||||
#include <cstdlib>
|
||||
#include <functional>
|
||||
#include <optional>
|
||||
@@ -27,8 +27,6 @@
|
||||
#include <type_traits>
|
||||
#include <utility>
|
||||
|
||||
#include <tiny-json.h>
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class Context;
|
||||
}
|
||||
@@ -39,74 +37,10 @@ namespace DefaultValues {
|
||||
#define OPT_BASE(type, group, enum, json, default) const P(type) P(enum) = P(default);
|
||||
#define OPT_STR(group, enum, json, default) const std::string_view P(enum) = P(default);
|
||||
#define OPT_STRARRAY(group, enum, json, default) OPT_STR(group, enum, json, default)
|
||||
#define OPT_STRENUM(group, enum, json, default) const uint64_t P(enum) = FEXCore::ToUnderlying(P(default));
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}
|
||||
|
||||
namespace JSON {
|
||||
struct JsonAllocator {
|
||||
jsonPool_t PoolObject;
|
||||
fextl::unique_ptr<fextl::list<json_t>> json_objects;
|
||||
};
|
||||
static_assert(offsetof(JsonAllocator, PoolObject) == 0, "This needs to be at offset zero");
|
||||
|
||||
json_t* PoolInit(jsonPool_t* Pool) {
|
||||
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
|
||||
alloc->json_objects = fextl::make_unique<fextl::list<json_t>>();
|
||||
return &*alloc->json_objects->emplace(alloc->json_objects->end());
|
||||
}
|
||||
|
||||
json_t* PoolAlloc(jsonPool_t* Pool) {
|
||||
JsonAllocator* alloc = reinterpret_cast<JsonAllocator*>(Pool);
|
||||
return &*alloc->json_objects->emplace(alloc->json_objects->end());
|
||||
}
|
||||
|
||||
static void LoadJSonConfig(const fextl::string &Config, std::function<void(const char *Name, const char *ConfigSring)> Func) {
|
||||
fextl::vector<char> Data;
|
||||
if (!FEXCore::FileLoading::LoadFile(Data, Config)) {
|
||||
return;
|
||||
}
|
||||
|
||||
JsonAllocator Pool {
|
||||
.PoolObject = {
|
||||
.init = PoolInit,
|
||||
.alloc = PoolAlloc,
|
||||
},
|
||||
};
|
||||
|
||||
json_t const *json = json_createWithPool(&Data.at(0), &Pool.PoolObject);
|
||||
if (!json) {
|
||||
LogMan::Msg::EFmt("Couldn't create json");
|
||||
return;
|
||||
}
|
||||
|
||||
json_t const* ConfigList = json_getProperty(json, "Config");
|
||||
|
||||
if (!ConfigList) {
|
||||
// This is a non-error if the configuration file exists but no Config section
|
||||
return;
|
||||
}
|
||||
|
||||
for (json_t const* ConfigItem = json_getChild(ConfigList);
|
||||
ConfigItem != nullptr;
|
||||
ConfigItem = json_getSibling(ConfigItem)) {
|
||||
const char* ConfigName = json_getName(ConfigItem);
|
||||
const char* ConfigString = json_getValue(ConfigItem);
|
||||
|
||||
if (!ConfigName) {
|
||||
LogMan::Msg::EFmt("Couldn't get config name");
|
||||
return;
|
||||
}
|
||||
|
||||
if (!ConfigString) {
|
||||
LogMan::Msg::EFmt("Couldn't get ConfigString for '{}'", ConfigName);
|
||||
return;
|
||||
}
|
||||
|
||||
Func(ConfigName, ConfigString);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
enum Paths {
|
||||
PATH_DATA_DIR = 0,
|
||||
PATH_CONFIG_DIR_LOCAL,
|
||||
@@ -518,7 +452,7 @@ namespace JSON {
|
||||
auto Value = FEXCore::Config::Get(Option);
|
||||
|
||||
if (!FEXCore::StrConv::Conv(**Value, &Result)) {
|
||||
assert(0 && "Attempted to convert invalid value");
|
||||
LOGMAN_MSG_A_FMT("Attempted to convert invalid value");
|
||||
}
|
||||
return Result;
|
||||
}
|
||||
@@ -583,154 +517,5 @@ namespace JSON {
|
||||
}
|
||||
}
|
||||
template void Value<fextl::string>::GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string> *List);
|
||||
|
||||
// Application loaders
|
||||
class MainLoader final : public FEXCore::Config::OptionMapper {
|
||||
public:
|
||||
explicit MainLoader(FEXCore::Config::LayerType Type);
|
||||
explicit MainLoader(fextl::string ConfigFile);
|
||||
void Load() override;
|
||||
|
||||
private:
|
||||
fextl::string Config;
|
||||
};
|
||||
|
||||
class AppLoader final : public FEXCore::Config::OptionMapper {
|
||||
public:
|
||||
explicit AppLoader(const fextl::string& Filename, FEXCore::Config::LayerType Type);
|
||||
void Load();
|
||||
|
||||
private:
|
||||
fextl::string Config;
|
||||
};
|
||||
|
||||
class EnvLoader final : public FEXCore::Config::Layer {
|
||||
public:
|
||||
explicit EnvLoader(char *const _envp[]);
|
||||
void Load() override;
|
||||
|
||||
private:
|
||||
char *const *envp;
|
||||
};
|
||||
|
||||
static const fextl::map<fextl::string, FEXCore::Config::ConfigOption, std::less<>> ConfigLookup = {{
|
||||
#define OPT_BASE(type, group, enum, json, default) {#json, FEXCore::Config::ConfigOption::CONFIG_##enum},
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}};
|
||||
static const fextl::vector<std::pair<const char*, FEXCore::Config::ConfigOption>> EnvConfigLookup = {{
|
||||
#define OPT_BASE(type, group, enum, json, default) {"FEX_" #enum, FEXCore::Config::ConfigOption::CONFIG_##enum},
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
}};
|
||||
|
||||
OptionMapper::OptionMapper(FEXCore::Config::LayerType Layer)
|
||||
: FEXCore::Config::Layer(Layer) {
|
||||
}
|
||||
|
||||
void OptionMapper::MapNameToOption(const char *ConfigName, const char *ConfigString) {
|
||||
auto it = ConfigLookup.find(ConfigName);
|
||||
if (it != ConfigLookup.end()) {
|
||||
Set(it->second, ConfigString);
|
||||
}
|
||||
}
|
||||
|
||||
MainLoader::MainLoader(FEXCore::Config::LayerType Type)
|
||||
: FEXCore::Config::OptionMapper(Type)
|
||||
, Config{FEXCore::Config::GetConfigFileLocation(Type == FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN)} {
|
||||
}
|
||||
|
||||
MainLoader::MainLoader(fextl::string ConfigFile)
|
||||
: FEXCore::Config::OptionMapper(FEXCore::Config::LayerType::LAYER_MAIN)
|
||||
, Config{std::move(ConfigFile)} {
|
||||
}
|
||||
|
||||
void MainLoader::Load() {
|
||||
JSON::LoadJSonConfig(Config, [this](const char *Name, const char *ConfigString) {
|
||||
MapNameToOption(Name, ConfigString);
|
||||
});
|
||||
}
|
||||
|
||||
AppLoader::AppLoader(const fextl::string& Filename, FEXCore::Config::LayerType Type)
|
||||
: FEXCore::Config::OptionMapper(Type) {
|
||||
const bool Global = Type == FEXCore::Config::LayerType::LAYER_GLOBAL_STEAM_APP ||
|
||||
Type == FEXCore::Config::LayerType::LAYER_GLOBAL_APP;
|
||||
Config = FEXCore::Config::GetApplicationConfig(Filename, Global);
|
||||
|
||||
// Immediately load so we can reload the meta layer
|
||||
Load();
|
||||
}
|
||||
|
||||
void AppLoader::Load() {
|
||||
JSON::LoadJSonConfig(Config, [this](const char *Name, const char *ConfigString) {
|
||||
MapNameToOption(Name, ConfigString);
|
||||
});
|
||||
}
|
||||
|
||||
EnvLoader::EnvLoader(char *const _envp[])
|
||||
: FEXCore::Config::Layer(FEXCore::Config::LayerType::LAYER_ENVIRONMENT)
|
||||
, envp {_envp} {
|
||||
}
|
||||
|
||||
void EnvLoader::Load() {
|
||||
using EnvMapType = fextl::unordered_map<std::string_view, std::string_view>;
|
||||
EnvMapType EnvMap;
|
||||
|
||||
for(const char *const *pvar=envp; pvar && *pvar; pvar++) {
|
||||
std::string_view Var(*pvar);
|
||||
size_t pos = Var.rfind('=');
|
||||
if (fextl::string::npos == pos)
|
||||
continue;
|
||||
|
||||
std::string_view Key = Var.substr(0,pos);
|
||||
std::string_view Value {Var.substr(pos+1)};
|
||||
|
||||
#define ENVLOADER
|
||||
#include <FEXCore/Config/ConfigOptions.inl>
|
||||
|
||||
EnvMap[Key] = Value;
|
||||
}
|
||||
|
||||
auto GetVar = [](EnvMapType &EnvMap, const std::string_view id) -> std::optional<std::string_view> {
|
||||
if (EnvMap.find(id) != EnvMap.end())
|
||||
return EnvMap.at(id);
|
||||
|
||||
// If envp[] was empty, search using std::getenv()
|
||||
const char* vs = std::getenv(id.data());
|
||||
if (vs) {
|
||||
return vs;
|
||||
}
|
||||
else {
|
||||
return std::nullopt;
|
||||
}
|
||||
};
|
||||
|
||||
std::optional<std::string_view> Value;
|
||||
|
||||
for (auto &it : EnvConfigLookup) {
|
||||
if ((Value = GetVar(EnvMap, it.first)).has_value()) {
|
||||
Set(it.second, fextl::string(*Value));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateGlobalMainLayer() {
|
||||
return fextl::make_unique<FEXCore::Config::MainLoader>(FEXCore::Config::LayerType::LAYER_GLOBAL_MAIN);
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(fextl::string const *File) {
|
||||
if (File) {
|
||||
return fextl::make_unique<FEXCore::Config::MainLoader>(*File);
|
||||
}
|
||||
else {
|
||||
return fextl::make_unique<FEXCore::Config::MainLoader>(FEXCore::Config::LayerType::LAYER_MAIN);
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const fextl::string& Filename, FEXCore::Config::LayerType Type) {
|
||||
return fextl::make_unique<FEXCore::Config::AppLoader>(Filename, Type);
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::Config::Layer> CreateEnvironmentLayer(char *const _envp[]) {
|
||||
return fextl::make_unique<FEXCore::Config::EnvLoader>(_envp);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -251,6 +251,20 @@
|
||||
"Set this in an application configuration for injecting in to only specific applications.",
|
||||
"\tNote: If x86/x86_64 libSegFault.so isn't installed then this option won't work."
|
||||
]
|
||||
},
|
||||
"Disassemble": {
|
||||
"Type": "strenum",
|
||||
"Default": "FEXCore::Config::Disassemble::OFF",
|
||||
"Enums": {
|
||||
"DISPATCHER": "dispatcher",
|
||||
"BLOCKS": "blocks"
|
||||
},
|
||||
"Desc": [
|
||||
"Allows controlling of the vixl disassembler.",
|
||||
"\toff: No disassembly will be output",
|
||||
"\tdispatcher: Will enable disassembly of the JIT dispatcher loop",
|
||||
"\tblocks: Will enable disassembly of the translated instruction code blocks"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Logging": {
|
||||
|
||||
+7
-37
@@ -1,5 +1,4 @@
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/Core.h"
|
||||
#include "Interface/Core/OpcodeDispatcher.h"
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
@@ -27,7 +26,9 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
bool FEXCore::Context::ContextImpl::InitializeContext() {
|
||||
return FEXCore::CPU::CreateCPUCore(this);
|
||||
// This should be used for generating things that are shared between threads
|
||||
CPUID.Init(this);
|
||||
return true;
|
||||
}
|
||||
|
||||
void FEXCore::Context::ContextImpl::SetExitHandler(ExitHandler handler) {
|
||||
@@ -66,10 +67,6 @@ namespace FEXCore::Context {
|
||||
CustomCPUFactory = std::move(Factory);
|
||||
}
|
||||
|
||||
bool FEXCore::Context::ContextImpl::AddVirtualMemoryMapping([[maybe_unused]] uint64_t VirtualAddress, [[maybe_unused]] uint64_t PhysicalAddress, [[maybe_unused]] uint64_t Size) {
|
||||
return false;
|
||||
}
|
||||
|
||||
HostFeatures FEXCore::Context::ContextImpl::GetHostFeatures() const {
|
||||
return HostFeatures;
|
||||
}
|
||||
@@ -87,38 +84,11 @@ namespace FEXCore::Context {
|
||||
return CPUID.RunFunction(Function, Leaf);
|
||||
}
|
||||
|
||||
FEXCore::CPUID::XCRResults FEXCore::Context::ContextImpl::RunXCRFunction(uint32_t Function) {
|
||||
return CPUID.RunXCRFunction(Function);
|
||||
}
|
||||
|
||||
FEXCore::CPUID::FunctionResults FEXCore::Context::ContextImpl::RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) {
|
||||
return CPUID.RunFunctionName(Function, Leaf, CPU);
|
||||
}
|
||||
|
||||
namespace Debug {
|
||||
//void CompileRIP(FEXCore::Context::Context *CTX, uint64_t RIP) {
|
||||
// CTX->CompileRIP(CTX->ParentThread, RIP);
|
||||
//}
|
||||
//uint64_t GetThreadCount(FEXCore::Context::Context *CTX) {
|
||||
// return CTX->GetThreadCount();
|
||||
//}
|
||||
|
||||
//FEXCore::Core::RuntimeStats *GetRuntimeStatsForThread(FEXCore::Context::Context *CTX, uint64_t Thread) {
|
||||
// return CTX->GetRuntimeStatsForThread(Thread);
|
||||
//}
|
||||
|
||||
//bool GetDebugDataForRIP(FEXCore::Context::Context *CTX, uint64_t RIP, FEXCore::Core::DebugData *Data) {
|
||||
// return CTX->GetDebugDataForRIP(RIP, Data);
|
||||
//}
|
||||
|
||||
//bool FindHostCodeForRIP(FEXCore::Context::Context *CTX, uint64_t RIP, uint8_t **Code) {
|
||||
// return CTX->FindHostCodeForRIP(RIP, Code);
|
||||
//}
|
||||
|
||||
// XXX:
|
||||
// bool FindIRForRIP(FEXCore::Context::Context *CTX, uint64_t RIP, FEXCore::IR::IntrusiveIRList **ir) {
|
||||
// return CTX->FindIRForRIP(RIP, ir);
|
||||
// }
|
||||
|
||||
// void SetIRForRIP(FEXCore::Context::Context *CTX, uint64_t RIP, FEXCore::IR::IntrusiveIRList *const ir) {
|
||||
// CTX->SetIRForRIP(RIP, ir);
|
||||
// }
|
||||
}
|
||||
|
||||
}
|
||||
+49
-27
@@ -1,7 +1,6 @@
|
||||
#pragma once
|
||||
|
||||
#include "Common/JitSymbols.h"
|
||||
#include "FEXHeaderUtils/ScopedSignalMask.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/X86HelperGen.h"
|
||||
#include "Interface/Core/ObjectCache/ObjectCacheService.h"
|
||||
@@ -14,6 +13,7 @@
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/DeferredSignalMutex.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/set.h>
|
||||
@@ -100,8 +100,6 @@ namespace FEXCore::Context {
|
||||
|
||||
void SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) override;
|
||||
|
||||
bool AddVirtualMemoryMapping(uint64_t VirtualAddress, uint64_t PhysicalAddress, uint64_t Size) override;
|
||||
|
||||
HostFeatures GetHostFeatures() const override;
|
||||
|
||||
void HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) override;
|
||||
@@ -154,39 +152,44 @@ namespace FEXCore::Context {
|
||||
* @param Thread The internal FEX thread state object
|
||||
*/
|
||||
void DestroyThread(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
void CleanupAfterFork(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
|
||||
#ifndef _WIN32
|
||||
void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) override;
|
||||
void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child) override;
|
||||
#endif
|
||||
void SetSignalDelegator(FEXCore::SignalDelegator *SignalDelegation) override;
|
||||
void SetSyscallHandler(FEXCore::HLE::SyscallHandler *Handler) override;
|
||||
|
||||
FEXCore::CPUID::FunctionResults RunCPUIDFunction(uint32_t Function, uint32_t Leaf) override;
|
||||
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) override;
|
||||
FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) override;
|
||||
|
||||
FEXCore::IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(const fextl::string& Name) override;
|
||||
void UnloadAOTIRCacheEntry(FEXCore::IR::AOTIRCacheEntry *Entry) override;
|
||||
|
||||
void SetAOTIRLoader(std::function<int(const fextl::string&)> CacheReader) override {
|
||||
IRCaptureCache.SetAOTIRLoader(CacheReader);
|
||||
void SetAOTIRLoader(AOTIRLoaderCBFn CacheReader) override {
|
||||
IRCaptureCache.SetAOTIRLoader(std::move(CacheReader));
|
||||
}
|
||||
void SetAOTIRWriter(std::function<fextl::unique_ptr<AOTIRWriter>(const fextl::string&)> CacheWriter) override {
|
||||
IRCaptureCache.SetAOTIRWriter(CacheWriter);
|
||||
void SetAOTIRWriter(AOTIRWriterCBFn CacheWriter) override {
|
||||
IRCaptureCache.SetAOTIRWriter(std::move(CacheWriter));
|
||||
}
|
||||
void SetAOTIRRenamer(std::function<void(const fextl::string&)> CacheRenamer) override {
|
||||
IRCaptureCache.SetAOTIRRenamer(CacheRenamer);
|
||||
void SetAOTIRRenamer(AOTIRRenamerCBFn CacheRenamer) override {
|
||||
IRCaptureCache.SetAOTIRRenamer(std::move(CacheRenamer));
|
||||
}
|
||||
|
||||
void FinalizeAOTIRCache() override {
|
||||
IRCaptureCache.FinalizeAOTIRCache();
|
||||
}
|
||||
void WriteFilesWithCode(std::function<void(const fextl::string& fileid, const fextl::string& filename)> Writer) override {
|
||||
void WriteFilesWithCode(AOTIRCodeFileWriterFn Writer) override {
|
||||
IRCaptureCache.WriteFilesWithCode(Writer);
|
||||
}
|
||||
void InvalidateGuestCodeRange(uint64_t Start, uint64_t Length) override;
|
||||
void InvalidateGuestCodeRange(uint64_t Start, uint64_t Length, std::function<void(uint64_t start, uint64_t Length)> callback) override;
|
||||
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) override;
|
||||
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, CodeRangeInvalidationFn callback) override;
|
||||
void MarkMemoryShared() override;
|
||||
|
||||
void ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, fextl::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) override;
|
||||
// returns false if a handler was already registered
|
||||
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)> Handler, void *Creator = nullptr, void *Data = nullptr) override;
|
||||
CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void *Creator = nullptr, void *Data = nullptr) override;
|
||||
|
||||
void AppendThunkDefinitions(fextl::vector<FEXCore::IR::ThunkDefinition> const& Definitions) override;
|
||||
|
||||
@@ -253,7 +256,7 @@ namespace FEXCore::Context {
|
||||
Event PauseWait;
|
||||
bool Running{};
|
||||
|
||||
std::shared_mutex CodeInvalidationMutex;
|
||||
FEXCore::ForkableSharedMutex CodeInvalidationMutex;
|
||||
|
||||
FEXCore::CPUIDEmu CPUID;
|
||||
FEXCore::HLE::SyscallHandler *SyscallHandler{};
|
||||
@@ -289,7 +292,8 @@ namespace FEXCore::Context {
|
||||
|
||||
template<auto Fn>
|
||||
static uint64_t ThreadExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
|
||||
FHU::ScopedSignalMaskWithSharedLock lk(static_cast<ContextImpl*>(Frame->Thread->CTX)->CodeInvalidationMutex);
|
||||
auto Thread = Frame->Thread;
|
||||
ScopedDeferredSignalWithForkableSharedLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
return Fn(Frame, record);
|
||||
}
|
||||
@@ -300,20 +304,13 @@ namespace FEXCore::Context {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
LogMan::Throw::AFmt(Thread->ThreadManager.GetTID() == FHU::Syscalls::gettid(), "Must be called from owning thread {}, not {}", Thread->ThreadManager.GetTID(), FHU::Syscalls::gettid());
|
||||
|
||||
FHU::ScopedSignalMaskWithUniqueLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex);
|
||||
ScopedDeferredSignalWithForkableUniqueLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
ThreadRemoveCodeEntry(Thread, GuestRIP);
|
||||
}
|
||||
|
||||
void RemoveCustomIREntrypoint(uintptr_t Entrypoint);
|
||||
|
||||
// Debugger interface
|
||||
uint64_t GetThreadCount() const;
|
||||
FEXCore::Core::RuntimeStats *GetRuntimeStatsForThread(uint64_t Thread);
|
||||
bool GetDebugDataForRIP(uint64_t RIP, FEXCore::Core::DebugData *Data);
|
||||
bool FindHostCodeForRIP(uint64_t RIP, uint8_t **Code);
|
||||
|
||||
struct GenerateIRResult {
|
||||
FEXCore::IR::IRListView* IRList;
|
||||
FEXCore::IR::RegisterAllocationData::UniquePtr RAData;
|
||||
@@ -372,11 +369,34 @@ namespace FEXCore::Context {
|
||||
FEXCore::Utils::PooledAllocatorVirtual OpDispatcherAllocator;
|
||||
FEXCore::Utils::PooledAllocatorVirtual FrontendAllocator;
|
||||
|
||||
bool IsTSOEnabled() { return (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled; }
|
||||
// If Atomic-based TSO emulation is enabled or not.
|
||||
bool IsAtomicTSOEnabled() const { return AtomicTSOEmulationEnabled; }
|
||||
|
||||
void SetHardwareTSOSupport(bool HardwareTSOSupported) override {
|
||||
SupportsHardwareTSO = HardwareTSOSupported;
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
}
|
||||
|
||||
void EnableExitOnHLT() override { ExitOnHLT = true; }
|
||||
|
||||
bool ExitOnHLTEnabled() const { return ExitOnHLT; }
|
||||
|
||||
FEXCore::CPU::CPUBackendFeatures BackendFeatures;
|
||||
|
||||
protected:
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState *Thread);
|
||||
|
||||
void UpdateAtomicTSOEmulationConfig() {
|
||||
if (SupportsHardwareTSO) {
|
||||
// If the hardware supports TSO then we don't need to emulate it through atomics.
|
||||
AtomicTSOEmulationEnabled = false;
|
||||
}
|
||||
else {
|
||||
// Atomic TSO emulation only enabled if the config option is enabled.
|
||||
AtomicTSOEmulationEnabled = (IsMemoryShared || !Config.TSOAutoMigration) && Config.TSOEnabled;
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
/**
|
||||
* @brief Does some final thread initialization
|
||||
@@ -411,11 +431,13 @@ namespace FEXCore::Context {
|
||||
|
||||
bool StartPaused = false;
|
||||
bool IsMemoryShared = false;
|
||||
bool SupportsHardwareTSO = false;
|
||||
bool AtomicTSOEmulationEnabled = true;
|
||||
bool ExitOnHLT = false;
|
||||
FEX_CONFIG_OPT(AppFilename, APP_FILENAME);
|
||||
|
||||
std::shared_mutex CustomIRMutex;
|
||||
fextl::unordered_map<uint64_t, std::tuple<std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)>, void *, void *>> CustomIRHandlers;
|
||||
FEXCore::CPU::CPUBackendFeatures BackendFeatures;
|
||||
fextl::unordered_map<uint64_t, std::tuple<CustomIREntrypointHandler, void *, void *>> CustomIRHandlers;
|
||||
FEXCore::CPU::DispatcherConfig DispatcherConfig;
|
||||
};
|
||||
|
||||
|
||||
+214
-79
@@ -20,6 +20,131 @@
|
||||
#include <utility>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
// Register x18 is unused in the current configuration.
|
||||
// This is due to it being a platform register on wine platforms.
|
||||
// TODO: Allow x18 register allocation in the future to gain one more register.
|
||||
|
||||
namespace x64 {
|
||||
// All but x19 and x29 are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 16> SRA = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
|
||||
FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
|
||||
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 9> RA = {
|
||||
// All these callee saved
|
||||
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
|
||||
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
|
||||
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
|
||||
FEXCore::ARMEmitter::Reg::r30,
|
||||
};
|
||||
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 4> RAPair = {{
|
||||
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
|
||||
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
|
||||
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
|
||||
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
|
||||
}};
|
||||
|
||||
// All are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 16> SRAFPR = {
|
||||
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17,
|
||||
FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
|
||||
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21,
|
||||
FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
|
||||
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
|
||||
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
|
||||
};
|
||||
|
||||
// v8..v15 = (lower 64bits) Callee saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 12> RAFPR = {
|
||||
// v0 ~ v3 are used as temps.
|
||||
// FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
|
||||
// FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
|
||||
|
||||
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
|
||||
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
|
||||
};
|
||||
}
|
||||
|
||||
namespace x32 {
|
||||
// All but x19 and x29 are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 8> SRA = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
|
||||
FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 17> RA = {
|
||||
// All these callee saved
|
||||
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
|
||||
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
|
||||
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
|
||||
|
||||
// Registers only available on 32-bit
|
||||
// All these are caller saved (except for r19).
|
||||
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r29, FEXCore::ARMEmitter::Reg::r30,
|
||||
|
||||
FEXCore::ARMEmitter::Reg::r19,
|
||||
};
|
||||
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 8> RAPair = {{
|
||||
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
|
||||
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
|
||||
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
|
||||
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
|
||||
|
||||
{FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13},
|
||||
{FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15},
|
||||
{FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17},
|
||||
{FEXCore::ARMEmitter::Reg::r29, FEXCore::ARMEmitter::Reg::r30},
|
||||
}};
|
||||
|
||||
// All are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 8> SRAFPR = {
|
||||
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17,
|
||||
FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
|
||||
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21,
|
||||
FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
|
||||
};
|
||||
|
||||
// v8..v15 = (lower 64bits) Callee saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 20> RAFPR = {
|
||||
// v0 ~ v3 are used as temps.
|
||||
// FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
|
||||
// FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
|
||||
|
||||
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
|
||||
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
|
||||
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
|
||||
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
|
||||
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
|
||||
};
|
||||
}
|
||||
|
||||
// We want vixl to not allocate a default buffer. Jit and dispatcher will manually create one.
|
||||
Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, size_t size)
|
||||
@@ -29,22 +154,21 @@ Arm64Emitter::Arm64Emitter(FEXCore::Context::ContextImpl *ctx, size_t size)
|
||||
|
||||
// Number of register available is dependent on what operating mode the proccess is in.
|
||||
if (EmitterCTX->Config.Is64BitMode()) {
|
||||
ConfiguredGPRs = NumGPRs64;
|
||||
ConfiguredSRAGPRs = NumSRAGPRs64;
|
||||
ConfiguredGPRPairs = NumGPRPairs64;
|
||||
ConfiguredFPRs = NumFPRs64;
|
||||
ConfiguredSRAFPRs = NumSRAFPRs64;
|
||||
ConfiguredDynamicGPRs = NumGPRs64 - NumGPRs64; // Will be zero, just to be consistent with 32-bit side
|
||||
ConfiguredDynamicRegisterBase = nullptr;
|
||||
StaticRegisters = x64::SRA;
|
||||
GeneralRegisters = x64::RA;
|
||||
GeneralPairRegisters = x64::RAPair;
|
||||
StaticFPRegisters = x64::SRAFPR;
|
||||
GeneralFPRegisters = x64::RAFPR;
|
||||
}
|
||||
else {
|
||||
ConfiguredGPRs = NumGPRs32;
|
||||
ConfiguredSRAGPRs = NumSRAGPRs32;
|
||||
ConfiguredGPRPairs = NumGPRPairs32;
|
||||
ConfiguredFPRs = NumFPRs32;
|
||||
ConfiguredSRAFPRs = NumSRAFPRs32;
|
||||
ConfiguredDynamicGPRs = NumGPRs32 - NumGPRs64; // Will be 8
|
||||
ConfiguredDynamicRegisterBase = &RA64[9];
|
||||
ConfiguredDynamicRegisterBase = std::span(x32::RA.begin() + 8, 8);
|
||||
|
||||
StaticRegisters = x32::SRA;
|
||||
GeneralRegisters = x32::RA;
|
||||
GeneralPairRegisters = x32::RAPair;
|
||||
|
||||
StaticFPRegisters = x32::SRAFPR;
|
||||
GeneralFPRegisters = x32::RAFPR;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -68,6 +192,23 @@ void Arm64Emitter::LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, ui
|
||||
return;
|
||||
}
|
||||
|
||||
if ((Constant >> 32) == 0) {
|
||||
// If the upper 32-bits is all zero, we can now switch to a 32-bit move.
|
||||
s = ARMEmitter::Size::i32Bit;
|
||||
Is64Bit = false;
|
||||
Segments = 2;
|
||||
}
|
||||
|
||||
// If this can be loaded with a mov bitmask.
|
||||
const auto IsImm = vixl::aarch64::Assembler::IsImmLogical(Constant, RegSizeInBits(s));
|
||||
if (IsImm) {
|
||||
orr(s, Reg, ARMEmitter::Reg::zr, Constant);
|
||||
if (NOPPad) {
|
||||
nop(); nop(); nop();
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
int NumMoves = 1;
|
||||
int RequiredMoveSegments{};
|
||||
|
||||
@@ -219,14 +360,14 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Emitter::SpillStaticRegs(bool FPRs, uint32_t GPRSpillMask, uint32_t FPRSpillMask) {
|
||||
void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FPRs, uint32_t GPRSpillMask, uint32_t FPRSpillMask) {
|
||||
if (!StaticRegisterAllocation()) {
|
||||
return;
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < ConfiguredSRAGPRs; i+=2) {
|
||||
auto Reg1 = SRA64[i];
|
||||
auto Reg2 = SRA64[i+1];
|
||||
for (size_t i = 0; i < StaticRegisters.size(); i+=2) {
|
||||
auto Reg1 = StaticRegisters[i];
|
||||
auto Reg2 = StaticRegisters[i+1];
|
||||
if (((1U << Reg1.Idx()) & GPRSpillMask) &&
|
||||
((1U << Reg2.Idx()) & GPRSpillMask)) {
|
||||
stp<ARMEmitter::IndexType::OFFSET>(Reg1.X(), Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
|
||||
@@ -241,8 +382,8 @@ void Arm64Emitter::SpillStaticRegs(bool FPRs, uint32_t GPRSpillMask, uint32_t FP
|
||||
|
||||
if (FPRs) {
|
||||
if (EmitterCTX->HostFeatures.SupportsAVX) {
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i++) {
|
||||
const auto Reg = SRAFPR[i];
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i++) {
|
||||
const auto Reg = StaticFPRegisters[i];
|
||||
|
||||
if (((1U << Reg.Idx()) & FPRSpillMask) != 0) {
|
||||
mov(ARMEmitter::Size::i64Bit, TMP4.R(), offsetof(Core::CpuStateFrame, State.xmm.avx.data[i][0]));
|
||||
@@ -252,22 +393,20 @@ void Arm64Emitter::SpillStaticRegs(bool FPRs, uint32_t GPRSpillMask, uint32_t FP
|
||||
} else {
|
||||
if (GPRSpillMask && FPRSpillMask == ~0U) {
|
||||
// Optimize the common case where we can spill four registers per instruction
|
||||
auto TmpReg = SRA64[FindFirstSetBit(GPRSpillMask)];
|
||||
|
||||
// Load the sse offset in to the temporary register
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[0][0]));
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i += 4) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
const auto Reg3 = SRAFPR[i + 2];
|
||||
const auto Reg4 = SRAFPR[i + 3];
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
const auto Reg3 = StaticFPRegisters[i + 2];
|
||||
const auto Reg4 = StaticFPRegisters[i + 3];
|
||||
st1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
|
||||
}
|
||||
}
|
||||
else {
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i += 2) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 2) {
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
|
||||
if (((1U << Reg1.Idx()) & FPRSpillMask) &&
|
||||
((1U << Reg2.Idx()) & FPRSpillMask)) {
|
||||
@@ -299,8 +438,8 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
ptrue<ARMEmitter::SubRegSize::i8Bit>(PRED_TMP_16B, ARMEmitter::PredicatePattern::SVE_VL16);
|
||||
ptrue<ARMEmitter::SubRegSize::i8Bit>(PRED_TMP_32B, ARMEmitter::PredicatePattern::SVE_VL32);
|
||||
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i++) {
|
||||
const auto Reg = SRAFPR[i];
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i++) {
|
||||
const auto Reg = StaticFPRegisters[i];
|
||||
if (((1U << Reg.Idx()) & FPRFillMask) != 0) {
|
||||
mov(ARMEmitter::Size::i64Bit, TMP4.R(), offsetof(Core::CpuStateFrame, State.xmm.avx.data[i][0]));
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Reg.Z(), PRED_TMP_32B.Zeroing(), STATE.R(), TMP4.R());
|
||||
@@ -310,22 +449,22 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
if (GPRFillMask && FPRFillMask == ~0U) {
|
||||
// Optimize the common case where we can fill four registers per instruction.
|
||||
// Use one of the filling static registers before we fill it.
|
||||
auto TmpReg = SRA64[FindFirstSetBit(GPRFillMask)];
|
||||
auto TmpReg = StaticRegisters[FindFirstSetBit(GPRFillMask)];
|
||||
|
||||
// Load the sse offset in to the temporary register
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.xmm.sse.data[0][0]));
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i += 4) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
const auto Reg3 = SRAFPR[i + 2];
|
||||
const auto Reg4 = SRAFPR[i + 3];
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
const auto Reg3 = StaticFPRegisters[i + 2];
|
||||
const auto Reg4 = StaticFPRegisters[i + 3];
|
||||
ld1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
|
||||
}
|
||||
}
|
||||
else {
|
||||
for (size_t i = 0; i < ConfiguredSRAFPRs; i += 2) {
|
||||
const auto Reg1 = SRAFPR[i];
|
||||
const auto Reg2 = SRAFPR[i + 1];
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); i += 2) {
|
||||
const auto Reg1 = StaticFPRegisters[i];
|
||||
const auto Reg2 = StaticFPRegisters[i + 1];
|
||||
|
||||
if (((1U << Reg1.Idx()) & FPRFillMask) &&
|
||||
((1U << Reg2.Idx()) & FPRFillMask)) {
|
||||
@@ -342,9 +481,9 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
}
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < ConfiguredSRAGPRs; i+=2) {
|
||||
auto Reg1 = SRA64[i];
|
||||
auto Reg2 = SRA64[i+1];
|
||||
for (size_t i = 0; i < StaticRegisters.size(); i+=2) {
|
||||
auto Reg1 = StaticRegisters[i];
|
||||
auto Reg2 = StaticRegisters[i+1];
|
||||
if (((1U << Reg1.Idx()) & GPRFillMask) &&
|
||||
((1U << Reg2.Idx()) & GPRFillMask)) {
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(Reg1.X(), Reg2.X(), STATE.R(), offsetof(FEXCore::Core::CpuStateFrame, State.gregs[i]));
|
||||
@@ -360,10 +499,10 @@ void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRF
|
||||
|
||||
void Arm64Emitter::PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg) {
|
||||
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
|
||||
const auto GPRSize = (ConfiguredDynamicGPRs + 1) * Core::CPUState::GPR_REG_SIZE;
|
||||
const auto GPRSize = (ConfiguredDynamicRegisterBase.size() + 1) * Core::CPUState::GPR_REG_SIZE;
|
||||
const auto FPRRegSize = CanUseSVE ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto FPRSize = ConfiguredFPRs * FPRRegSize;
|
||||
const auto FPRSize = GeneralFPRegisters.size() * FPRRegSize;
|
||||
const uint64_t SPOffset = AlignUp(GPRSize + FPRSize, 16);
|
||||
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, SPOffset);
|
||||
@@ -372,31 +511,29 @@ void Arm64Emitter::PushDynamicRegsAndLR(FEXCore::ARMEmitter::Register TmpReg) {
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, ARMEmitter::Reg::rsp, 0);
|
||||
|
||||
if (CanUseSVE) {
|
||||
for (size_t i = 0; i < ConfiguredFPRs; i += 4) {
|
||||
const auto Reg1 = RAFPR[i];
|
||||
const auto Reg2 = RAFPR[i + 1];
|
||||
const auto Reg3 = RAFPR[i + 2];
|
||||
const auto Reg4 = RAFPR[i + 3];
|
||||
for (size_t i = 0; i < GeneralFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = GeneralFPRegisters[i];
|
||||
const auto Reg2 = GeneralFPRegisters[i + 1];
|
||||
const auto Reg3 = GeneralFPRegisters[i + 2];
|
||||
const auto Reg4 = GeneralFPRegisters[i + 3];
|
||||
st4b(Reg1.Z(), Reg2.Z(), Reg3.Z(), Reg4.Z(), PRED_TMP_32B, TmpReg, 0);
|
||||
add(ARMEmitter::Size::i64Bit, TmpReg, TmpReg, 32 * 4);
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(ConfiguredFPRs % 4 == 0, "Needs to have multiple of 4 FPRs for RA");
|
||||
for (size_t i = 0; i < ConfiguredFPRs; i += 4) {
|
||||
const auto Reg1 = RAFPR[i];
|
||||
const auto Reg2 = RAFPR[i + 1];
|
||||
const auto Reg3 = RAFPR[i + 2];
|
||||
const auto Reg4 = RAFPR[i + 3];
|
||||
LOGMAN_THROW_A_FMT(GeneralFPRegisters.size() % 4 == 0, "Needs to have multiple of 4 FPRs for RA");
|
||||
for (size_t i = 0; i < GeneralFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = GeneralFPRegisters[i];
|
||||
const auto Reg2 = GeneralFPRegisters[i + 1];
|
||||
const auto Reg3 = GeneralFPRegisters[i + 2];
|
||||
const auto Reg4 = GeneralFPRegisters[i + 3];
|
||||
st1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), TmpReg, 64);
|
||||
}
|
||||
}
|
||||
|
||||
if (ConfiguredDynamicRegisterBase) {
|
||||
for (size_t i = 0; i < ConfiguredDynamicGPRs; i += 2) {
|
||||
const auto Reg1 = ConfiguredDynamicRegisterBase[i];
|
||||
const auto Reg2 = ConfiguredDynamicRegisterBase[i + 1];
|
||||
stp<ARMEmitter::IndexType::POST>(Reg1.X(), Reg2.X(), TmpReg, 16);
|
||||
}
|
||||
for (size_t i = 0; i < ConfiguredDynamicRegisterBase.size(); i += 2) {
|
||||
const auto Reg1 = ConfiguredDynamicRegisterBase[i];
|
||||
const auto Reg2 = ConfiguredDynamicRegisterBase[i + 1];
|
||||
stp<ARMEmitter::IndexType::POST>(Reg1.X(), Reg2.X(), TmpReg, 16);
|
||||
}
|
||||
|
||||
str(ARMEmitter::XReg::lr, TmpReg, 0);
|
||||
@@ -406,30 +543,28 @@ void Arm64Emitter::PopDynamicRegsAndLR() {
|
||||
const auto CanUseSVE = EmitterCTX->HostFeatures.SupportsAVX;
|
||||
|
||||
if (CanUseSVE) {
|
||||
for (size_t i = 0; i < ConfiguredFPRs; i += 4) {
|
||||
const auto Reg1 = RAFPR[i];
|
||||
const auto Reg2 = RAFPR[i + 1];
|
||||
const auto Reg3 = RAFPR[i + 2];
|
||||
const auto Reg4 = RAFPR[i + 3];
|
||||
for (size_t i = 0; i < GeneralFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = GeneralFPRegisters[i];
|
||||
const auto Reg2 = GeneralFPRegisters[i + 1];
|
||||
const auto Reg3 = GeneralFPRegisters[i + 2];
|
||||
const auto Reg4 = GeneralFPRegisters[i + 3];
|
||||
ld4b(Reg1.Z(), Reg2.Z(), Reg3.Z(), Reg4.Z(), PRED_TMP_32B.Zeroing(), ARMEmitter::Reg::rsp);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 32 * 4);
|
||||
}
|
||||
} else {
|
||||
for (size_t i = 0; i < ConfiguredFPRs; i += 4) {
|
||||
const auto Reg1 = RAFPR[i];
|
||||
const auto Reg2 = RAFPR[i + 1];
|
||||
const auto Reg3 = RAFPR[i + 2];
|
||||
const auto Reg4 = RAFPR[i + 3];
|
||||
for (size_t i = 0; i < GeneralFPRegisters.size(); i += 4) {
|
||||
const auto Reg1 = GeneralFPRegisters[i];
|
||||
const auto Reg2 = GeneralFPRegisters[i + 1];
|
||||
const auto Reg3 = GeneralFPRegisters[i + 2];
|
||||
const auto Reg4 = GeneralFPRegisters[i + 3];
|
||||
ld1<ARMEmitter::SubRegSize::i64Bit>(Reg1.Q(), Reg2.Q(), Reg3.Q(), Reg4.Q(), ARMEmitter::Reg::rsp, 64);
|
||||
}
|
||||
}
|
||||
|
||||
if (ConfiguredDynamicRegisterBase) {
|
||||
for (size_t i = 0; i < ConfiguredDynamicGPRs; i += 2) {
|
||||
const auto Reg1 = ConfiguredDynamicRegisterBase[i];
|
||||
const auto Reg2 = ConfiguredDynamicRegisterBase[i + 1];
|
||||
ldp<ARMEmitter::IndexType::POST>(Reg1.X(), Reg2.X(), ARMEmitter::Reg::rsp, 16);
|
||||
}
|
||||
for (size_t i = 0; i < ConfiguredDynamicRegisterBase.size(); i += 2) {
|
||||
const auto Reg1 = ConfiguredDynamicRegisterBase[i];
|
||||
const auto Reg2 = ConfiguredDynamicRegisterBase[i + 1];
|
||||
ldp<ARMEmitter::IndexType::POST>(Reg1.X(), Reg2.X(), ARMEmitter::Reg::rsp, 16);
|
||||
}
|
||||
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
|
||||
@@ -26,87 +26,9 @@
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <utility>
|
||||
#include <span>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
// Register x18 is unused in the current configuration.
|
||||
// This is due to it being a platform register on wine platforms.
|
||||
// TODO: Allow x18 register allocation in the future to gain one more register.
|
||||
|
||||
// All but x19 and x29 are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 16> SRA64 = {
|
||||
FEXCore::ARMEmitter::Reg::r4, FEXCore::ARMEmitter::Reg::r5,
|
||||
FEXCore::ARMEmitter::Reg::r6, FEXCore::ARMEmitter::Reg::r7,
|
||||
FEXCore::ARMEmitter::Reg::r8, FEXCore::ARMEmitter::Reg::r9,
|
||||
FEXCore::ARMEmitter::Reg::r10, FEXCore::ARMEmitter::Reg::r11,
|
||||
// Registers that don't exist on 32-bit
|
||||
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29
|
||||
};
|
||||
|
||||
constexpr std::array<FEXCore::ARMEmitter::Register, 9 + 8> RA64 = {
|
||||
// All these callee saved
|
||||
FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21,
|
||||
FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23,
|
||||
FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25,
|
||||
FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27,
|
||||
FEXCore::ARMEmitter::Reg::r30,
|
||||
|
||||
// Registers only available on 32-bit
|
||||
// All these are caller saved (except for r19).
|
||||
FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13,
|
||||
FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15,
|
||||
FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17,
|
||||
FEXCore::ARMEmitter::Reg::r19, FEXCore::ARMEmitter::Reg::r29
|
||||
};
|
||||
|
||||
constexpr std::array<std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>, 4 + 3> RA64Pair = {{
|
||||
{FEXCore::ARMEmitter::Reg::r20, FEXCore::ARMEmitter::Reg::r21},
|
||||
{FEXCore::ARMEmitter::Reg::r22, FEXCore::ARMEmitter::Reg::r23},
|
||||
{FEXCore::ARMEmitter::Reg::r24, FEXCore::ARMEmitter::Reg::r25},
|
||||
{FEXCore::ARMEmitter::Reg::r26, FEXCore::ARMEmitter::Reg::r27},
|
||||
|
||||
// Registers only available on 32-bit
|
||||
{FEXCore::ARMEmitter::Reg::r12, FEXCore::ARMEmitter::Reg::r13},
|
||||
{FEXCore::ARMEmitter::Reg::r14, FEXCore::ARMEmitter::Reg::r15},
|
||||
{FEXCore::ARMEmitter::Reg::r16, FEXCore::ARMEmitter::Reg::r17}
|
||||
}};
|
||||
|
||||
// All are caller saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 16> SRAFPR = {
|
||||
FEXCore::ARMEmitter::VReg::v16, FEXCore::ARMEmitter::VReg::v17,
|
||||
FEXCore::ARMEmitter::VReg::v18, FEXCore::ARMEmitter::VReg::v19,
|
||||
FEXCore::ARMEmitter::VReg::v20, FEXCore::ARMEmitter::VReg::v21,
|
||||
FEXCore::ARMEmitter::VReg::v22, FEXCore::ARMEmitter::VReg::v23,
|
||||
|
||||
// Registers that don't exist on 32-bit
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
|
||||
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
|
||||
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
|
||||
};
|
||||
|
||||
// v8..v15 = (lower 64bits) Callee saved
|
||||
constexpr std::array<FEXCore::ARMEmitter::VRegister, 12 + 8> RAFPR = {
|
||||
// v0 ~ v3 are used as temps.
|
||||
// FEXCore::ARMEmitter::VReg::v0, FEXCore::ARMEmitter::VReg::v1,
|
||||
// FEXCore::ARMEmitter::VReg::v2, FEXCore::ARMEmitter::VReg::v3,
|
||||
|
||||
FEXCore::ARMEmitter::VReg::v4, FEXCore::ARMEmitter::VReg::v5,
|
||||
FEXCore::ARMEmitter::VReg::v6, FEXCore::ARMEmitter::VReg::v7,
|
||||
FEXCore::ARMEmitter::VReg::v8, FEXCore::ARMEmitter::VReg::v9,
|
||||
FEXCore::ARMEmitter::VReg::v10, FEXCore::ARMEmitter::VReg::v11,
|
||||
FEXCore::ARMEmitter::VReg::v12, FEXCore::ARMEmitter::VReg::v13,
|
||||
FEXCore::ARMEmitter::VReg::v14, FEXCore::ARMEmitter::VReg::v15,
|
||||
|
||||
// Registers only available on 32-bit
|
||||
FEXCore::ARMEmitter::VReg::v24, FEXCore::ARMEmitter::VReg::v25,
|
||||
FEXCore::ARMEmitter::VReg::v26, FEXCore::ARMEmitter::VReg::v27,
|
||||
FEXCore::ARMEmitter::VReg::v28, FEXCore::ARMEmitter::VReg::v29,
|
||||
FEXCore::ARMEmitter::VReg::v30, FEXCore::ARMEmitter::VReg::v31
|
||||
};
|
||||
|
||||
// Contains the address to the currently available CPU state
|
||||
constexpr auto STATE = FEXCore::ARMEmitter::XReg::x28;
|
||||
|
||||
@@ -139,32 +61,16 @@ protected:
|
||||
FEXCore::Context::ContextImpl *EmitterCTX;
|
||||
vixl::aarch64::CPU CPU;
|
||||
|
||||
uint32_t ConfiguredGPRs;
|
||||
uint32_t ConfiguredSRAGPRs;
|
||||
uint32_t ConfiguredGPRPairs;
|
||||
uint32_t ConfiguredFPRs;
|
||||
uint32_t ConfiguredSRAFPRs;
|
||||
uint32_t ConfiguredDynamicGPRs;
|
||||
const FEXCore::ARMEmitter::Register *ConfiguredDynamicRegisterBase{};
|
||||
std::span<const FEXCore::ARMEmitter::Register> ConfiguredDynamicRegisterBase{};
|
||||
std::span<const FEXCore::ARMEmitter::Register> StaticRegisters{};
|
||||
std::span<const FEXCore::ARMEmitter::Register> GeneralRegisters{};
|
||||
std::span<const std::pair<FEXCore::ARMEmitter::Register, FEXCore::ARMEmitter::Register>> GeneralPairRegisters{};
|
||||
std::span<const FEXCore::ARMEmitter::VRegister> StaticFPRegisters{};
|
||||
std::span<const FEXCore::ARMEmitter::VRegister> GeneralFPRegisters{};
|
||||
|
||||
/**
|
||||
* @name Register Allocation
|
||||
* @{ */
|
||||
// 64-bit gets removal of additional pairs
|
||||
constexpr static uint32_t NumGPRs64 = RA64.size() - 8;
|
||||
constexpr static uint32_t NumSRAGPRs64 = SRA64.size();
|
||||
constexpr static uint32_t NumFPRs64 = RAFPR.size() - 8;
|
||||
constexpr static uint32_t NumSRAFPRs64 = SRAFPR.size();
|
||||
constexpr static uint32_t NumGPRPairs64 = RA64Pair.size() - 3;
|
||||
|
||||
// 32-bit gets full array of GPR registers
|
||||
// SRA registers remove the additional 8
|
||||
constexpr static uint32_t NumGPRs32 = RA64.size();
|
||||
constexpr static uint32_t NumSRAGPRs32 = SRA64.size() - 8;
|
||||
constexpr static uint32_t NumFPRs32 = RAFPR.size();
|
||||
constexpr static uint32_t NumSRAFPRs32 = SRAFPR.size() - 8;
|
||||
constexpr static uint32_t NumGPRPairs32 = RA64Pair.size();
|
||||
|
||||
constexpr static uint32_t RegisterClasses = 6;
|
||||
|
||||
constexpr static uint64_t GPRBase = (0ULL << 32);
|
||||
@@ -183,7 +89,7 @@ protected:
|
||||
// NOTE: These functions WILL clobber the register TMP4 if AVX support is enabled
|
||||
// and FPRs are being spilled or filled. If only GPRs are spilled/filled, then
|
||||
// TMP4 is left alone.
|
||||
void SpillStaticRegs(bool FPRs = true, uint32_t GPRSpillMask = ~0U, uint32_t FPRSpillMask = ~0U);
|
||||
void SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FPRs = true, uint32_t GPRSpillMask = ~0U, uint32_t FPRSpillMask = ~0U);
|
||||
void FillStaticRegs(bool FPRs = true, uint32_t GPRFillMask = ~0U, uint32_t FPRFillMask = ~0U);
|
||||
|
||||
// Register 0-18 + 29 + 30 are caller saved
|
||||
@@ -269,6 +175,7 @@ protected:
|
||||
#endif
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
vixl::aarch64::PrintDisassembler Disasm {stderr};
|
||||
FEX_CONFIG_OPT(Disassemble, DISASSEMBLE);
|
||||
#endif
|
||||
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
|
||||
};
|
||||
|
||||
+164
-4
@@ -129,7 +129,9 @@ public:
|
||||
constexpr uint32_t Op = 0b0011'0001'0 << 23;
|
||||
DataProcessing_AddSub_Imm(Op, s, rd, rn, Imm, LSL12);
|
||||
}
|
||||
|
||||
void cmn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, uint32_t Imm, bool LSL12 = false) {
|
||||
adds(s, FEXCore::ARMEmitter::Reg::zr, rn, Imm, LSL12);
|
||||
}
|
||||
void sub(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t Imm, bool LSL12 = false) {
|
||||
constexpr uint32_t Op = 0b0101'0001'0 << 23;
|
||||
DataProcessing_AddSub_Imm(Op, s, rd, rn, Imm, LSL12);
|
||||
@@ -145,6 +147,27 @@ public:
|
||||
DataProcessing_AddSub_Imm(Op, s, rd, rn, Imm, LSL12);
|
||||
}
|
||||
|
||||
// Min/max immediate
|
||||
void smax(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, int64_t Imm) {
|
||||
LOGMAN_THROW_A_FMT(Imm >= -128 && Imm <= 127, "{} Immediate too large", __func__);
|
||||
MinMaxImmediate(0b0000, s, rd, rn, Imm);
|
||||
}
|
||||
|
||||
void umax(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
|
||||
LOGMAN_THROW_A_FMT(Imm <= 255, "{} Immediate too large", __func__);
|
||||
MinMaxImmediate(0b0001, s, rd, rn, Imm);
|
||||
}
|
||||
|
||||
void smin(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, int64_t Imm) {
|
||||
LOGMAN_THROW_A_FMT(Imm >= -128 && Imm <= 127, "{} Immediate too large", __func__);
|
||||
MinMaxImmediate(0b0010, s, rd, rn, Imm);
|
||||
}
|
||||
|
||||
void umin(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
|
||||
LOGMAN_THROW_A_FMT(Imm <= 255, "{} Immediate too large", __func__);
|
||||
MinMaxImmediate(0b0011, s, rd, rn, Imm);
|
||||
}
|
||||
|
||||
// Logical immediate
|
||||
void and_(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
|
||||
uint32_t n, immr, imms;
|
||||
@@ -198,6 +221,10 @@ public:
|
||||
eor(s, rd, rn, n, immr, imms);
|
||||
}
|
||||
|
||||
void tst(ARMEmitter::Size s, Register rn, uint64_t imm) {
|
||||
ands(s, Reg::zr, rn, imm);
|
||||
}
|
||||
|
||||
// Move wide immediate
|
||||
void movn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, uint32_t Imm, uint32_t Offset = 0) {
|
||||
LOGMAN_THROW_A_FMT((Imm & 0xFFFF0000U) == 0, "Upper bits of move wide not valid");
|
||||
@@ -265,6 +292,9 @@ public:
|
||||
LOGMAN_THROW_A_FMT((lsb + width) <= RegSizeInBits(s), "Tried to sbfx a region larger than the register");
|
||||
sbfm(s, rd, rn, lsb, lsb + width - 1);
|
||||
}
|
||||
void sbfiz(ARMEmitter::Size s, Register rd, Register rn, uint32_t lsb, uint32_t width) {
|
||||
xbfiz_helper(true, s, rd, rn, lsb, width);
|
||||
}
|
||||
void asr(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t shift) {
|
||||
LOGMAN_THROW_A_FMT(shift <= RegSizeInBits(s), "Tried to asr a region larger than the register");
|
||||
sbfm(s, rd, rn, shift, RegSizeInBits(s) - 1);
|
||||
@@ -285,6 +315,10 @@ public:
|
||||
DataProcessing_Logical_Imm(Op, s, rd, rn, s == ARMEmitter::Size::i64Bit, immr, imms);
|
||||
}
|
||||
|
||||
void ubfiz(ARMEmitter::Size s, Register rd, Register rn, uint32_t lsb, uint32_t width) {
|
||||
xbfiz_helper(false, s, rd, rn, lsb, width);
|
||||
}
|
||||
|
||||
void lsl(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t shift) {
|
||||
const auto RegSize = RegSizeInBits(s);
|
||||
LOGMAN_THROW_A_FMT(shift < RegSize, "Tried to lsl a region larger than the register");
|
||||
@@ -303,10 +337,23 @@ public:
|
||||
|
||||
void bfi(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t lsb, uint32_t width) {
|
||||
const auto RegSize = RegSizeInBits(s);
|
||||
LOGMAN_THROW_A_FMT(width > 0, "bfi needs width > 0");
|
||||
LOGMAN_THROW_A_FMT((lsb + width) <= RegSize, "Tried to bfi a region larger than the register");
|
||||
LOGMAN_THROW_A_FMT(width > 0, "bfc/bfi needs width > 0");
|
||||
LOGMAN_THROW_A_FMT((lsb + width) <= RegSize, "Tried to bfc/bfi a region larger than the register");
|
||||
bfm(s, rd, rn, (RegSize - lsb) & (RegSize - 1), width - 1);
|
||||
}
|
||||
void bfc(ARMEmitter::Size s, Register rd, uint32_t lsb, uint32_t width) {
|
||||
bfi(s, rd, Reg::zr, lsb, width);
|
||||
}
|
||||
void bfxil(ARMEmitter::Size s, Register rd, Register rn, uint32_t lsb, uint32_t width) {
|
||||
[[maybe_unused]] const auto reg_size_bits = RegSizeInBits(s);
|
||||
const auto lsb_p_width = lsb + width;
|
||||
|
||||
LOGMAN_THROW_A_FMT(width >= 1, "bfxil needs width >= 1");
|
||||
LOGMAN_THROW_A_FMT(lsb_p_width <= reg_size_bits, "bfxil lsb + width ({}) must be <= {}. lsb={}, width={}",
|
||||
lsb_p_width, reg_size_bits, lsb, width);
|
||||
|
||||
bfm(s, rd, rn, lsb, lsb_p_width - 1);
|
||||
}
|
||||
|
||||
// Extract
|
||||
void extr(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, uint32_t Imm) {
|
||||
@@ -381,6 +428,26 @@ public:
|
||||
(0b0101'10U << 10);
|
||||
DataProcessing_2Source(Op, ARMEmitter::Size::i32Bit, rd, rn, rm);
|
||||
}
|
||||
void smax(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0110'00U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void umax(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0110'01U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void smin(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0110'10U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void umin(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0110'11U << 10);
|
||||
DataProcessing_2Source(Op, s, rd, rn, rm);
|
||||
}
|
||||
void subp(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm) {
|
||||
constexpr uint32_t Op = (0b001'1010'110U << 21) |
|
||||
(0b0000'00U << 10);
|
||||
@@ -467,7 +534,24 @@ public:
|
||||
(s == ARMEmitter::Size::i64Bit ? (1U << 10) : 0);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
|
||||
void ctz(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0001'10U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void cnt(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0001'11U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
void abs(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn) {
|
||||
constexpr uint32_t Op = (0b101'1010'110U << 21) |
|
||||
(0b0'0000U << 16) |
|
||||
(0b0010'00U << 10);
|
||||
DataProcessing_1Source(Op, s, rd, rn);
|
||||
}
|
||||
|
||||
// TODO: PAUTH
|
||||
|
||||
@@ -506,6 +590,9 @@ public:
|
||||
constexpr uint32_t Op = 0b010'1010'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void tst(ARMEmitter::Size s, Register rn, Register rm, ShiftType shift = ShiftType::LSL, uint32_t amt = 0) {
|
||||
ands(s, Reg::zr, rn, rm, shift, amt);
|
||||
}
|
||||
|
||||
void orn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
constexpr uint32_t Op = 0b010'1010'001U << 21;
|
||||
@@ -527,6 +614,9 @@ public:
|
||||
void adds(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
adds(ARMEmitter::Size::i64Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void cmn(FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
adds(ARMEmitter::Size::i64Bit, FEXCore::ARMEmitter::XReg::zr, rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void sub(FEXCore::ARMEmitter::XRegister rd, FEXCore::ARMEmitter::XRegister rn, FEXCore::ARMEmitter::XRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
sub(ARMEmitter::Size::i64Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
@@ -549,6 +639,9 @@ public:
|
||||
void adds(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
adds(ARMEmitter::Size::i32Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void cmn(FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
adds(ARMEmitter::Size::i32Bit, FEXCore::ARMEmitter::WReg::zr, rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
void sub(FEXCore::ARMEmitter::WRegister rd, FEXCore::ARMEmitter::WRegister rn, FEXCore::ARMEmitter::WRegister rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
sub(ARMEmitter::Size::i32Bit, rd.R(), rn.R(), rm.R(), Shift, amt);
|
||||
}
|
||||
@@ -575,6 +668,9 @@ public:
|
||||
constexpr uint32_t Op = 0b010'1011'000U << 21;
|
||||
DataProcessing_Shifted_Reg(Op, s, rd, rn, rm, Shift, amt);
|
||||
}
|
||||
void cmn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
adds(s, FEXCore::ARMEmitter::Reg::zr, rn, rm, Shift, amt);
|
||||
}
|
||||
void sub(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift = FEXCore::ARMEmitter::ShiftType::LSL, uint32_t amt = 0) {
|
||||
LOGMAN_THROW_AA_FMT(Shift != FEXCore::ARMEmitter::ShiftType::ROR, "Doesn't support ROR");
|
||||
constexpr uint32_t Op = 0b100'1011'000U << 21;
|
||||
@@ -606,6 +702,9 @@ public:
|
||||
constexpr uint32_t Op = 0b010'1011'001U << 21;
|
||||
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, Option, Shift);
|
||||
}
|
||||
void cmn(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ExtendedType Option, uint32_t Shift = 0) {
|
||||
adds(s, FEXCore::ARMEmitter::Reg::zr, rn, rm, Option, Shift);
|
||||
}
|
||||
void sub(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ExtendedType Option, uint32_t Shift = 0) {
|
||||
constexpr uint32_t Op = 0b100'1011'001U << 21;
|
||||
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, Option, Shift);
|
||||
@@ -636,6 +735,12 @@ public:
|
||||
constexpr uint32_t Op = 0b0111'1010'000U << 21;
|
||||
DataProcessing_Extended_Reg(Op, s, rd, rn, rm, FEXCore::ARMEmitter::ExtendedType::UXTB, 0);
|
||||
}
|
||||
void ngc(ARMEmitter::Size s, Register rd, Register rm) {
|
||||
sbc(s, rd, Reg::zr, rm);
|
||||
}
|
||||
void ngcs(ARMEmitter::Size s, Register rd, Register rm) {
|
||||
sbcs(s, rd, Reg::zr, rm);
|
||||
}
|
||||
|
||||
// Rotate right into flags
|
||||
void rmif(XRegister rn, uint32_t shift, uint32_t mask) {
|
||||
@@ -703,6 +808,18 @@ public:
|
||||
constexpr uint32_t Op = 0b0001'1010'100 << 21;
|
||||
ConditionalCompare(Op, 1, 0b01, s, rd, rn, rm, Cond);
|
||||
}
|
||||
void cneg(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Condition Cond) {
|
||||
csneg(s, rd, rn, rn, InvertCondition(Cond));
|
||||
}
|
||||
void cinc(ARMEmitter::Size s, Register rd, Register rn, Condition cond) {
|
||||
csinc(s, rd, rn, rn, InvertCondition(cond));
|
||||
}
|
||||
void cinv(ARMEmitter::Size s, Register rd, Register rn, Condition cond) {
|
||||
csinv(s, rd, rn, rn, InvertCondition(cond));
|
||||
}
|
||||
void csetm(ARMEmitter::Size s, Register rd, Condition cond) {
|
||||
csinv(s, rd, Reg::zr, Reg::zr, InvertCondition(cond));
|
||||
}
|
||||
|
||||
// Data processing - 3 source
|
||||
void madd(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::Register ra) {
|
||||
@@ -757,6 +874,13 @@ public:
|
||||
}
|
||||
|
||||
private:
|
||||
static constexpr Condition InvertCondition(Condition cond) {
|
||||
// These behave as always, so it makes no sense to allow inverting these.
|
||||
LOGMAN_THROW_AA_FMT(cond != Condition::CC_AL && cond != Condition::CC_NV,
|
||||
"Cannot invert CC_AL or CC_NV");
|
||||
return static_cast<Condition>(FEXCore::ToUnderlying(cond) ^ 1);
|
||||
}
|
||||
|
||||
void and_(FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint32_t n, uint32_t immr, uint32_t imms) {
|
||||
constexpr uint32_t Op = 0b001'0010'00 << 22;
|
||||
DataProcessing_Logical_Imm(Op, s, rd, rn, n, immr, imms);
|
||||
@@ -819,6 +943,21 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Min/max immediate
|
||||
void MinMaxImmediate(uint32_t opc, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, uint64_t Imm) {
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
uint32_t Instr = 0b1'0001'11U << 22;
|
||||
|
||||
Instr |= SF;
|
||||
Instr |= opc << 18;
|
||||
Instr |= (Imm & 0xFF) << 10;
|
||||
Instr |= Encode_rn(rn);
|
||||
Instr |= Encode_rd(rd);
|
||||
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
// Move Wide
|
||||
void DataProcessing_MoveWide(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, uint32_t Imm, uint32_t Offset) {
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
@@ -849,6 +988,24 @@ private:
|
||||
dc32(Instr);
|
||||
}
|
||||
|
||||
void xbfiz_helper(bool is_signed, ARMEmitter::Size s, Register rd, Register rn, uint32_t lsb, uint32_t width) {
|
||||
[[maybe_unused]] const auto lsb_p_width = lsb + width;
|
||||
const auto reg_size_bits = RegSizeInBits(s);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(lsb_p_width <= reg_size_bits, "lsb + width ({}) must be <= {}. lsb={}, width={}",
|
||||
lsb_p_width, reg_size_bits, lsb, width);
|
||||
LOGMAN_THROW_AA_FMT(width >= 1, "xbfiz width must be >= 1");
|
||||
|
||||
const auto immr = (reg_size_bits - lsb) & (reg_size_bits - 1);
|
||||
const auto imms = width - 1;
|
||||
|
||||
if (is_signed) {
|
||||
sbfm(s, rd, rn, immr, imms);
|
||||
} else {
|
||||
ubfm(s, rd, rn, immr, imms);
|
||||
}
|
||||
}
|
||||
|
||||
void DataProcessing_Extract(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, uint32_t Imm) {
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
@@ -899,6 +1056,9 @@ private:
|
||||
// AddSub - shifted register
|
||||
void DataProcessing_Shifted_Reg(uint32_t Op, FEXCore::ARMEmitter::Size s, FEXCore::ARMEmitter::Register rd, FEXCore::ARMEmitter::Register rn, FEXCore::ARMEmitter::Register rm, FEXCore::ARMEmitter::ShiftType Shift, uint32_t amt) {
|
||||
LOGMAN_THROW_AA_FMT((amt & ~0b11'1111U) == 0, "Shift amount too large");
|
||||
if (s == FEXCore::ARMEmitter::Size::i32Bit) {
|
||||
LOGMAN_THROW_AA_FMT(amt < 32, "Shift amount for 32-bit must be below 32");
|
||||
}
|
||||
|
||||
const uint32_t SF = s == FEXCore::ARMEmitter::Size::i64Bit ? (1U << 31) : 0;
|
||||
|
||||
|
||||
@@ -6,6 +6,7 @@
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
|
||||
#include <aarch64/assembler-aarch64.h>
|
||||
@@ -207,58 +208,94 @@ namespace FEXCore::ARMEmitter {
|
||||
*/
|
||||
class SVEMemOperand final {
|
||||
public:
|
||||
// Used for scalar + vector variants to determine
|
||||
// extension behavior on the index values.
|
||||
enum class ModType : uint8_t {
|
||||
MOD_UXTW,
|
||||
MOD_SXTW,
|
||||
MOD_LSL,
|
||||
MOD_NONE,
|
||||
};
|
||||
|
||||
enum class Type {
|
||||
ScalarPlusScalar,
|
||||
ScalarPlusImm,
|
||||
ScalarPlusVector,
|
||||
VectorPlusImm,
|
||||
};
|
||||
|
||||
SVEMemOperand(XRegister rn, XRegister rm = XReg::zr)
|
||||
: rn {rn}
|
||||
, MemType{Type::ScalarPlusScalar}
|
||||
, MetaType {
|
||||
.ScalarScalarType {
|
||||
.Header = { .MemType = TYPE_SCALAR_SCALAR },
|
||||
.rm = rm,
|
||||
}
|
||||
} {}
|
||||
SVEMemOperand(XRegister rn, int32_t imm = 0)
|
||||
: rn {rn}
|
||||
, MemType{Type::ScalarPlusImm}
|
||||
, MetaType {
|
||||
.ScalarImmType {
|
||||
.Header = { .MemType = TYPE_SCALAR_IMM },
|
||||
.Imm = imm,
|
||||
}
|
||||
} {}
|
||||
SVEMemOperand(XRegister rn, ZRegister zm, ModType mod = ModType::MOD_NONE, uint8_t scale = 0)
|
||||
: rn{rn}
|
||||
, MemType{Type::ScalarPlusVector}
|
||||
, MetaType {
|
||||
.ScalarVectorType {
|
||||
.zm = zm,
|
||||
.mod = mod,
|
||||
.scale = scale,
|
||||
}
|
||||
} {}
|
||||
SVEMemOperand(ZRegister zn, uint32_t imm)
|
||||
: rn{Register{zn.Idx()}}
|
||||
, MemType{Type::VectorPlusImm}
|
||||
, MetaType {
|
||||
.VectorImmType{
|
||||
.Imm = imm,
|
||||
}
|
||||
} {}
|
||||
|
||||
Register rn;
|
||||
enum Type {
|
||||
TYPE_SCALAR_SCALAR,
|
||||
TYPE_SCALAR_IMM,
|
||||
TYPE_SCALAR_VECTOR,
|
||||
TYPE_VECTOR_IMM,
|
||||
};
|
||||
struct HeaderStruct {
|
||||
Type MemType;
|
||||
};
|
||||
[[nodiscard]] bool IsScalarPlusScalar() const {
|
||||
return MemType == Type::ScalarPlusScalar;
|
||||
}
|
||||
[[nodiscard]] bool IsScalarPlusImm() const {
|
||||
return MemType == Type::ScalarPlusImm;
|
||||
}
|
||||
[[nodiscard]] bool IsScalarPlusVector() const {
|
||||
return MemType == Type::ScalarPlusVector;
|
||||
}
|
||||
[[nodiscard]] bool IsVectorPlusImm() const {
|
||||
return MemType == Type::VectorPlusImm;
|
||||
}
|
||||
|
||||
union {
|
||||
HeaderStruct Header;
|
||||
union Data {
|
||||
struct {
|
||||
HeaderStruct Header;
|
||||
Register rm;
|
||||
} ScalarScalarType;
|
||||
|
||||
struct {
|
||||
HeaderStruct Header;
|
||||
int32_t Imm;
|
||||
} ScalarImmType;
|
||||
|
||||
struct {
|
||||
HeaderStruct Header;
|
||||
ZRegister zm;
|
||||
// TODO: Implement support for modifier
|
||||
ModType mod;
|
||||
uint8_t scale;
|
||||
} ScalarVectorType;
|
||||
|
||||
struct {
|
||||
HeaderStruct Header;
|
||||
// rn will be a ZRegister
|
||||
int32_t Imm;
|
||||
uint32_t Imm;
|
||||
} VectorImmType;
|
||||
} MetaType;
|
||||
};
|
||||
|
||||
Register rn;
|
||||
Type MemType;
|
||||
Data MetaType;
|
||||
};
|
||||
|
||||
/* This `ExtendedMemOperand` class is used for the helper load-store instructions.
|
||||
@@ -475,6 +512,20 @@ namespace FEXCore::ARMEmitter {
|
||||
SVE_ALL = 0b11111,
|
||||
};
|
||||
|
||||
// Used with SVE FP immediate arithmetic instructions
|
||||
enum class SVEFAddSubImm : uint32_t {
|
||||
_0_5,
|
||||
_1_0,
|
||||
};
|
||||
enum class SVEFMulImm : uint32_t {
|
||||
_0_5,
|
||||
_2_0,
|
||||
};
|
||||
enum class SVEFMaxMinImm : uint32_t {
|
||||
_0_0,
|
||||
_1_0,
|
||||
};
|
||||
|
||||
/* This `BackwardLabel` struct used for retaining a location for PC-Relative instructions.
|
||||
* This is specifically a label for a target that is logically `below` an instruction that uses it.
|
||||
* Which means that a branch would jump backwards.
|
||||
|
||||
+436
-546
File diff suppressed because it is too large.
Load diff
+1356
-1238
File diff suppressed because it is too large.
Load diff
+25
-16
@@ -76,10 +76,6 @@ static uint32_t GetCPUID() {
|
||||
return CPU;
|
||||
}
|
||||
|
||||
// TODO: Replace usages with CTX->HostFeatures.EnableAVX
|
||||
// when AVX implementations are further along.
|
||||
constexpr uint32_t SUPPORTS_AVX = 0;
|
||||
|
||||
#ifdef CPUID_AMD
|
||||
constexpr uint32_t FAMILY_IDENTIFIER =
|
||||
0 | // Stepping
|
||||
@@ -399,8 +395,6 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
uint32_t CoreCount = Cores();
|
||||
// XXX: Enable once the rest of the SSE4.2 instructions are emulated
|
||||
uint32_t SupportsSSE42 = CTX->HostFeatures.SupportsCRC && false ? 1 : 0;
|
||||
|
||||
// Hypervisor bit is normally set but some applications have issues with it.
|
||||
uint32_t Hypervisor = HideHypervisorBit() ? 0 : 1;
|
||||
@@ -433,15 +427,15 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) {
|
||||
(0 << 17) | // Process-context identifiers
|
||||
(0 << 18) | // Prefetching from memory mapped device
|
||||
(1 << 19) | // SSE4.1
|
||||
(SupportsSSE42 << 20) | // SSE4.2
|
||||
(CTX->HostFeatures.SupportsCRC << 20) | // SSE4.2
|
||||
(0 << 21) | // X2APIC
|
||||
(1 << 22) | // MOVBE
|
||||
(1 << 23) | // POPCNT
|
||||
(0 << 24) | // APIC TSC-Deadline
|
||||
(CTX->HostFeatures.SupportsAES << 25) | // AES
|
||||
(0 << 26) | // XSAVE
|
||||
(0 << 27) | // OSXSAVE
|
||||
(SUPPORTS_AVX << 28) | // AVX
|
||||
(SupportsAVX() << 26) | // XSAVE
|
||||
(SupportsAVX() << 27) | // OSXSAVE
|
||||
(SupportsAVX() << 28) | // AVX
|
||||
(0 << 29) | // F16C
|
||||
(CTX->HostFeatures.SupportsRAND << 30) | // RDRAND
|
||||
(Hypervisor << 31);
|
||||
@@ -630,12 +624,12 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) {
|
||||
(1 << 0) | // FS/GS support
|
||||
(0 << 1) | // TSC adjust MSR
|
||||
(0 << 2) | // SGX
|
||||
(1 << 3) | // BMI1
|
||||
(SupportsAVX() << 3) | // BMI1
|
||||
(0 << 4) | // Intel Hardware Lock Elison
|
||||
(0 << 5) | // AVX2 support
|
||||
(1 << 6) | // FPU data pointer updated only on exception
|
||||
(1 << 7) | // SMEP support
|
||||
(1 << 8) | // BMI2
|
||||
(SupportsAVX() << 8) | // BMI2
|
||||
(0 << 9) | // Enhanced REP MOVSB/STOSB
|
||||
(1 << 10) | // INVPCID for system software control of process-context
|
||||
(0 << 11) | // Restricted transactional memory
|
||||
@@ -736,13 +730,13 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0Dh(uint32_t Leaf) {
|
||||
// Leaf 0
|
||||
FEXCore::CPUID::FunctionResults Res{};
|
||||
|
||||
uint32_t XFeatureSupportedSizeMax = SUPPORTS_AVX ? 0x0000'0340 : 0x0000'0240; // XFeatureEnabledSizeMax: Legacy Header + FPU/SSE + AVX
|
||||
uint32_t XFeatureSupportedSizeMax = SupportsAVX() ? 0x0000'0340 : 0x0000'0240; // XFeatureEnabledSizeMax: Legacy Header + FPU/SSE + AVX
|
||||
if (Leaf == 0) {
|
||||
// XFeatureSupportedMask[31:0]
|
||||
Res.eax =
|
||||
(1 << 0) | // X87 support
|
||||
(1 << 1) | // 128-bit SSE support
|
||||
(SUPPORTS_AVX << 2) | // 256-bit AVX support
|
||||
(SupportsAVX() << 2) | // 256-bit AVX support
|
||||
(0b00 << 3) | // MPX State
|
||||
(0b000 << 5) | // AVX-512 state
|
||||
(0 << 8) | // "Used for IA32_XSS" ... Used for what?
|
||||
@@ -776,8 +770,8 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_0Dh(uint32_t Leaf) {
|
||||
Res.edx = 0;
|
||||
}
|
||||
else if (Leaf == 2) {
|
||||
Res.eax = SUPPORTS_AVX ? 0x0000'0100 : 0; // YmmSaveStateSize
|
||||
Res.ebx = SUPPORTS_AVX ? 0x0000'0240 : 0; // YmmSaveStateOffset
|
||||
Res.eax = SupportsAVX() ? 0x0000'0100 : 0; // YmmSaveStateSize
|
||||
Res.ebx = SupportsAVX() ? 0x0000'0240 : 0; // YmmSaveStateOffset
|
||||
|
||||
// Reserved
|
||||
Res.ecx = 0;
|
||||
@@ -1212,11 +1206,26 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_Reserved(uint32_t Leaf) {
|
||||
return Res;
|
||||
}
|
||||
|
||||
FEXCore::CPUID::XCRResults CPUIDEmu::XCRFunction_0h() {
|
||||
// This just returns XCR0
|
||||
FEXCore::CPUID::XCRResults Res{
|
||||
.eax = static_cast<uint32_t>(XCR0),
|
||||
.edx = static_cast<uint32_t>(XCR0 >> 32),
|
||||
};
|
||||
|
||||
return Res;
|
||||
}
|
||||
|
||||
void CPUIDEmu::Init(FEXCore::Context::ContextImpl *ctx) {
|
||||
CTX = ctx;
|
||||
|
||||
// Setup some state tracking
|
||||
SetupHostHybridFlag();
|
||||
|
||||
// TODO: Enable once AVX is supported.
|
||||
if (false && CTX->HostFeatures.SupportsAVX) {
|
||||
XCR0 |= XCR0_AVX;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+41
@@ -63,13 +63,52 @@ public:
|
||||
return Function_8000_0004h(Leaf, CPU % PerCPUData.size());
|
||||
}
|
||||
|
||||
FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) {
|
||||
if (Function >= 1) {
|
||||
// XCR function 1 is not yet supported.
|
||||
return {};
|
||||
}
|
||||
|
||||
return XCRFunction_0h();
|
||||
}
|
||||
|
||||
private:
|
||||
FEXCore::Context::ContextImpl *CTX;
|
||||
bool Hybrid{};
|
||||
FEX_CONFIG_OPT(Cores, THREADS);
|
||||
FEX_CONFIG_OPT(HideHypervisorBit, HIDEHYPERVISORBIT);
|
||||
|
||||
// XFEATURE_ENABLED_MASK
|
||||
// Mask that configures what features are enabled on the CPU.
|
||||
// Affects XSAVE and XRSTOR when modified.
|
||||
// Bit layout is as follows.
|
||||
// [0] - x87 enabled
|
||||
// [1] - SSE enabled
|
||||
// [2] - YMM enabled (256-bit SSE)
|
||||
// [8:3] - Reserved. MBZ.
|
||||
// [9] - MPK
|
||||
// [10] - Reserved. MBZ.
|
||||
// [11] - CET_U
|
||||
// [12] - CET_S
|
||||
// [61:13] - Reserved. MBZ.
|
||||
// [62] - LWP (Lightweight profiling)
|
||||
// [63] - Reserved for XCR bit vector expansion. MBZ.
|
||||
// Always enable x87 and SSE by default.
|
||||
constexpr static uint64_t XCR0_X87 = 1ULL << 0;
|
||||
constexpr static uint64_t XCR0_SSE = 1ULL << 1;
|
||||
constexpr static uint64_t XCR0_AVX = 1ULL << 2;
|
||||
|
||||
uint64_t XCR0 {
|
||||
XCR0_X87 |
|
||||
XCR0_SSE
|
||||
};
|
||||
|
||||
uint32_t SupportsAVX() const {
|
||||
return (XCR0 & XCR0_AVX) ? 1 : 0;
|
||||
}
|
||||
|
||||
using FunctionHandler = FEXCore::CPUID::FunctionResults (CPUIDEmu::*)(uint32_t Leaf);
|
||||
|
||||
struct CPUData {
|
||||
const char *ProductName{};
|
||||
#ifdef _M_ARM_64
|
||||
@@ -109,6 +148,8 @@ private:
|
||||
FEXCore::CPUID::FunctionResults Function_8000_001Dh(uint32_t Leaf);
|
||||
FEXCore::CPUID::FunctionResults Function_Reserved(uint32_t Leaf);
|
||||
|
||||
FEXCore::CPUID::XCRResults XCRFunction_0h();
|
||||
|
||||
void SetupHostHybridFlag();
|
||||
static constexpr std::array<FunctionHandler, 27> Primary = {
|
||||
// 0: Highest function parameter and ID
|
||||
|
||||
+100
-123
@@ -8,9 +8,9 @@ $end_info$
|
||||
*/
|
||||
|
||||
#include <cstdint>
|
||||
#include "FEXCore/Utils/DeferredSignalMutex.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/LookupCache.h"
|
||||
#include "Interface/Core/Core.h"
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/Frontend.h"
|
||||
#include "Interface/Core/GdbServer.h"
|
||||
@@ -24,6 +24,8 @@ $end_info$
|
||||
#include "Interface/IR/Passes/RegisterAllocationPass.h"
|
||||
#include "Interface/IR/Passes.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
#include "Utils/Allocator.h"
|
||||
#include "Utils/Allocator/HostAllocator.h"
|
||||
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Core/CodeLoader.h>
|
||||
@@ -42,6 +44,7 @@ $end_info$
|
||||
#include <FEXCore/IR/RegisterAllocationData.h>
|
||||
#include <FEXCore/Utils/Allocator.h>
|
||||
#include <FEXCore/Utils/Event.h>
|
||||
#include <FEXCore/Utils/File.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Threads.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
@@ -73,22 +76,11 @@ $end_info$
|
||||
#include <utility>
|
||||
#include <xxhash.h>
|
||||
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
bool CreateCPUCore(Context::ContextImpl *CTX) {
|
||||
// This should be used for generating things that are shared between threads
|
||||
CTX->CPUID.Init(CTX);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct ThreadLocalData {
|
||||
FEXCore::Core::InternalThreadState* Thread;
|
||||
};
|
||||
|
||||
thread_local ThreadLocalData ThreadData{};
|
||||
|
||||
constexpr std::array<std::string_view const, 22> FlagNames = {
|
||||
"CF",
|
||||
"",
|
||||
@@ -162,6 +154,9 @@ namespace FEXCore::Context {
|
||||
// Only initialize symbols file if enabled. Ensures we don't pollute /tmp with empty files.
|
||||
Symbols.InitFile();
|
||||
}
|
||||
|
||||
// Track atomic TSO emulation configuration.
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
}
|
||||
|
||||
ContextImpl::~ContextImpl() {
|
||||
@@ -187,43 +182,38 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
static FEXCore::Core::CPUState CreateDefaultCPUState() {
|
||||
FEXCore::Core::CPUState NewThreadState{};
|
||||
|
||||
// Initialize default CPU state
|
||||
NewThreadState.rip = ~0ULL;
|
||||
for (auto& greg : NewThreadState.gregs) {
|
||||
greg = 0;
|
||||
}
|
||||
|
||||
for (auto& xmm : NewThreadState.xmm.avx.data) {
|
||||
xmm[0] = 0xDEADBEEFULL;
|
||||
xmm[1] = 0xBAD0DAD1ULL;
|
||||
xmm[2] = 0xDEADCAFEULL;
|
||||
xmm[3] = 0xBAD2CAD3ULL;
|
||||
}
|
||||
memset(NewThreadState.flags, 0, Core::CPUState::NUM_EFLAG_BITS);
|
||||
NewThreadState.flags[1] = 1;
|
||||
NewThreadState.flags[9] = 1;
|
||||
NewThreadState.FCW = 0x37F;
|
||||
NewThreadState.FTW = 0xFFFF;
|
||||
return NewThreadState;
|
||||
}
|
||||
|
||||
uint64_t ContextImpl::RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState *Thread, uint64_t HostPC) {
|
||||
const auto Frame = Thread->CurrentFrame;
|
||||
const uint64_t BlockBegin = Frame->State.InlineJITBlockHeader;
|
||||
const CPU::CPUBackend::JITCodeHeader *InlineHeader = reinterpret_cast<const CPU::CPUBackend::JITCodeHeader *>(BlockBegin);
|
||||
auto InlineHeader = reinterpret_cast<const CPU::CPUBackend::JITCodeHeader *>(BlockBegin);
|
||||
|
||||
if (InlineHeader) {
|
||||
const CPU::CPUBackend::JITCodeTail *InlineTail = reinterpret_cast<const CPU::CPUBackend::JITCodeTail *>(Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail);
|
||||
auto InlineTail = reinterpret_cast<const CPU::CPUBackend::JITCodeTail *>(Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail);
|
||||
auto RIPEntries = reinterpret_cast<const CPU::CPUBackend::JITRIPReconstructEntries *>(Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail + InlineTail->OffsetToRIPEntries);
|
||||
|
||||
// Check if the host PC is currently within a code block.
|
||||
// If it is then RIP can be reconstructed from the beginning of the code block.
|
||||
// This is currently as close as FEX can get RIP reconstructions.
|
||||
if (HostPC >= reinterpret_cast<uint64_t>(BlockBegin) &&
|
||||
HostPC < reinterpret_cast<uint64_t>(BlockBegin + InlineTail->Size)) {
|
||||
return InlineTail->RIP;
|
||||
|
||||
// Reconstruct RIP from JIT entries for this block.
|
||||
uint64_t StartingHostPC = BlockBegin;
|
||||
uint64_t StartingGuestRIP = InlineTail->RIP;
|
||||
|
||||
for (uint32_t i = 0; i < InlineTail->NumberOfRIPEntries; ++i) {
|
||||
const auto &RIPEntry = RIPEntries[i];
|
||||
if (HostPC >= (StartingHostPC + RIPEntry.HostPCOffset)) {
|
||||
// We are beyond this entry, keep going forward.
|
||||
StartingHostPC += RIPEntry.HostPCOffset;
|
||||
StartingGuestRIP += RIPEntry.GuestRIPOffset;
|
||||
}
|
||||
else {
|
||||
// Passed where the Host PC is at. Break now.
|
||||
break;
|
||||
}
|
||||
}
|
||||
return StartingGuestRIP;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -304,12 +294,13 @@ namespace FEXCore::Context {
|
||||
StopGdbServer();
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
ThunkHandler = FEXCore::ThunkHandler::Create();
|
||||
#endif
|
||||
|
||||
using namespace FEXCore::Core;
|
||||
|
||||
FEXCore::Core::CPUState NewThreadState = CreateDefaultCPUState();
|
||||
FEXCore::Core::InternalThreadState *Thread = CreateThread(&NewThreadState, 0);
|
||||
FEXCore::Core::InternalThreadState *Thread = CreateThread(nullptr, 0);
|
||||
|
||||
// We are the parent thread
|
||||
ParentThread = Thread;
|
||||
@@ -552,7 +543,9 @@ namespace FEXCore::Context {
|
||||
Thread->ThreadManager.TID = FHU::Syscalls::gettid();
|
||||
Thread->ThreadManager.PID = ::getpid();
|
||||
SignalDelegation->RegisterTLSState(Thread);
|
||||
ThunkHandler->RegisterTLSState(Thread);
|
||||
if (ThunkHandler) {
|
||||
ThunkHandler->RegisterTLSState(Thread);
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::RunThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
@@ -616,7 +609,9 @@ namespace FEXCore::Context {
|
||||
FEXCore::Core::InternalThreadState *Thread = new FEXCore::Core::InternalThreadState{};
|
||||
|
||||
// Copy over the new thread state to the new object
|
||||
memcpy(Thread->CurrentFrame, NewThreadState, sizeof(FEXCore::Core::CPUState));
|
||||
if (NewThreadState) {
|
||||
memcpy(Thread->CurrentFrame, NewThreadState, sizeof(FEXCore::Core::CPUState));
|
||||
}
|
||||
Thread->CurrentFrame->Thread = Thread;
|
||||
|
||||
// Set up the thread manager state
|
||||
@@ -625,6 +620,9 @@ namespace FEXCore::Context {
|
||||
InitializeCompiler(Thread);
|
||||
InitializeThreadData(Thread);
|
||||
|
||||
Thread->CurrentFrame->State.DeferredSignalRefCount.Store(0);
|
||||
Thread->CurrentFrame->State.DeferredSignalFaultAddress = reinterpret_cast<Core::NonAtomicRefCounter<uint64_t>*>(FEXCore::Allocator::VirtualAlloc(4096));
|
||||
|
||||
// Insert after the Thread object has been fully initialized
|
||||
{
|
||||
std::lock_guard lk(ThreadCreationMutex);
|
||||
@@ -650,10 +648,23 @@ namespace FEXCore::Context {
|
||||
// To be able to delete a thread from itself, we need to detached the std::thread object
|
||||
Thread->ExecutionThread->detach();
|
||||
}
|
||||
|
||||
FEXCore::Allocator::VirtualFree(reinterpret_cast<void*>(Thread->CurrentFrame->State.DeferredSignalFaultAddress), 4096);
|
||||
delete Thread;
|
||||
}
|
||||
|
||||
void ContextImpl::CleanupAfterFork(FEXCore::Core::InternalThreadState *LiveThread) {
|
||||
#ifndef _WIN32
|
||||
void ContextImpl::UnlockAfterFork(FEXCore::Core::InternalThreadState *LiveThread, bool Child) {
|
||||
Allocator::UnlockAfterFork(LiveThread, Child);
|
||||
|
||||
if (Child) {
|
||||
CodeInvalidationMutex.StealAndDropActiveLocks();
|
||||
}
|
||||
else {
|
||||
CodeInvalidationMutex.unlock();
|
||||
return;
|
||||
}
|
||||
|
||||
// This function is called after fork
|
||||
// We need to cleanup some of the thread data that is dead
|
||||
for (auto &DeadThread : Threads) {
|
||||
@@ -692,6 +703,12 @@ namespace FEXCore::Context {
|
||||
FEXCore::Threads::Thread::CleanupAfterFork();
|
||||
}
|
||||
|
||||
void ContextImpl::LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) {
|
||||
CodeInvalidationMutex.lock();
|
||||
Allocator::LockBeforeFork(Thread);
|
||||
}
|
||||
#endif
|
||||
|
||||
void ContextImpl::AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr) {
|
||||
Thread->LookupCache->AddBlockMapping(Address, Ptr);
|
||||
}
|
||||
@@ -712,36 +729,30 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
static void IRDumper(FEXCore::Core::InternalThreadState *Thread, IR::IREmitter *IREmitter, uint64_t GuestRIP, IR::RegisterAllocationData* RA) {
|
||||
#ifndef _WIN32
|
||||
int FD {-1};
|
||||
bool CloseAfter = false;
|
||||
FEXCore::File::File FD;
|
||||
const auto DumpIRStr = static_cast<ContextImpl*>(Thread->CTX)->Config.DumpIR();
|
||||
|
||||
// DumpIRStr might be no if not dumping but ShouldDump is set in OpDisp
|
||||
if (DumpIRStr =="stderr" || DumpIRStr =="no") {
|
||||
FD = STDERR_FILENO;
|
||||
FD = FEXCore::File::File::GetStdERR();
|
||||
}
|
||||
else if (DumpIRStr =="stdout") {
|
||||
FD = STDOUT_FILENO;
|
||||
FD = FEXCore::File::File::GetStdOUT();
|
||||
}
|
||||
else {
|
||||
const auto fileName = fextl::fmt::format("{}/{:x}{}", DumpIRStr, GuestRIP, RA ? "-post.ir" : "-pre.ir");
|
||||
constexpr int USER_PERMS = S_IRWXU | S_IRWXG | S_IRWXO;
|
||||
FD = open(fileName.c_str(), O_CREAT | O_WRONLY | O_TRUNC | O_CLOEXEC, USER_PERMS, USER_PERMS);
|
||||
CloseAfter = true;
|
||||
FD = FEXCore::File::File(fileName.c_str(),
|
||||
FEXCore::File::FileModes::WRITE |
|
||||
FEXCore::File::FileModes::CREATE |
|
||||
FEXCore::File::FileModes::TRUNCATE);
|
||||
}
|
||||
|
||||
if (FD != -1) {
|
||||
if (FD.IsValid()) {
|
||||
fextl::stringstream out;
|
||||
auto NewIR = IREmitter->ViewIR();
|
||||
FEXCore::IR::Dump(&out, &NewIR, RA);
|
||||
fextl::fmt::print(FD, "IR-{} 0x{:x}:\n{}\n@@@@@\n", RA ? "post" : "pre", GuestRIP, out.str());
|
||||
|
||||
if (CloseAfter) {
|
||||
close(FD);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
};
|
||||
|
||||
static void ValidateIR(ContextImpl *ctx, IR::IREmitter *IREmitter) {
|
||||
@@ -795,7 +806,7 @@ namespace FEXCore::Context {
|
||||
|
||||
Thread->FrontendDecoder->DecodeInstructionsAtEntry(GuestCode, GuestRIP, [Thread](uint64_t BlockEntry, uint64_t Start, uint64_t Length) {
|
||||
if (Thread->LookupCache->AddBlockExecutableRange(BlockEntry, Start, Length)) {
|
||||
static_cast<ContextImpl*>(Thread->CTX)->SyscallHandler->MarkGuestExecutableRange(Start, Length);
|
||||
static_cast<ContextImpl*>(Thread->CTX)->SyscallHandler->MarkGuestExecutableRange(Thread, Start, Length);
|
||||
}
|
||||
});
|
||||
|
||||
@@ -825,7 +836,7 @@ namespace FEXCore::Context {
|
||||
DecodedInfo = &Block.DecodedInstructions[i];
|
||||
bool IsLocked = DecodedInfo->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK;
|
||||
|
||||
if (ExtendedDebugInfo) {
|
||||
if (ExtendedDebugInfo || Thread->OpDispatcher->CanHaveSideEffects(TableInfo, DecodedInfo)) {
|
||||
Thread->OpDispatcher->_GuestOpcode(Block.Entry + BlockInstructionsLength - GuestRIP);
|
||||
}
|
||||
|
||||
@@ -852,14 +863,14 @@ namespace FEXCore::Context {
|
||||
|
||||
if (TableInfo && TableInfo->OpcodeDispatcher) {
|
||||
auto Fn = TableInfo->OpcodeDispatcher;
|
||||
Thread->OpDispatcher->HandledLock = false;
|
||||
Thread->OpDispatcher->ResetHandledLock();
|
||||
Thread->OpDispatcher->ResetDecodeFailure();
|
||||
std::invoke(Fn, Thread->OpDispatcher, DecodedInfo);
|
||||
if (Thread->OpDispatcher->HadDecodeFailure()) {
|
||||
HadDispatchError = true;
|
||||
}
|
||||
else {
|
||||
if (Thread->OpDispatcher->HandledLock != IsLocked) {
|
||||
if (Thread->OpDispatcher->HasHandledLock() != IsLocked) {
|
||||
HadDispatchError = true;
|
||||
LogMan::Msg::EFmt("Missing LOCK HANDLER at 0x{:x}{{'{}'}}", Block.Entry + BlockInstructionsLength, TableInfo->Name ?: "UND");
|
||||
}
|
||||
@@ -909,7 +920,7 @@ namespace FEXCore::Context {
|
||||
|
||||
IR::IREmitter *IREmitter = Thread->OpDispatcher.get();
|
||||
|
||||
auto ShouldDump = static_cast<ContextImpl*>(Thread->CTX)->Config.DumpIR() != "no" || Thread->OpDispatcher->ShouldDump;
|
||||
auto ShouldDump = static_cast<ContextImpl*>(Thread->CTX)->Config.DumpIR() != "no" || Thread->OpDispatcher->ShouldDumpIR();
|
||||
// Debug
|
||||
{
|
||||
if (ShouldDump) {
|
||||
@@ -974,7 +985,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
if (SourcecodeResolver && Config.GDBSymbols()) {
|
||||
auto AOTIRCacheEntry = SyscallHandler->LookupAOTIRCacheEntry(GuestRIP);
|
||||
auto AOTIRCacheEntry = SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
|
||||
if (AOTIRCacheEntry.Entry && !AOTIRCacheEntry.Entry->ContainsCode) {
|
||||
AOTIRCacheEntry.Entry->SourcecodeMap =
|
||||
SourcecodeResolver->GenerateMap(AOTIRCacheEntry.Entry->Filename, AOTIRCacheEntry.Entry->FileId);
|
||||
@@ -983,7 +994,7 @@ namespace FEXCore::Context {
|
||||
|
||||
// AOT IR bookkeeping and cache
|
||||
{
|
||||
auto [IRCopy, RACopy, DebugDataCopy, _StartAddr, _Length, _GeneratedIR] = IRCaptureCache.PreGenerateIRFetch(GuestRIP, IRList);
|
||||
auto [IRCopy, RACopy, DebugDataCopy, _StartAddr, _Length, _GeneratedIR] = IRCaptureCache.PreGenerateIRFetch(Thread, GuestRIP, IRList);
|
||||
if (_GeneratedIR) {
|
||||
// Setup pointers to internal structures
|
||||
IRList = IRCopy;
|
||||
@@ -1006,9 +1017,6 @@ namespace FEXCore::Context {
|
||||
StartAddr = _StartAddr;
|
||||
Length = _Length;
|
||||
|
||||
// Increment stats
|
||||
Thread->Stats.BlocksCompiled.fetch_add(1);
|
||||
|
||||
// These blocks aren't already in the cache
|
||||
GeneratedIR = true;
|
||||
}
|
||||
@@ -1047,7 +1055,7 @@ namespace FEXCore::Context {
|
||||
auto Thread = Frame->Thread;
|
||||
|
||||
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
|
||||
std::shared_lock lk(CodeInvalidationMutex);
|
||||
ScopedDeferredSignalWithForkableSharedLock lk(CodeInvalidationMutex, Thread);
|
||||
|
||||
// Is the code in the cache?
|
||||
// The backends only check L1 and L2, not L3
|
||||
@@ -1079,7 +1087,7 @@ namespace FEXCore::Context {
|
||||
auto FragmentBasePtr = reinterpret_cast<uint8_t *>(CodePtr);
|
||||
|
||||
if (DebugData) {
|
||||
auto GuestRIPLookup = SyscallHandler->LookupAOTIRCacheEntry(GuestRIP);
|
||||
auto GuestRIPLookup = SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
|
||||
|
||||
if (DebugData->Subblocks.size()) {
|
||||
for (auto& Subblock: DebugData->Subblocks) {
|
||||
@@ -1143,10 +1151,12 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
void ContextImpl::ExecutionThread(FEXCore::Core::InternalThreadState *Thread) {
|
||||
Core::ThreadData.Thread = Thread;
|
||||
Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_WAITING;
|
||||
|
||||
InitializeThreadTLSData(Thread);
|
||||
#ifndef _WIN32
|
||||
Alloc::OSAllocator::RegisterTLSData(Thread);
|
||||
#endif
|
||||
|
||||
++IdleWaitRefCount;
|
||||
|
||||
@@ -1191,6 +1201,9 @@ namespace FEXCore::Context {
|
||||
--IdleWaitRefCount;
|
||||
IdleWaitCV.notify_all();
|
||||
|
||||
#ifndef _WIN32
|
||||
Alloc::OSAllocator::UninstallTLSData(Thread);
|
||||
#endif
|
||||
SignalDelegation->UninstallTLSState(Thread);
|
||||
|
||||
// If the parent thread is waiting to join, then we can't destroy our thread object
|
||||
@@ -1221,14 +1234,20 @@ namespace FEXCore::Context {
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::InvalidateGuestCodeRange(uint64_t Start, uint64_t Length) {
|
||||
FHU::ScopedSignalMaskWithUniqueLock CodeInvalidationLock(CodeInvalidationMutex);
|
||||
void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) {
|
||||
// Potential deferred since Thread might not be valid.
|
||||
// Thread object isn't valid very early in frontend's initialization.
|
||||
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
|
||||
ScopedPotentialDeferredSignalWithForkableUniqueLock lk(CodeInvalidationMutex, Thread);
|
||||
|
||||
InvalidateGuestCodeRangeInternal(this, Start, Length);
|
||||
}
|
||||
|
||||
void ContextImpl::InvalidateGuestCodeRange(uint64_t Start, uint64_t Length, std::function<void(uint64_t start, uint64_t Length)> CallAfter) {
|
||||
FHU::ScopedSignalMaskWithUniqueLock CodeInvalidationLock(CodeInvalidationMutex);
|
||||
void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, CodeRangeInvalidationFn CallAfter) {
|
||||
// Potential deferred since Thread might not be valid.
|
||||
// Thread object isn't valid very early in frontend's initialization.
|
||||
// To be more optimal the frontend should provide this code with a valid Thread object earlier.
|
||||
ScopedPotentialDeferredSignalWithForkableUniqueLock lk(CodeInvalidationMutex, Thread);
|
||||
|
||||
InvalidateGuestCodeRangeInternal(this, Start, Length);
|
||||
CallAfter(Start, Length);
|
||||
@@ -1237,6 +1256,7 @@ namespace FEXCore::Context {
|
||||
void ContextImpl::MarkMemoryShared() {
|
||||
if (!IsMemoryShared) {
|
||||
IsMemoryShared = true;
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
|
||||
if (Config.TSOAutoMigration) {
|
||||
std::lock_guard<std::mutex> lkThreads(ThreadCreationMutex);
|
||||
@@ -1255,7 +1275,7 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
void ContextImpl::ThreadAddBlockLink(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestDestination, uintptr_t HostLink, const std::function<void()> &delinker) {
|
||||
std::shared_lock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex);
|
||||
ScopedDeferredSignalWithForkableSharedLock lk(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
Thread->LookupCache->AddBlockLink(GuestDestination, HostLink, delinker);
|
||||
}
|
||||
@@ -1269,7 +1289,7 @@ namespace FEXCore::Context {
|
||||
Thread->LookupCache->Erase(GuestRIP);
|
||||
}
|
||||
|
||||
CustomIRResult ContextImpl::AddCustomIREntrypoint(uintptr_t Entrypoint, std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)> Handler, void *Creator, void *Data) {
|
||||
CustomIRResult ContextImpl::AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void *Creator, void *Data) {
|
||||
LOGMAN_THROW_A_FMT(Config.Is64BitMode || !(Entrypoint >> 32), "64-bit Entrypoint in 32-bit mode {:x}", Entrypoint);
|
||||
|
||||
std::unique_lock lk(CustomIRMutex);
|
||||
@@ -1290,56 +1310,11 @@ namespace FEXCore::Context {
|
||||
|
||||
std::scoped_lock lk(CustomIRMutex);
|
||||
|
||||
InvalidateGuestCodeRange(Entrypoint, 1, [this](uint64_t Entrypoint, uint64_t) {
|
||||
InvalidateGuestCodeRange(nullptr, Entrypoint, 1, [this](uint64_t Entrypoint, uint64_t) {
|
||||
CustomIRHandlers.erase(Entrypoint);
|
||||
});
|
||||
}
|
||||
|
||||
// Debug interface
|
||||
void Context::CompileRIP(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) {
|
||||
uint64_t RIPBackup = Thread->CurrentFrame->State.rip;
|
||||
Thread->CurrentFrame->State.rip = RIP;
|
||||
|
||||
auto CTX = static_cast<ContextImpl*>(Thread->CTX);
|
||||
|
||||
// Erase the RIP from all the storage backings if it exists
|
||||
CTX->ThreadRemoveCodeEntry(Thread, RIP);
|
||||
|
||||
// We don't care if compilation passes or not
|
||||
CTX->CompileBlock(Thread->CurrentFrame, RIP);
|
||||
|
||||
Thread->CurrentFrame->State.rip = RIPBackup;
|
||||
}
|
||||
|
||||
uint64_t ContextImpl::GetThreadCount() const {
|
||||
return Threads.size();
|
||||
}
|
||||
|
||||
FEXCore::Core::RuntimeStats *ContextImpl::GetRuntimeStatsForThread(uint64_t Thread) {
|
||||
return &Threads[Thread]->Stats;
|
||||
}
|
||||
|
||||
bool ContextImpl::GetDebugDataForRIP(uint64_t RIP, FEXCore::Core::DebugData *Data) {
|
||||
std::lock_guard<std::recursive_mutex> lk(ParentThread->LookupCache->WriteLock);
|
||||
auto it = ParentThread->DebugStore.find(RIP);
|
||||
if (it == ParentThread->DebugStore.end()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
memcpy(Data, it->second.DebugData.get(), sizeof(FEXCore::Core::DebugData));
|
||||
return true;
|
||||
}
|
||||
|
||||
bool ContextImpl::FindHostCodeForRIP(uint64_t RIP, uint8_t **Code) {
|
||||
uintptr_t HostCode = ParentThread->LookupCache->FindBlock(RIP);
|
||||
if (!HostCode) {
|
||||
return false;
|
||||
}
|
||||
|
||||
*Code = reinterpret_cast<uint8_t*>(HostCode);
|
||||
return true;
|
||||
}
|
||||
|
||||
uint64_t HandleSyscall(FEXCore::HLE::SyscallHandler *Handler, FEXCore::Core::CpuStateFrame *Frame, FEXCore::HLE::SyscallArguments *Args) {
|
||||
uint64_t Result{};
|
||||
Result = Handler->HandleSyscall(Frame, Args);
|
||||
@@ -1362,7 +1337,9 @@ namespace FEXCore::Context {
|
||||
}
|
||||
|
||||
void ContextImpl::AppendThunkDefinitions(fextl::vector<FEXCore::IR::ThunkDefinition> const& Definitions) {
|
||||
ThunkHandler->AppendThunkDefinitions(Definitions);
|
||||
if (ThunkHandler) {
|
||||
ThunkHandler->AppendThunkDefinitions(Definitions);
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, fextl::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) {
|
||||
|
||||
-23
@@ -1,23 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
namespace FEXCore {
|
||||
class CodeLoader;
|
||||
}
|
||||
|
||||
namespace FEXCore::Context {
|
||||
class ContextImpl;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
/**
|
||||
* @brief Create the CPU core backend for the context passed in
|
||||
*
|
||||
* @param CTX
|
||||
*
|
||||
* @return true if core was able to be create
|
||||
*/
|
||||
bool CreateCPUCore(FEXCore::Context::ContextImpl *CTX);
|
||||
|
||||
bool LoadCode(FEXCore::Context::ContextImpl *CTX, FEXCore::CodeLoader *Loader);
|
||||
}
|
||||
@@ -129,13 +129,6 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg.R(), ARMEmitter::Reg::r3);
|
||||
}
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// VIXL simulator can't run syscalls.
|
||||
constexpr bool SignalSafeCompile = false;
|
||||
#else
|
||||
constexpr bool SignalSafeCompile = true;
|
||||
#endif
|
||||
|
||||
ARMEmitter::ForwardLabel NoBlock;
|
||||
|
||||
{
|
||||
@@ -183,7 +176,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
{
|
||||
ThreadStopHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
ThreadStopHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
@@ -197,28 +190,11 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
{
|
||||
ExitFunctionLinkerAddress = GetCursorAddress<uint64_t>();
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
#ifndef _WIN32
|
||||
if (SignalSafeCompile) {
|
||||
// When compiling code, mask all signals to reduce the chance of reentrant allocations
|
||||
// Args:
|
||||
// X0: SETMASK
|
||||
// X1: Pointer to mask value (uint64_t)
|
||||
// X2: Pointer to old mask value (uint64_t)
|
||||
// X3: Size of mask, sizeof(uint64_t)
|
||||
// X8: Syscall
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ~0ULL);
|
||||
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, ARMEmitter::Reg::rsp, -16);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SIG_SETMASK);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::rsp, 0);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::rsp, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, 8);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, SYS_rt_sigprocmask);
|
||||
svc(0);
|
||||
}
|
||||
#endif
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
|
||||
mov(ARMEmitter::XReg::x0, STATE);
|
||||
mov(ARMEmitter::XReg::x1, ARMEmitter::XReg::lr);
|
||||
@@ -230,28 +206,17 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
#endif
|
||||
|
||||
#ifndef _WIN32
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
|
||||
mov(ARMEmitter::XReg::x4, ARMEmitter::XReg::x0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SIG_SETMASK);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::rsp, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, 8);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, SYS_rt_sigprocmask);
|
||||
svc(0);
|
||||
|
||||
// Bring stack back
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 16);
|
||||
mov(ARMEmitter::XReg::x0, ARMEmitter::XReg::x4);
|
||||
}
|
||||
#endif
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
FillStaticRegs();
|
||||
|
||||
ldr(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
subs(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x1, ARMEmitter::XReg::x1, 1);
|
||||
str(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
|
||||
// Trigger segfault if any deferred signals are pending
|
||||
ldr(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress));
|
||||
str(ARMEmitter::XReg::zr, ARMEmitter::XReg::x1, 0);
|
||||
|
||||
br(ARMEmitter::Reg::r0);
|
||||
}
|
||||
|
||||
@@ -260,31 +225,11 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
Bind(&NoBlock);
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
#ifndef _WIN32
|
||||
if (SignalSafeCompile) {
|
||||
// When compiling code, mask all signals to reduce the chance of reentrant allocations
|
||||
// Args:
|
||||
// X0: SETMASK
|
||||
// X1: Pointer to mask value (uint64_t)
|
||||
// X2: Pointer to old mask value (uint64_t)
|
||||
// X3: Size of mask, sizeof(uint64_t)
|
||||
// X8: Syscall
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ~0ULL);
|
||||
stp<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::x0, ARMEmitter::XReg::x2, ARMEmitter::Reg::rsp, -16);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SIG_SETMASK);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::rsp, 0);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::rsp, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, 8);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, SYS_rt_sigprocmask);
|
||||
svc(0);
|
||||
|
||||
// Reload x2 to bring back RIP
|
||||
ldr(ARMEmitter::XReg::x2, ARMEmitter::Reg::rsp, 8);
|
||||
}
|
||||
#endif
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, &l_CTX);
|
||||
mov(ARMEmitter::XReg::x1, STATE);
|
||||
@@ -297,25 +242,17 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
blr(ARMEmitter::Reg::r3); // { CTX, Frame, RIP}
|
||||
#endif
|
||||
|
||||
#ifndef _WIN32
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SIG_SETMASK);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, ARMEmitter::Reg::rsp, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, 0);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, 8);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, SYS_rt_sigprocmask);
|
||||
svc(0);
|
||||
|
||||
// Bring stack back
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, 16);
|
||||
}
|
||||
#endif
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
FillStaticRegs();
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
subs(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
|
||||
// Trigger segfault if any deferred signals are pending
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress));
|
||||
str(ARMEmitter::XReg::zr, TMP1, 0);
|
||||
|
||||
b(&LoopTop);
|
||||
}
|
||||
|
||||
@@ -341,7 +278,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
GuestSignal_SIGILL = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
hlt(0);
|
||||
}
|
||||
@@ -352,7 +289,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
GuestSignal_SIGTRAP = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
brk(0);
|
||||
}
|
||||
@@ -363,20 +300,28 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
GuestSignal_SIGSEGV = GetCursorAddress<uint64_t>();
|
||||
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
// hlt/udf = SIGILL
|
||||
// brk = SIGTRAP
|
||||
// ??? = SIGSEGV
|
||||
// Force a SIGSEGV by loading zero
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, 0);
|
||||
ldr(ARMEmitter::XReg::x1, ARMEmitter::Reg::r1);
|
||||
if (CTX->ExitOnHLTEnabled()) {
|
||||
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, ReturningStackLocation));
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::r0, 0);
|
||||
PopCalleeSavedRegisters();
|
||||
ret();
|
||||
}
|
||||
else {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, 0);
|
||||
ldr(ARMEmitter::XReg::x1, ARMEmitter::Reg::r1);
|
||||
}
|
||||
}
|
||||
|
||||
{
|
||||
ThreadPauseHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
|
||||
if (config.StaticRegisterAllocation)
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
ThreadPauseHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
// We are pausing, this means the frontend should be waiting for this thread to idle
|
||||
@@ -453,7 +398,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
LUDIVHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(ARMEmitter::Reg::r3);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUDIV));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
@@ -475,7 +420,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
LDIVHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(ARMEmitter::Reg::r3);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LDIV));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
@@ -497,7 +442,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
LUREMHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(ARMEmitter::Reg::r3);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUREM));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
@@ -519,7 +464,7 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
LREMHandlerAddress = GetCursorAddress<uint64_t>();
|
||||
|
||||
PushDynamicRegsAndLR(ARMEmitter::Reg::r3);
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(ARMEmitter::Reg::r3);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LREM));
|
||||
|
||||
@@ -558,8 +503,10 @@ void Arm64Dispatcher::EmitDispatcher() {
|
||||
}
|
||||
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
const auto DisasmEnd = GetCursorAddress<const vixl::aarch64::Instruction*>();
|
||||
Disasm.DisassembleBuffer(DisasmBegin, DisasmEnd);
|
||||
if (Disassemble() & FEXCore::Config::Disassemble::DISPATCHER) {
|
||||
const auto DisasmEnd = GetCursorAddress<const vixl::aarch64::Instruction*>();
|
||||
Disasm.DisassembleBuffer(DisasmBegin, DisasmEnd);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
|
||||
@@ -31,22 +31,22 @@ class Arm64Dispatcher final : public Dispatcher, public Arm64Emitter {
|
||||
void EmitDispatcher();
|
||||
|
||||
uint16_t GetSRAGPRCount() const override {
|
||||
return SRA64.size();
|
||||
return StaticRegisters.size();
|
||||
}
|
||||
|
||||
uint16_t GetSRAFPRCount() const override {
|
||||
return SRAFPR.size();
|
||||
return StaticFPRegisters.size();
|
||||
}
|
||||
|
||||
void GetSRAGPRMapping(uint8_t Mapping[16]) const override {
|
||||
for (size_t i = 0; i < SRA64.size(); ++i) {
|
||||
Mapping[i] = SRA64[i].Idx();
|
||||
for (size_t i = 0; i < StaticRegisters.size(); ++i) {
|
||||
Mapping[i] = StaticRegisters[i].Idx();
|
||||
}
|
||||
}
|
||||
|
||||
void GetSRAFPRMapping(uint8_t Mapping[16]) const override {
|
||||
for (size_t i = 0; i < SRAFPR.size(); ++i) {
|
||||
Mapping[i] = SRAFPR[i].Idx();
|
||||
for (size_t i = 0; i < StaticFPRegisters.size(); ++i) {
|
||||
Mapping[i] = StaticFPRegisters[i].Idx();
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -90,6 +90,8 @@ public:
|
||||
virtual void GetSRAFPRMapping(uint8_t Mapping[16]) const {
|
||||
}
|
||||
|
||||
const DispatcherConfig& GetConfig() const { return config; }
|
||||
|
||||
protected:
|
||||
Dispatcher(FEXCore::Context::ContextImpl *ctx, const DispatcherConfig &Config)
|
||||
: CTX {ctx}
|
||||
|
||||
@@ -170,38 +170,11 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::ContextImpl *ctx, const Dispatche
|
||||
ret();
|
||||
}
|
||||
|
||||
constexpr bool SignalSafeCompile = true;
|
||||
// Block creation
|
||||
{
|
||||
L(NoBlock);
|
||||
|
||||
#ifndef _WIN32
|
||||
if (SignalSafeCompile) {
|
||||
// When compiling code, mask all signals to reduce the chance of reentrant allocations
|
||||
// RDI: SETMASK
|
||||
// RSI: Pointer to mask value (uint64_t)
|
||||
// RDX: Pointer to old mask value (uint64_t)
|
||||
// R10: Size of mask, sizeof(uint64_t)
|
||||
// RAX: Syscall
|
||||
|
||||
// Backup rdx
|
||||
mov(r9, rdx);
|
||||
|
||||
mov(rdi, ~0ULL);
|
||||
sub(rsp, 16);
|
||||
mov(qword [rsp], rdi);
|
||||
mov(qword [rsp + 8], rdi);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, rsp);
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
|
||||
mov(rdx, r9);
|
||||
}
|
||||
#endif
|
||||
inc(qword [STATE + offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount)]);
|
||||
|
||||
// {rdi, rsi, rdx}
|
||||
mov(rdi, reinterpret_cast<uint64_t>(CTX));
|
||||
@@ -210,26 +183,15 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::ContextImpl *ctx, const Dispatche
|
||||
|
||||
call(rax);
|
||||
|
||||
#ifndef _WIN32
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
// Backup rdx
|
||||
mov(r9, rdx);
|
||||
dec(qword [STATE + offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount)]);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, 0); // Don't care about result
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
Label AfterStore;
|
||||
// Skip the deferred fault address if the refcount isn't zero
|
||||
jne(AfterStore);
|
||||
mov(rax, qword [STATE + offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress)]);
|
||||
mov(rax, qword [rax]);
|
||||
|
||||
// Bring stack back
|
||||
add(rsp, 16);
|
||||
|
||||
mov(rdx, r9);
|
||||
}
|
||||
#endif
|
||||
L(AfterStore);
|
||||
|
||||
// rdx already contains RIP here
|
||||
jmp(LoopTop);
|
||||
@@ -237,34 +199,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::ContextImpl *ctx, const Dispatche
|
||||
|
||||
{
|
||||
ExitFunctionLinkerAddress = getCurr<uint64_t>();
|
||||
|
||||
#ifndef _WIN32
|
||||
if (SignalSafeCompile) {
|
||||
// When compiling code, mask all signals to reduce the chance of reentrant allocations
|
||||
// RDI: SETMASK
|
||||
// RSI: Pointer to mask value (uint64_t)
|
||||
// RDX: Pointer to old mask value (uint64_t)
|
||||
// R10: Size of mask, sizeof(uint64_t)
|
||||
// RAX: Syscall
|
||||
|
||||
// Backup rax
|
||||
mov(r9, rax);
|
||||
|
||||
mov(rdi, ~0ULL);
|
||||
sub(rsp, 16);
|
||||
mov(qword [rsp], rdi);
|
||||
mov(qword [rsp + 8], rdi);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, rsp);
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
|
||||
mov(rax, r9);
|
||||
}
|
||||
#endif
|
||||
inc(qword [STATE + offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount)]);
|
||||
|
||||
// {rdi, rsi}
|
||||
mov(rdi, STATE);
|
||||
@@ -272,30 +207,17 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::ContextImpl *ctx, const Dispatche
|
||||
|
||||
call(qword STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
|
||||
|
||||
#ifndef _WIN32
|
||||
if (SignalSafeCompile) {
|
||||
// Now restore the signal mask
|
||||
// Living in the same location
|
||||
// Backup rax
|
||||
mov(r9, rax);
|
||||
dec(qword [STATE + offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount)]);
|
||||
|
||||
mov(rdi, SIG_SETMASK);
|
||||
mov(rsi, rsp);
|
||||
mov(rdx, 0); // Don't care about result
|
||||
mov(r10, 8);
|
||||
mov(rax, SYS_rt_sigprocmask);
|
||||
syscall();
|
||||
Label AfterStore;
|
||||
// Skip the deferred fault address if the refcount isn't zero
|
||||
jne(AfterStore);
|
||||
mov(rbx, qword [STATE + offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress)]);
|
||||
mov(qword [rbx], rbx);
|
||||
|
||||
// Bring stack back
|
||||
add(rsp, 16);
|
||||
L(AfterStore);
|
||||
|
||||
jmp(r9);
|
||||
}
|
||||
else
|
||||
#endif
|
||||
{
|
||||
jmp(rax);
|
||||
}
|
||||
jmp(rax);
|
||||
}
|
||||
|
||||
{
|
||||
|
||||
+18
-13
@@ -10,7 +10,6 @@ $end_info$
|
||||
#include "Interface/Core/X86Tables/X86Tables.h"
|
||||
|
||||
#include <array>
|
||||
#include <assert.h>
|
||||
#include <algorithm>
|
||||
#include <cstring>
|
||||
#include <FEXCore/Config/Config.h>
|
||||
@@ -239,8 +238,19 @@ void Decoder::DecodeModRM_64(X86Tables::DecodedOperand *Operand, X86Tables::ModR
|
||||
Operand->Data.SIB.Scale = 1 << SIB.scale;
|
||||
|
||||
// The invalid encoding types are described at Table 1-12. "promoted nsigned is always non-zero"
|
||||
Operand->Data.SIB.Index = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_X ? 1 : 0, SIB.index, false, false, false, false, 0b100);
|
||||
Operand->Data.SIB.Base = MapModRMToReg(DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B ? 1 : 0, SIB.base, false, false, false, false, ModRM.mod == 0 ? 0b101 : 16);
|
||||
{
|
||||
// If we have a VSIB byte (as opposed to SIB), then the index register is a vector.
|
||||
const bool IsIndexVector = (DecodeInst->TableInfo->Flags & InstFlags::FLAGS_VEX_VSIB) != 0;
|
||||
if (IsIndexVector) {
|
||||
DecodeInst->Flags |= X86Tables::DecodeFlags::FLAG_VSIB_BYTE;
|
||||
}
|
||||
|
||||
const uint8_t IndexREX = (DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_X) != 0 ? 1 : 0;
|
||||
const uint8_t BaseREX = (DecodeInst->Flags & DecodeFlags::FLAG_REX_XGPR_B) != 0 ? 1 : 0;
|
||||
|
||||
Operand->Data.SIB.Index = MapModRMToReg(IndexREX, SIB.index, false, false, IsIndexVector, false, 0b100);
|
||||
Operand->Data.SIB.Base = MapModRMToReg(BaseREX, SIB.base, false, false, false, false, ModRM.mod == 0 ? 0b101 : 16);
|
||||
}
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Displacement <= 4, "Number of bytes should be <= 4 for literal src");
|
||||
|
||||
@@ -671,9 +681,6 @@ bool Decoder::NormalOpHeader(FEXCore::X86Tables::X86InstInfo const *Info, uint16
|
||||
DecodeInst->ModRM = ModRMByte;
|
||||
DecodeInst->DecodedModRM = true;
|
||||
|
||||
FEXCore::X86Tables::ModRMDecoded ModRM;
|
||||
ModRM.Hex = DecodeInst->ModRM;
|
||||
|
||||
uint16_t X87Op = ((Op - 0xD8) << 8) | ModRMByte;
|
||||
return NormalOp(&X87Ops[X87Op], X87Op);
|
||||
}
|
||||
@@ -957,7 +964,7 @@ bool Decoder::DecodeInstruction(uint64_t PC) {
|
||||
}
|
||||
|
||||
if (DecodeInst->Dest.IsGPR()) {
|
||||
assert(DecodeInst->Dest.Data.GPR.GPR != FEXCore::X86State::REG_INVALID);
|
||||
LOGMAN_THROW_A_FMT(DecodeInst->Dest.Data.GPR.GPR != FEXCore::X86State::REG_INVALID, "Destination GPR was invalid");
|
||||
}
|
||||
|
||||
return true;
|
||||
@@ -1015,15 +1022,13 @@ void Decoder::BranchTargetInMultiblockRange() {
|
||||
|
||||
// If we are conditional then a target can be the instruction past the conditional instruction
|
||||
uint64_t FallthroughRIP = DecodeInst->PC + DecodeInst->InstSize;
|
||||
if (HasBlocks.find(FallthroughRIP) == HasBlocks.end() &&
|
||||
BlocksToDecode.find(FallthroughRIP) == BlocksToDecode.end()) {
|
||||
BlocksToDecode.emplace(FallthroughRIP);
|
||||
if (!HasBlocks.contains(FallthroughRIP)) {
|
||||
BlocksToDecode.insert(FallthroughRIP);
|
||||
}
|
||||
}
|
||||
|
||||
if (HasBlocks.find(TargetRIP) == HasBlocks.end() &&
|
||||
BlocksToDecode.find(TargetRIP) == BlocksToDecode.end()) {
|
||||
BlocksToDecode.emplace(TargetRIP);
|
||||
if (!HasBlocks.contains(TargetRIP)) {
|
||||
BlocksToDecode.insert(TargetRIP);
|
||||
}
|
||||
} else {
|
||||
if (ExternalBranches) {
|
||||
|
||||
@@ -54,7 +54,12 @@ HostFeatures::HostFeatures() {
|
||||
#ifdef VIXL_SIMULATOR
|
||||
auto Features = vixl::CPUFeatures::All();
|
||||
#else
|
||||
#ifndef _WIN32
|
||||
auto Features = vixl::CPUFeatures::InferFromOS();
|
||||
#else
|
||||
// Need to use ID registers in WINE.
|
||||
auto Features = vixl::CPUFeatures::InferFromIDRegisters();
|
||||
#endif
|
||||
#endif
|
||||
SupportsAES = Features.Has(vixl::CPUFeatures::Feature::kAES);
|
||||
SupportsCRC = Features.Has(vixl::CPUFeatures::Feature::kCRC32);
|
||||
@@ -66,6 +71,7 @@ HostFeatures::HostFeatures() {
|
||||
SupportsRCPC = Features.Has(vixl::CPUFeatures::Feature::kRCpc);
|
||||
SupportsTSOImm9 = Features.Has(vixl::CPUFeatures::Feature::kRCpcImm);
|
||||
SupportsPMULL_128Bit = Features.Has(vixl::CPUFeatures::Feature::kPmull1Q);
|
||||
SupportsCSSC = Features.Has(vixl::CPUFeatures::Feature::kCSSC);
|
||||
|
||||
Supports3DNow = true;
|
||||
SupportsSSE4A = true;
|
||||
|
||||
@@ -113,6 +113,22 @@ DEF_OP(Neg) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Abs) {
|
||||
auto Op = IROp->C<IR::IROp_Abs>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
const int64_t Src = *GetSrc<int64_t*>(Data->SSAData, Op->Src);
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
GD = std::abs(static_cast<int32_t>(Src));
|
||||
break;
|
||||
case 8:
|
||||
GD = std::abs(static_cast<int64_t>(Src));
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unknown Abs Size: {}\n", OpSize); break;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Mul) {
|
||||
auto Op = IROp->C<IR::IROp_Mul>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
@@ -869,12 +885,8 @@ DEF_OP(Select) {
|
||||
auto Op = IROp->C<IR::IROp_Select>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
const uint64_t Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp1);
|
||||
const uint64_t Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp2);
|
||||
|
||||
uint64_t ArgTrue;
|
||||
uint64_t ArgFalse;
|
||||
|
||||
if (OpSize == 4) {
|
||||
ArgTrue = *GetSrc<uint32_t*>(Data->SSAData, Op->TrueVal);
|
||||
ArgFalse = *GetSrc<uint32_t*>(Data->SSAData, Op->FalseVal);
|
||||
@@ -885,10 +897,25 @@ DEF_OP(Select) {
|
||||
|
||||
bool CompResult;
|
||||
|
||||
if (Op->CompareSize == 4)
|
||||
if (Op->CompareSize == 4) {
|
||||
const auto Src1 = *GetSrc<uint32_t*>(Data->SSAData, Op->Cmp1);
|
||||
const auto Src2 = *GetSrc<uint32_t*>(Data->SSAData, Op->Cmp2);
|
||||
CompResult = IsConditionTrue<uint32_t, int32_t, float>(Op->Cond.Val, Src1, Src2);
|
||||
else
|
||||
}
|
||||
else if (Op->CompareSize == 8) {
|
||||
const auto Src1 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp1);
|
||||
const auto Src2 = *GetSrc<uint64_t*>(Data->SSAData, Op->Cmp2);
|
||||
CompResult = IsConditionTrue<uint64_t, int64_t, double>(Op->Cond.Val, Src1, Src2);
|
||||
}
|
||||
else if (Op->CompareSize == 16) {
|
||||
const auto Src1 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Cmp1);
|
||||
const auto Src2 = *GetSrc<__uint128_t*>(Data->SSAData, Op->Cmp2);
|
||||
CompResult = IsConditionTrue<__uint128_t, __int128_t, double>(Op->Cond.Val, Src1, Src2);
|
||||
}
|
||||
else {
|
||||
LOGMAN_MSG_A_FMT("Unknown select size: {}", Op->CompareSize);
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
GD = CompResult ? ArgTrue : ArgFalse;
|
||||
}
|
||||
|
||||
@@ -143,6 +143,15 @@ DEF_OP(CPUID) {
|
||||
memcpy(DstPtr, &Results, sizeof(uint32_t) * 4);
|
||||
}
|
||||
|
||||
DEF_OP(XGETBV) {
|
||||
auto Op = IROp->C<IR::IROp_XGetBV>();
|
||||
uint32_t *DstPtr = GetDest<uint32_t*>(Data->SSAData, Node);
|
||||
const uint32_t Function = *GetSrc<uint32_t*>(Data->SSAData, Op->Function);
|
||||
|
||||
auto Results = Data->State->CTX->RunXCRFunction(Function);
|
||||
memcpy(DstPtr, &Results, sizeof(uint32_t) * 2);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -211,7 +211,7 @@ DEF_OP(Vector_FToF) {
|
||||
// Little bit tricky here
|
||||
// Sometimes is used to convert from a 128bit vector register
|
||||
// in to a 64bit vector register with different sized elements
|
||||
// eg: %ssa5 i32v2 = Vector_FToF %ssa4 i128, #0x8
|
||||
// eg: %5 i32v2 = Vector_FToF %4 i128, #0x8
|
||||
uint8_t Elements = OpSize == 8 ? 2 : OpSize / Op->SrcElementSize;
|
||||
DO_VECTOR_1SRC_2TYPE_OP_NOSIZE(float, double, Func, 0, 0)
|
||||
break;
|
||||
|
||||
+1
-1
@@ -3,7 +3,7 @@
|
||||
#include <cstdint>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
enum IROps : uint8_t;
|
||||
enum IROps : uint16_t;
|
||||
}
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
|
||||
+1
-1
@@ -46,7 +46,7 @@ struct OpHandlers<IR::OP_VPCMPESTRX> {
|
||||
// Main PCMPXSTRX algorithm body. Allows for reuse with both implicit and explicit length variants.
|
||||
static uint32_t MainBody(const __uint128_t& lhs, int valid_lhs, const __uint128_t& rhs, int valid_rhs, uint16_t control) {
|
||||
const uint32_t aggregation = PerformAggregation(lhs, valid_lhs, rhs, valid_rhs, control);
|
||||
const uint32_t upper_limit = (16U >> (control & 1)) - 1;
|
||||
const int32_t upper_limit = (16 >> (control & 1)) - 1;
|
||||
|
||||
// Bits are arranged as:
|
||||
// Bit #: 3 2 1 0
|
||||
|
||||
@@ -52,6 +52,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(ADD, Add);
|
||||
REGISTER_OP(SUB, Sub);
|
||||
REGISTER_OP(NEG, Neg);
|
||||
REGISTER_OP(ABS, Abs);
|
||||
REGISTER_OP(MUL, Mul);
|
||||
REGISTER_OP(UMUL, UMul);
|
||||
REGISTER_OP(DIV, Div);
|
||||
@@ -118,6 +119,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(THREADREMOVECODEENTRY, ThreadRemoveCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
REGISTER_OP(XGETBV, XGETBV);
|
||||
|
||||
// Conversion ops
|
||||
REGISTER_OP(VINSGPR, VInsGPR);
|
||||
@@ -265,6 +267,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(VUMULL2, VUMull2);
|
||||
REGISTER_OP(VSMULL2, VSMull2);
|
||||
REGISTER_OP(VUABDL, VUABDL);
|
||||
REGISTER_OP(VUABDL2, VUABDL2);
|
||||
REGISTER_OP(VTBL1, VTBL1);
|
||||
REGISTER_OP(VREV64, VRev64);
|
||||
REGISTER_OP(VPCMPESTRX, VPCMPESTRX);
|
||||
|
||||
@@ -85,6 +85,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(Add);
|
||||
DEF_OP(Sub);
|
||||
DEF_OP(Neg);
|
||||
DEF_OP(Abs);
|
||||
DEF_OP(Mul);
|
||||
DEF_OP(UMul);
|
||||
DEF_OP(Div);
|
||||
@@ -154,6 +155,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(ThreadRemoveCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
DEF_OP(XGETBV);
|
||||
|
||||
///< Conversion ops
|
||||
DEF_OP(VInsGPR);
|
||||
@@ -291,6 +293,7 @@ namespace FEXCore::CPU {
|
||||
DEF_OP(VUMull2);
|
||||
DEF_OP(VSMull2);
|
||||
DEF_OP(VUABDL);
|
||||
DEF_OP(VUABDL2);
|
||||
DEF_OP(VTBL1);
|
||||
DEF_OP(VRev64);
|
||||
DEF_OP(VPCMPESTRX);
|
||||
@@ -347,6 +350,15 @@ namespace FEXCore::CPU {
|
||||
template<typename unsigned_type, typename signed_type, typename float_type>
|
||||
[[nodiscard]] static bool IsConditionTrue(uint8_t Cond, uint64_t Src1, uint64_t Src2) {
|
||||
bool CompResult = false;
|
||||
if constexpr (sizeof(unsigned_type) == 16) {
|
||||
LOGMAN_THROW_A_FMT(Cond != FEXCore::IR::COND_FLU &&
|
||||
Cond != FEXCore::IR::COND_FGE &&
|
||||
Cond != FEXCore::IR::COND_FLEU &&
|
||||
Cond != FEXCore::IR::COND_FGT &&
|
||||
Cond != FEXCore::IR::COND_FU &&
|
||||
Cond != FEXCore::IR::COND_FNU, "Unsupported comparison for 128-bit floats");
|
||||
}
|
||||
|
||||
switch (Cond) {
|
||||
case FEXCore::IR::COND_EQ:
|
||||
CompResult = static_cast<unsigned_type>(Src1) == static_cast<unsigned_type>(Src2);
|
||||
|
||||
@@ -190,19 +190,31 @@ DEF_OP(LoadFlag) {
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += offsetof(FEXCore::Core::CPUState, flags[0]);
|
||||
ContextPtr += Op->Flag;
|
||||
uint8_t const *MemData = reinterpret_cast<uint8_t const*>(ContextPtr);
|
||||
GD = *MemData;
|
||||
|
||||
if (Op->Flag == 24 /* NZCV */) {
|
||||
uint32_t const *MemData = reinterpret_cast<uint32_t const*>(ContextPtr);
|
||||
GD = *MemData;
|
||||
} else {
|
||||
uint8_t const *MemData = reinterpret_cast<uint8_t const*>(ContextPtr);
|
||||
GD = *MemData;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreFlag) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
uint8_t Arg = *GetSrc<uint8_t*>(Data->SSAData, Op->Value);
|
||||
uint32_t Arg = *GetSrc<uint32_t*>(Data->SSAData, Op->Value);
|
||||
|
||||
uintptr_t ContextPtr = reinterpret_cast<uintptr_t>(Data->State->CurrentFrame);
|
||||
ContextPtr += offsetof(FEXCore::Core::CPUState, flags[0]);
|
||||
ContextPtr += Op->Flag;
|
||||
uint8_t *MemData = reinterpret_cast<uint8_t*>(ContextPtr);
|
||||
*MemData = Arg;
|
||||
|
||||
if (Op->Flag == 24 /* NZCV */) {
|
||||
uint32_t *MemData = reinterpret_cast<uint32_t*>(ContextPtr);
|
||||
*MemData = Arg;
|
||||
} else {
|
||||
uint8_t *MemData = reinterpret_cast<uint8_t*>(ContextPtr);
|
||||
*MemData = Arg;
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadMem) {
|
||||
|
||||
@@ -2227,6 +2227,33 @@ DEF_OP(VUABDL) {
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VUABDL2) {
|
||||
const auto Op = IROp->C<IR::IROp_VUABDL2>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
void *Src1 = GetSrc<void*>(Data->SSAData, Op->Vector1);
|
||||
void *Src2 = GetSrc<void*>(Data->SSAData, Op->Vector2);
|
||||
|
||||
uint8_t Tmp[Core::CPUState::XMM_AVX_REG_SIZE];
|
||||
|
||||
const uint8_t ElementSize = Op->Header.ElementSize;
|
||||
const uint8_t Elements = OpSize / ElementSize;
|
||||
|
||||
const auto Func8 = [](auto a, auto b) { return std::abs((int16_t)a - (int16_t)b); };
|
||||
const auto Func16 = [](auto a, auto b) { return std::abs((int32_t)a - (int32_t)b); };
|
||||
const auto Func32 = [](auto a, auto b) { return std::abs((int64_t)a - (int64_t)b); };
|
||||
|
||||
switch (ElementSize) {
|
||||
DO_VECTOR_2SRC_2TYPE_OP_TOP_SRC(2, uint16_t, uint8_t, Func8)
|
||||
DO_VECTOR_2SRC_2TYPE_OP_TOP_SRC(4, uint32_t, uint16_t, Func16)
|
||||
DO_VECTOR_2SRC_2TYPE_OP_TOP_SRC(8, uint64_t, uint32_t, Func32)
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
memcpy(GDP, Tmp, OpSize);
|
||||
}
|
||||
|
||||
DEF_OP(VTBL1) {
|
||||
const auto Op = IROp->C<IR::IROp_VTBL1>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
@@ -2280,16 +2307,13 @@ DEF_OP(VRev64) {
|
||||
|
||||
DEF_OP(VPCMPESTRX) {
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPESTRX>();
|
||||
const auto Is64Bit = Op->GPRSize == 8;
|
||||
const auto Control = Op->Control;
|
||||
|
||||
const auto RAX = *GetSrc<uint64_t*>(Data->SSAData, Op->RAX);
|
||||
const auto RDX = *GetSrc<uint64_t*>(Data->SSAData, Op->RDX);
|
||||
const auto LHS = *GetSrc<__uint128_t*>(Data->SSAData, Op->LHS);
|
||||
const auto RHS = *GetSrc<__uint128_t*>(Data->SSAData, Op->RHS);
|
||||
|
||||
// We can be cheeky and encode the size at bit 8 to save a parameter
|
||||
const auto Control = Op->Control | (uint16_t(Is64Bit) << 8);
|
||||
|
||||
const auto Result = OpHandlers<IR::OP_VPCMPESTRX>::handle(RAX, RDX, LHS, RHS, Control);
|
||||
|
||||
memset(GDP, 0, sizeof(uint64_t));
|
||||
|
||||
+113
-14
@@ -4,6 +4,7 @@ tags: backend|arm64
|
||||
$end_info$
|
||||
*/
|
||||
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Emitter.h"
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
@@ -83,6 +84,21 @@ DEF_OP(Add) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(TestNZ) {
|
||||
auto Op = IROp->C<IR::IROp_TestNZ>();
|
||||
const uint8_t OpSize = Op->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src1.ID());
|
||||
tst(EmitSize, Src, Src);
|
||||
|
||||
// TODO: Optimize this out
|
||||
mrs(Dst, ARMEmitter::SystemRegister::NZCV);
|
||||
}
|
||||
|
||||
DEF_OP(Sub) {
|
||||
auto Op = IROp->C<IR::IROp_Sub>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
@@ -108,6 +124,26 @@ DEF_OP(Neg) {
|
||||
neg(EmitSize, GetReg(Node), GetReg(Op->Src.ID()));
|
||||
}
|
||||
|
||||
DEF_OP(Abs) {
|
||||
auto Op = IROp->C<IR::IROp_Abs>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
if (CTX->HostFeatures.SupportsCSSC) {
|
||||
// On CSSC supporting processors, this turns in to one instruction and doesn't modify flags.
|
||||
abs(EmitSize, Dst, Src);
|
||||
}
|
||||
else {
|
||||
cmp(EmitSize, Src, 0);
|
||||
cneg(EmitSize, Dst, Src, ARMEmitter::Condition::CC_MI);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Mul) {
|
||||
auto Op = IROp->C<IR::IROp_Mul>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
@@ -310,6 +346,40 @@ DEF_OP(Or) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Orlshl) {
|
||||
auto Op = IROp->C<IR::IROp_Orlshl>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
orr(EmitSize, Dst, Src1, Const << Op->BitShift);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
orr(EmitSize, Dst, Src1, Src2, ARMEmitter::ShiftType::LSL, Op->BitShift);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Orlshr) {
|
||||
auto Op = IROp->C<IR::IROp_Orlshr>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
uint64_t Const;
|
||||
if (IsInlineConstant(Op->Src2, &Const)) {
|
||||
orr(EmitSize, Dst, Src1, Const >> Op->BitShift);
|
||||
} else {
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
orr(EmitSize, Dst, Src1, Src2, ARMEmitter::ShiftType::LSR, Op->BitShift);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(And) {
|
||||
auto Op = IROp->C<IR::IROp_And>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
@@ -477,9 +547,9 @@ DEF_OP(PDep) {
|
||||
const auto IndexReg = TMP4.R();
|
||||
const auto ZeroReg = ARMEmitter::Reg::zr;
|
||||
|
||||
const auto InputReg = SRA64[0];
|
||||
const auto MaskReg = SRA64[1];
|
||||
const auto DestReg = SRA64[2];
|
||||
const auto InputReg = StaticRegisters[0];
|
||||
const auto MaskReg = StaticRegisters[1];
|
||||
const auto DestReg = StaticRegisters[2];
|
||||
|
||||
const auto SpillCode = 1U << InputReg.Idx() |
|
||||
1U << MaskReg.Idx() |
|
||||
@@ -494,7 +564,7 @@ DEF_OP(PDep) {
|
||||
// We sadly need to spill regs for this for the time being
|
||||
// TODO: Remove when scratch registers can be allocated
|
||||
// explicitly.
|
||||
SpillStaticRegs(false, SpillCode);
|
||||
SpillStaticRegs(TMP1, false, SpillCode);
|
||||
|
||||
|
||||
mov(EmitSize, InputReg, Input);
|
||||
@@ -558,7 +628,7 @@ DEF_OP(PExt) {
|
||||
// We sadly need to spill a reg for this for the time being
|
||||
// TODO: Remove when scratch registers can be allocated
|
||||
// explicitly.
|
||||
SpillStaticRegs(false, 1U << Mask.Idx());
|
||||
SpillStaticRegs(TMP2, false, 1U << Mask.Idx());
|
||||
mov(EmitSize, Mask, ZeroReg);
|
||||
|
||||
// Main loop
|
||||
@@ -633,7 +703,10 @@ DEF_OP(LDiv) {
|
||||
mov(EmitSize, ARMEmitter::Reg::r2, Divisor);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LDIVHandler));
|
||||
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
|
||||
// Move result to its destination register
|
||||
mov(EmitSize, Dst, ARMEmitter::Reg::r0);
|
||||
@@ -697,7 +770,11 @@ DEF_OP(LUDiv) {
|
||||
mov(EmitSize, ARMEmitter::Reg::r2, Divisor);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LUDIVHandler));
|
||||
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
|
||||
// Move result to its destination register
|
||||
mov(EmitSize, Dst, ARMEmitter::Reg::r0);
|
||||
|
||||
@@ -768,7 +845,11 @@ DEF_OP(LRem) {
|
||||
mov(EmitSize, ARMEmitter::Reg::r2, Divisor);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LREMHandler));
|
||||
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
|
||||
// Move result to its destination register
|
||||
mov(EmitSize, Dst, ARMEmitter::Reg::r0);
|
||||
|
||||
@@ -833,7 +914,11 @@ DEF_OP(LURem) {
|
||||
mov(EmitSize, ARMEmitter::Reg::r2, Divisor);
|
||||
|
||||
ldr(ARMEmitter::XReg::x3, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LUREMHandler));
|
||||
|
||||
str<ARMEmitter::IndexType::PRE>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, -16);
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
ldr<ARMEmitter::IndexType::POST>(ARMEmitter::XReg::lr, ARMEmitter::Reg::rsp, 16);
|
||||
|
||||
// Move result to its destination register
|
||||
mov(EmitSize, Dst, ARMEmitter::Reg::r0);
|
||||
|
||||
@@ -1020,14 +1105,22 @@ DEF_OP(Bfi) {
|
||||
const auto SrcDst = GetReg(Op->Dest.ID());
|
||||
const auto Src = GetReg(Op->Src.ID());
|
||||
|
||||
mov(EmitSize, TMP1, SrcDst);
|
||||
bfi(EmitSize, TMP1, Src, Op->lsb, Op->Width);
|
||||
|
||||
if (OpSize == 8) {
|
||||
mov(EmitSize, Dst, TMP1.R());
|
||||
if (Dst == SrcDst) {
|
||||
// If Dst and SrcDst match then this turns in to a simple BFI instruction.
|
||||
bfi(EmitSize, Dst, Src, Op->lsb, Op->Width);
|
||||
}
|
||||
else {
|
||||
ubfx(EmitSize, Dst, TMP1, 0, OpSize * 8);
|
||||
// Destination didn't match the dst source register.
|
||||
// TODO: Inefficient until FEX can have RA constraints here.
|
||||
mov(EmitSize, TMP1, SrcDst);
|
||||
bfi(EmitSize, TMP1, Src, Op->lsb, Op->Width);
|
||||
|
||||
if (OpSize == 8) {
|
||||
mov(EmitSize, Dst, TMP1.R());
|
||||
}
|
||||
else {
|
||||
ubfx(EmitSize, Dst, TMP1, 0, OpSize * 8);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1085,6 +1178,7 @@ DEF_OP(Select) {
|
||||
const auto CompareEmitSize = Op->CompareSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
uint64_t Const;
|
||||
auto cc = MapSelectCC(Op->Cond);
|
||||
|
||||
if (IsGPR(Op->Cmp1.ID())) {
|
||||
const auto Src1 = GetReg(Op->Cmp1.ID());
|
||||
@@ -1095,7 +1189,14 @@ DEF_OP(Select) {
|
||||
const auto Src2 = GetReg(Op->Cmp2.ID());
|
||||
cmp(CompareEmitSize, Src1, Src2);
|
||||
}
|
||||
} else if (IsFPR(Op->Cmp1.ID())) {
|
||||
}
|
||||
else if (IsGPRPair(Op->Cmp1.ID())) {
|
||||
const auto Src1 = GetRegPair(Op->Cmp1.ID());
|
||||
const auto Src2 = GetRegPair(Op->Cmp2.ID());
|
||||
cmp(EmitSize, Src1.first, Src2.first);
|
||||
ccmp(EmitSize, Src1.second, Src2.second, ARMEmitter::StatusFlags::None, cc);
|
||||
}
|
||||
else if (IsFPR(Op->Cmp1.ID())) {
|
||||
const auto Src1 = GetVReg(Op->Cmp1.ID());
|
||||
const auto Src2 = GetVReg(Op->Cmp2.ID());
|
||||
fcmp(Op->CompareSize == 8 ? ARMEmitter::ScalarRegSize::i64Bit : ARMEmitter::ScalarRegSize::i32Bit, Src1, Src2);
|
||||
@@ -1103,8 +1204,6 @@ DEF_OP(Select) {
|
||||
LOGMAN_MSG_A_FMT("Select: Expected GPR or FPR");
|
||||
}
|
||||
|
||||
auto cc = MapSelectCC(Op->Cond);
|
||||
|
||||
uint64_t const_true, const_false;
|
||||
bool is_const_true = IsInlineConstant(Op->TrueVal, &const_true);
|
||||
bool is_const_false = IsInlineConstant(Op->FalseVal, &const_false);
|
||||
|
||||
@@ -22,7 +22,7 @@ namespace FEXCore::CPU {
|
||||
|
||||
DEF_OP(CallbackReturn) {
|
||||
// spill back to CTX
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
// First we must reset the stack
|
||||
ResetStack();
|
||||
@@ -175,7 +175,7 @@ DEF_OP(Syscall) {
|
||||
FPRSpillMask = CALLER_FPR_MASK;
|
||||
}
|
||||
|
||||
SpillStaticRegs(true, GPRSpillMask, FPRSpillMask);
|
||||
SpillStaticRegs(TMP1, true, GPRSpillMask, FPRSpillMask);
|
||||
|
||||
// Now that we are spilled, store in the state that we are in a syscall
|
||||
// Still without overwriting registers that matter
|
||||
@@ -216,8 +216,11 @@ DEF_OP(Syscall) {
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Move result to its destination register
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), ARMEmitter::Reg::r0);
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NORETURNEDRESULT) != FEXCore::IR::SyscallFlags::NORETURNEDRESULT) {
|
||||
// Move result to its destination register.
|
||||
// Only if `NORETURNEDRESULT` wasn't set, otherwise we might overwrite the CPUState refilled with `FillStaticRegs`
|
||||
mov(ARMEmitter::Size::i64Bit, GetReg(Node), ARMEmitter::Reg::r0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -257,7 +260,7 @@ DEF_OP(InlineSyscall) {
|
||||
// Ordering is incredibly important here
|
||||
// We must spill any overlapping registers first THEN claim we are in a syscall without invalidating state at all
|
||||
// Only spill the registers that intersect with our usage
|
||||
SpillStaticRegs(false, SpillMask);
|
||||
SpillStaticRegs(TMP1, false, SpillMask);
|
||||
|
||||
// Now that we are spilled, store in the state that we are in a syscall
|
||||
// Still without overwriting registers that matter
|
||||
@@ -325,7 +328,7 @@ DEF_OP(Thunk) {
|
||||
// X0: CTX
|
||||
// X1: Args (from guest stack)
|
||||
|
||||
SpillStaticRegs(); // spill to ctx before ra64 spill
|
||||
SpillStaticRegs(TMP1); // spill to ctx before ra64 spill
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -400,12 +403,12 @@ DEF_OP(ThreadRemoveCodeEntry) {
|
||||
// X1: RIP
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, STATE.R());
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, Entry);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.ThreadRemoveCodeEntryFromJIT));
|
||||
SpillStaticRegs();
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<void, void*, void*>(ARMEmitter::Reg::r2);
|
||||
#else
|
||||
@@ -421,7 +424,7 @@ DEF_OP(CPUID) {
|
||||
auto Op = IROp->C<IR::IROp_CPUID>();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
// x0 = CPUID Handler
|
||||
// x1 = CPUID Function
|
||||
@@ -447,6 +450,34 @@ DEF_OP(CPUID) {
|
||||
mov(ARMEmitter::Size::i64Bit, Dst.second, ARMEmitter::Reg::r1);
|
||||
}
|
||||
|
||||
DEF_OP(XGETBV) {
|
||||
auto Op = IROp->C<IR::IROp_XGetBV>();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
// x0 = CPUID Handler
|
||||
// x1 = XCR Function
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj));
|
||||
ldr(ARMEmitter::XReg::x2, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.XCRFunction));
|
||||
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, GetReg(Op->Function.ID()));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint64_t, void*, uint32_t>(ARMEmitter::Reg::r2);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
#endif
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
// Results are in x0
|
||||
// Results want to be in a i32v2 vector
|
||||
auto Dst = GetRegPair(Node);
|
||||
mov(ARMEmitter::Size::i32Bit, Dst.first, ARMEmitter::Reg::r0);
|
||||
lsr(ARMEmitter::Size::i64Bit, Dst.second, ARMEmitter::Reg::r0, 32);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
}
|
||||
|
||||
+98
-50
@@ -83,7 +83,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
} else {
|
||||
switch(Info.ABI) {
|
||||
case FABI_VOID_U16:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -103,7 +103,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
|
||||
case FABI_F80_F32:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
@@ -127,7 +127,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
|
||||
case FABI_F80_F64:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -153,7 +153,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
|
||||
case FABI_F80_I16:
|
||||
case FABI_F80_I32: {
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -183,7 +183,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
|
||||
case FABI_F32_F80:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -209,7 +209,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
|
||||
case FABI_F64_F80:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -235,7 +235,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
|
||||
case FABI_F64_F64: {
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -259,7 +259,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
|
||||
case FABI_F64_F64_F64: {
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -285,7 +285,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
|
||||
case FABI_I16_F80:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -310,7 +310,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
case FABI_I32_F80:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -335,7 +335,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
case FABI_I64_F80:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -360,7 +360,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
case FABI_I64_F80_F80:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -388,7 +388,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
case FABI_F80_F80:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -415,7 +415,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
case FABI_F80_F80_F80:{
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
@@ -446,20 +446,17 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
case FABI_I32_I64_I64_I128_I128_I16: {
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPESTRX>();
|
||||
const auto Is64Bit = Op->GPRSize == 8;
|
||||
const auto Control = Op->Control;
|
||||
|
||||
const auto Src1 = GetVReg(Op->LHS.ID());
|
||||
const auto Src2 = GetVReg(Op->RHS.ID());
|
||||
const auto SrcRAX = GetReg(Op->RAX.ID());
|
||||
const auto SrcRDX = GetReg(Op->RDX.ID());
|
||||
|
||||
// We can be cheeky and encode the size at bit 8 to save a parameter
|
||||
const auto Control = Op->Control | (uint16_t(Is64Bit) << 8);
|
||||
|
||||
mov(ARMEmitter::XReg::x0, SrcRAX.X());
|
||||
mov(ARMEmitter::XReg::x1, SrcRDX.X());
|
||||
|
||||
@@ -486,7 +483,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
break;
|
||||
}
|
||||
case FABI_I32_I128_I128_I16: {
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPISTRX>();
|
||||
@@ -552,7 +549,7 @@ static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Fram
|
||||
FEXCore::ARMEmitter::Emitter::ClearICache((void*)branch, 24);
|
||||
|
||||
// Add de-linking handler
|
||||
Context::ContextImpl::ThreadAddBlockLink(Thread, GuestRip, (uintptr_t)record, [branch, LinkerAddress]{
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [branch, LinkerAddress]{
|
||||
FEXCore::ARMEmitter::Emitter emit((uint8_t*)(branch), 24);
|
||||
FEXCore::ARMEmitter::ForwardLabel l_BranchHost;
|
||||
emit.ldr(FEXCore::ARMEmitter::XReg::x0, &l_BranchHost);
|
||||
@@ -566,7 +563,7 @@ static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Fram
|
||||
record[0] = HostCode;
|
||||
|
||||
// Add de-linking handler
|
||||
Context::ContextImpl::ThreadAddBlockLink(Thread, GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
record[0] = LinkerAddress;
|
||||
});
|
||||
}
|
||||
@@ -587,14 +584,14 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::In
|
||||
|
||||
RAPass->AllocateRegisterSet(RegisterClasses);
|
||||
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRClass, ConfiguredGPRs);
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRFixedClass, ConfiguredSRAGPRs);
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRClass, ConfiguredFPRs);
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRFixedClass, ConfiguredSRAFPRs);
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRPairClass, ConfiguredGPRPairs);
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRClass, GeneralRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRFixedClass, StaticRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRClass, GeneralFPRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::FPRFixedClass, StaticFPRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::GPRPairClass, GeneralPairRegisters.size());
|
||||
RAPass->AddRegisters(FEXCore::IR::ComplexClass, 1);
|
||||
|
||||
for (uint32_t i = 0; i < ConfiguredGPRPairs; ++i) {
|
||||
for (uint32_t i = 0; i < GeneralPairRegisters.size(); ++i) {
|
||||
RAPass->AddRegisterConflict(FEXCore::IR::GPRClass, i * 2, FEXCore::IR::GPRPairClass, i);
|
||||
RAPass->AddRegisterConflict(FEXCore::IR::GPRClass, i * 2 + 1, FEXCore::IR::GPRPairClass, i);
|
||||
}
|
||||
@@ -615,6 +612,11 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::In
|
||||
Common.CPUIDFunction = PMF.GetConvertedPointer();
|
||||
}
|
||||
|
||||
{
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::CPUIDEmu::RunXCRFunction);
|
||||
Common.XCRFunction = PMF.GetConvertedPointer();
|
||||
}
|
||||
|
||||
Common.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
|
||||
Common.SyscallHandlerFunc = reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall);
|
||||
Common.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Context::ContextImpl::ThreadExitFunctionLink<Arm64JITCore_ExitFunctionLink>);
|
||||
@@ -634,6 +636,25 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::In
|
||||
|
||||
// Must be done after Dispatcher init
|
||||
ClearCache();
|
||||
|
||||
// Setup dynamic dispatch.
|
||||
if (CTX->Dispatcher->GetConfig().StaticRegisterAllocation) {
|
||||
RT_LoadRegister = &Arm64JITCore::Op_LoadRegisterSRA;
|
||||
RT_StoreRegister = &Arm64JITCore::Op_StoreRegisterSRA;
|
||||
}
|
||||
else {
|
||||
RT_LoadRegister = &Arm64JITCore::Op_LoadRegister;
|
||||
RT_StoreRegister = &Arm64JITCore::Op_StoreRegister;
|
||||
}
|
||||
|
||||
if (ParanoidTSO()) {
|
||||
RT_LoadMemTSO = &Arm64JITCore::Op_ParanoidLoadMemTSO;
|
||||
RT_StoreMemTSO = &Arm64JITCore::Op_ParanoidStoreMemTSO;
|
||||
}
|
||||
else {
|
||||
RT_LoadMemTSO = &Arm64JITCore::Op_LoadMemTSO;
|
||||
RT_StoreMemTSO = &Arm64JITCore::Op_StoreMemTSO;
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64JITCore::EmitDetectionString() {
|
||||
@@ -703,6 +724,12 @@ bool Arm64JITCore::IsGPR(IR::NodeID Node) const {
|
||||
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
|
||||
}
|
||||
|
||||
bool Arm64JITCore::IsGPRPair(IR::NodeID Node) const {
|
||||
auto Class = GetRegClass(Node);
|
||||
|
||||
return Class == IR::GPRPairClass;
|
||||
}
|
||||
|
||||
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
FEXCore::IR::IRListView const *IR,
|
||||
FEXCore::Core::DebugData *DebugData,
|
||||
@@ -814,6 +841,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) {
|
||||
const auto ID = IR->GetID(CodeNode);
|
||||
switch (IROp->Op) {
|
||||
#define REGISTER_OP_RT(op, x) case FEXCore::IR::IROps::OP_##op: std::invoke(RT_##x, this, IROp, ID); break
|
||||
#define REGISTER_OP(op, x) case FEXCore::IR::IROps::OP_##op: Op_##x(IROp, ID); break
|
||||
// ALU ops
|
||||
REGISTER_OP(TRUNCELEMENTPAIR, TruncElementPair);
|
||||
@@ -823,8 +851,10 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
REGISTER_OP(INLINEENTRYPOINTOFFSET, InlineEntrypointOffset);
|
||||
REGISTER_OP(CYCLECOUNTER, CycleCounter);
|
||||
REGISTER_OP(ADD, Add);
|
||||
REGISTER_OP(TESTNZ, TestNZ);
|
||||
REGISTER_OP(SUB, Sub);
|
||||
REGISTER_OP(NEG, Neg);
|
||||
REGISTER_OP(ABS, Abs);
|
||||
REGISTER_OP(MUL, Mul);
|
||||
REGISTER_OP(UMUL, UMul);
|
||||
REGISTER_OP(DIV, Div);
|
||||
@@ -834,6 +864,8 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
REGISTER_OP(MULH, MulH);
|
||||
REGISTER_OP(UMULH, UMulH);
|
||||
REGISTER_OP(OR, Or);
|
||||
REGISTER_OP(ORLSHL, Orlshl);
|
||||
REGISTER_OP(ORLSHR, Orlshr);
|
||||
REGISTER_OP(AND, And);
|
||||
REGISTER_OP(ANDN, Andn);
|
||||
REGISTER_OP(XOR, Xor);
|
||||
@@ -891,6 +923,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(THREADREMOVECODEENTRY, ThreadRemoveCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
REGISTER_OP(XGETBV, XGETBV);
|
||||
|
||||
// Conversion ops
|
||||
REGISTER_OP(VINSGPR, VInsGPR);
|
||||
@@ -920,8 +953,8 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
// Memory ops
|
||||
REGISTER_OP(LOADCONTEXT, LoadContext);
|
||||
REGISTER_OP(STORECONTEXT, StoreContext);
|
||||
REGISTER_OP(LOADREGISTER, LoadRegister);
|
||||
REGISTER_OP(STOREREGISTER, StoreRegister);
|
||||
REGISTER_OP_RT(LOADREGISTER, LoadRegister);
|
||||
REGISTER_OP_RT(STOREREGISTER, StoreRegister);
|
||||
REGISTER_OP(LOADCONTEXTINDEXED, LoadContextIndexed);
|
||||
REGISTER_OP(STORECONTEXTINDEXED, StoreContextIndexed);
|
||||
REGISTER_OP(SPILLREGISTER, SpillRegister);
|
||||
@@ -930,22 +963,8 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
REGISTER_OP(STOREFLAG, StoreFlag);
|
||||
REGISTER_OP(LOADMEM, LoadMem);
|
||||
REGISTER_OP(STOREMEM, StoreMem);
|
||||
case FEXCore::IR::IROps::OP_LOADMEMTSO:
|
||||
if (ParanoidTSO()) {
|
||||
Op_ParanoidLoadMemTSO(IROp, ID);
|
||||
}
|
||||
else {
|
||||
Op_LoadMemTSO(IROp, ID);
|
||||
}
|
||||
break;
|
||||
case FEXCore::IR::IROps::OP_STOREMEMTSO:
|
||||
if (ParanoidTSO()) {
|
||||
Op_ParanoidStoreMemTSO(IROp, ID);
|
||||
}
|
||||
else {
|
||||
Op_StoreMemTSO(IROp, ID);
|
||||
}
|
||||
break;
|
||||
REGISTER_OP_RT(LOADMEMTSO, LoadMemTSO);
|
||||
REGISTER_OP_RT(STOREMEMTSO, StoreMemTSO);
|
||||
REGISTER_OP(VLOADVECTORMASKED, VLoadVectorMasked);
|
||||
REGISTER_OP(VSTOREVECTORMASKED, VStoreVectorMasked);
|
||||
|
||||
@@ -1063,6 +1082,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
REGISTER_OP(VUMULL2, VUMull2);
|
||||
REGISTER_OP(VSMULL2, VSMull2);
|
||||
REGISTER_OP(VUABDL, VUABDL);
|
||||
REGISTER_OP(VUABDL2, VUABDL2);
|
||||
REGISTER_OP(VTBL1, VTBL1);
|
||||
REGISTER_OP(VREV64, VRev64);
|
||||
#undef REGISTER_OP
|
||||
@@ -1088,28 +1108,54 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
}
|
||||
PendingTargetLabel = nullptr;
|
||||
|
||||
// CodeSize not including the tail data.
|
||||
const uint64_t CodeOnlySize = GetCursorAddress<uint8_t *>() - CodeData.BlockBegin;
|
||||
|
||||
// Add the JitCodeTail
|
||||
auto JITBlockTailLocation = GetCursorAddress<uint8_t *>();
|
||||
auto JITBlockTail = GetCursorAddress<JITCodeTail*>();
|
||||
CursorIncrement(sizeof(JITCodeTail));
|
||||
|
||||
auto JITRIPEntriesLocation = GetCursorAddress<uint8_t *>();
|
||||
auto JITRIPEntries = GetCursorAddress<JITRIPReconstructEntries*>();
|
||||
|
||||
CursorIncrement(sizeof(JITRIPReconstructEntries) * DebugData->GuestOpcodes.size());
|
||||
|
||||
// Put the block's RIP entry in the tail.
|
||||
// This will be used for RIP reconstruction in the future.
|
||||
// TODO: This needs to be a data RIP relocation once code caching works.
|
||||
// Current relocation code doesn't support this feature yet.
|
||||
JITBlockTail->RIP = Entry;
|
||||
|
||||
{
|
||||
// Store the RIP entries.
|
||||
JITBlockTail->NumberOfRIPEntries = DebugData->GuestOpcodes.size();
|
||||
JITBlockTail->OffsetToRIPEntries = JITRIPEntriesLocation - JITBlockTailLocation;
|
||||
uintptr_t CurrentRIPOffset = 0;
|
||||
uint64_t CurrentPCOffset = 0;
|
||||
for (size_t i = 0; i < DebugData->GuestOpcodes.size(); i++) {
|
||||
const auto &GuestOpcode = DebugData->GuestOpcodes[i];
|
||||
auto &RIPEntry = JITRIPEntries[i];
|
||||
RIPEntry.HostPCOffset = GuestOpcode.HostEntryOffset - CurrentPCOffset;
|
||||
RIPEntry.GuestRIPOffset = GuestOpcode.GuestEntryOffset - CurrentRIPOffset;
|
||||
CurrentPCOffset = GuestOpcode.HostEntryOffset;
|
||||
CurrentRIPOffset = GuestOpcode.GuestEntryOffset;
|
||||
}
|
||||
}
|
||||
|
||||
CodeHeader->OffsetToBlockTail = JITBlockTailLocation - CodeData.BlockBegin;
|
||||
|
||||
CodeData.Size = GetCursorAddress<uint8_t *>() - CodeData.BlockBegin;
|
||||
|
||||
JITBlockTail->Size = CodeData.Size;
|
||||
|
||||
ClearICache(CodeData.BlockBegin, CodeData.Size);
|
||||
ClearICache(CodeData.BlockBegin, CodeOnlySize);
|
||||
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
const auto DisasmEnd = GetCursorAddress<const vixl::aarch64::Instruction*>();
|
||||
Disasm.DisassembleBuffer(DisasmBegin, DisasmEnd);
|
||||
if (Disassemble() & FEXCore::Config::Disassemble::BLOCKS) {
|
||||
const auto DisasmEnd = reinterpret_cast<const vixl::aarch64::Instruction*>(JITBlockTailLocation);
|
||||
Disasm.DisassembleBuffer(DisasmBegin, DisasmEnd);
|
||||
}
|
||||
#endif
|
||||
|
||||
if (DebugData) {
|
||||
@@ -1144,7 +1190,9 @@ fextl::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::ContextImpl *
|
||||
|
||||
CPUBackendFeatures GetArm64JITBackendFeatures() {
|
||||
return CPUBackendFeatures {
|
||||
.SupportsStaticRegisterAllocation = true
|
||||
.SupportsStaticRegisterAllocation = true,
|
||||
.SupportsShiftedBitwise = true,
|
||||
.SupportsFlags = true,
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
+24
-5
@@ -70,9 +70,9 @@ private:
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::GPRFixedClass.Val) {
|
||||
return SRA64[Reg.Reg];
|
||||
return StaticRegisters[Reg.Reg];
|
||||
} else if (Reg.Class == IR::GPRClass.Val) {
|
||||
return RA64[Reg.Reg];
|
||||
return GeneralRegisters[Reg.Reg];
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
@@ -84,9 +84,9 @@ private:
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
if (Reg.Class == IR::FPRFixedClass.Val) {
|
||||
return SRAFPR[Reg.Reg];
|
||||
return StaticFPRegisters[Reg.Reg];
|
||||
} else if (Reg.Class == IR::FPRClass.Val) {
|
||||
return RAFPR[Reg.Reg];
|
||||
return GeneralFPRegisters[Reg.Reg];
|
||||
}
|
||||
|
||||
FEX_UNREACHABLE;
|
||||
@@ -97,7 +97,7 @@ private:
|
||||
|
||||
LOGMAN_THROW_AA_FMT(Reg.Class == IR::GPRPairClass.Val, "Unexpected Class: {}", Reg.Class);
|
||||
|
||||
return RA64Pair[Reg.Reg];
|
||||
return GeneralPairRegisters[Reg.Reg];
|
||||
}
|
||||
|
||||
[[nodiscard]] FEXCore::IR::RegisterClassType GetRegClass(IR::NodeID Node) const;
|
||||
@@ -112,6 +112,7 @@ private:
|
||||
|
||||
[[nodiscard]] bool IsFPR(IR::NodeID Node) const;
|
||||
[[nodiscard]] bool IsGPR(IR::NodeID Node) const;
|
||||
[[nodiscard]] bool IsGPRPair(IR::NodeID Node) const;
|
||||
|
||||
[[nodiscard]] FEXCore::ARMEmitter::ExtendedMemOperand GenerateMemOperand(uint8_t AccessSize,
|
||||
FEXCore::ARMEmitter::Register Base,
|
||||
@@ -210,6 +211,16 @@ private:
|
||||
/** @} */
|
||||
|
||||
uint32_t SpillSlots{};
|
||||
using OpType = void (Arm64JITCore::*)(IR::IROp_Header const *IROp, IR::NodeID Node);
|
||||
|
||||
// Runtime selection;
|
||||
// Load and store register style.
|
||||
OpType RT_LoadRegister;
|
||||
OpType RT_StoreRegister;
|
||||
// Load and store TSO memory style
|
||||
OpType RT_LoadMemTSO;
|
||||
OpType RT_StoreMemTSO;
|
||||
|
||||
#define DEF_OP(x) void Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
|
||||
///< Unhandled handler
|
||||
@@ -226,8 +237,10 @@ private:
|
||||
DEF_OP(InlineEntrypointOffset);
|
||||
DEF_OP(CycleCounter);
|
||||
DEF_OP(Add);
|
||||
DEF_OP(TestNZ);
|
||||
DEF_OP(Sub);
|
||||
DEF_OP(Neg);
|
||||
DEF_OP(Abs);
|
||||
DEF_OP(Mul);
|
||||
DEF_OP(UMul);
|
||||
DEF_OP(Div);
|
||||
@@ -237,6 +250,8 @@ private:
|
||||
DEF_OP(MulH);
|
||||
DEF_OP(UMulH);
|
||||
DEF_OP(Or);
|
||||
DEF_OP(Orlshl);
|
||||
DEF_OP(Orlshr);
|
||||
DEF_OP(And);
|
||||
DEF_OP(Andn);
|
||||
DEF_OP(Xor);
|
||||
@@ -297,6 +312,7 @@ private:
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(ThreadRemoveCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
DEF_OP(XGETBV);
|
||||
|
||||
///< Conversion ops
|
||||
DEF_OP(VInsGPR);
|
||||
@@ -318,6 +334,8 @@ private:
|
||||
DEF_OP(StoreContext);
|
||||
DEF_OP(LoadRegister);
|
||||
DEF_OP(StoreRegister);
|
||||
DEF_OP(LoadRegisterSRA);
|
||||
DEF_OP(StoreRegisterSRA);
|
||||
DEF_OP(LoadContextIndexed);
|
||||
DEF_OP(StoreContextIndexed);
|
||||
DEF_OP(SpillRegister);
|
||||
@@ -439,6 +457,7 @@ private:
|
||||
DEF_OP(VUMull2);
|
||||
DEF_OP(VSMull2);
|
||||
DEF_OP(VUABDL);
|
||||
DEF_OP(VUABDL2);
|
||||
DEF_OP(VTBL1);
|
||||
DEF_OP(VRev64);
|
||||
|
||||
|
||||
+306
-79
@@ -10,6 +10,7 @@ $end_info$
|
||||
#include "Interface/Core/CPUID.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
@@ -128,13 +129,177 @@ DEF_OP(LoadRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
[[maybe_unused]] const auto regId = (Op->Offset / Core::CPUState::GPR_REG_SIZE) - 1;
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0 || regOffs == 1, "unexpected regOffs");
|
||||
ldrb(GetReg(Node), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
ldrh(GetReg(Node), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
ldr(GetReg(Node).W(), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
ldr(GetReg(Node).X(), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadRegister GPR size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
else if (Op->Class == IR::FPRClass) {
|
||||
const auto regSize = HostSupportsSVE ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
[[maybe_unused]] const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE, "Unsupported code path!");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
|
||||
const auto host = GetVReg(Node);
|
||||
|
||||
const auto regOffs = Op->Offset & 15;
|
||||
|
||||
switch (OpSize) {
|
||||
case 1: {
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
|
||||
ldrb(host, STATE, Op->Offset);
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
|
||||
ldrh(host, STATE, Op->Offset);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 3) == 0, "unexpected regOffs: {}", regOffs);
|
||||
ldr(host.S(), STATE, Op->Offset);
|
||||
break;
|
||||
}
|
||||
|
||||
case 8: {
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 7) == 0, "unexpected regOffs: {}", regOffs);
|
||||
ldr(host.D(), STATE, Op->Offset);
|
||||
break;
|
||||
}
|
||||
|
||||
case 16: {
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
|
||||
ldr(host.Q(), STATE, Op->Offset);
|
||||
break;
|
||||
}
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreRegister) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
[[maybe_unused]] const auto regId = (Op->Offset / Core::CPUState::GPR_REG_SIZE) - 1;
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0 || regOffs == 1, "unexpected regOffs");
|
||||
strb(Src, STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
strh(Src, STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
str(Src.W(), STATE, Op->Offset);
|
||||
break;
|
||||
case 8:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs");
|
||||
str(Src.X(), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreRegister GPR size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
} else if (Op->Class == IR::FPRClass) {
|
||||
const auto regSize = HostSupportsSVE ? Core::CPUState::XMM_AVX_REG_SIZE
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
[[maybe_unused]] const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
|
||||
LOGMAN_THROW_A_FMT(HostSupportsSVE, "Unsupported code path!");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "regId out of range");
|
||||
|
||||
const auto host = GetVReg(Op->Value.ID());
|
||||
|
||||
const auto regOffs = Op->Offset & 15;
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
strb(host, STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 2:
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 1) == 0, "unexpected regOffs: {}", regOffs);
|
||||
strh(host, STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 4:
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 3) == 0, "unexpected regOffs: {}", regOffs);
|
||||
str(host.S(), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 8:
|
||||
LOGMAN_THROW_AA_FMT((regOffs & 7) == 0, "unexpected regOffs: {}", regOffs);
|
||||
str(host.D(), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
case 16:
|
||||
LOGMAN_THROW_AA_FMT(regOffs == 0, "unexpected regOffs: {}", regOffs);
|
||||
str(host.Q(), STATE, Op->Offset);
|
||||
break;
|
||||
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreRegister FPR size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
LOGMAN_THROW_AA_FMT(false, "Unhandled Op->Class {}", Op->Class);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(LoadRegisterSRA) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
if (Op->Class == IR::GPRClass) {
|
||||
const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.gregs[0])) / Core::CPUState::GPR_REG_SIZE;
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRA64.size(), "out of range regId");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
|
||||
const auto reg = SRA64[regId];
|
||||
const auto reg = StaticRegisters[regId];
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
@@ -170,9 +335,9 @@ DEF_OP(LoadRegister) {
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRAFPR.size(), "out of range regId");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
|
||||
const auto guest = SRAFPR[regId];
|
||||
const auto guest = StaticFPRegisters[regId];
|
||||
const auto host = GetVReg(Node);
|
||||
|
||||
if (HostSupportsSVE) {
|
||||
@@ -312,7 +477,7 @@ DEF_OP(LoadRegister) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(StoreRegister) {
|
||||
DEF_OP(StoreRegisterSRA) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -320,9 +485,9 @@ DEF_OP(StoreRegister) {
|
||||
const auto regId = (Op->Offset / Core::CPUState::GPR_REG_SIZE) - 1;
|
||||
const auto regOffs = Op->Offset & 7;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRA64.size(), "out of range regId");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "out of range regId");
|
||||
|
||||
const auto reg = SRA64[regId];
|
||||
const auto reg = StaticRegisters[regId];
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
|
||||
switch (OpSize) {
|
||||
@@ -356,9 +521,9 @@ DEF_OP(StoreRegister) {
|
||||
: Core::CPUState::XMM_SSE_REG_SIZE;
|
||||
const auto regId = (Op->Offset - offsetof(Core::CpuStateFrame, State.xmm.avx.data[0][0])) / regSize;
|
||||
|
||||
LOGMAN_THROW_A_FMT(regId < SRAFPR.size(), "regId out of range");
|
||||
LOGMAN_THROW_A_FMT(regId < StaticFPRegisters.size(), "regId out of range");
|
||||
|
||||
const auto guest = SRAFPR[regId];
|
||||
const auto guest = StaticFPRegisters[regId];
|
||||
const auto host = GetVReg(Op->Value.ID());
|
||||
|
||||
if (HostSupportsSVE) {
|
||||
@@ -500,7 +665,6 @@ DEF_OP(StoreRegister) {
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
DEF_OP(LoadContextIndexed) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -513,10 +677,7 @@ DEF_OP(LoadContextIndexed) {
|
||||
case 2:
|
||||
case 4:
|
||||
case 8: {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Op->Stride);
|
||||
mul(ARMEmitter::Size::i64Bit, TMP1, Index, TMP1);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, TMP1.R());
|
||||
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, Index, FEXCore::ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Op->Stride));
|
||||
const auto Dst = GetReg(Node);
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
@@ -553,10 +714,7 @@ DEF_OP(LoadContextIndexed) {
|
||||
case 8:
|
||||
case 16:
|
||||
case 32: {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Op->Stride);
|
||||
mul(ARMEmitter::Size::i64Bit, TMP1, Index, TMP1);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, TMP1.R());
|
||||
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, Index, FEXCore::ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Op->Stride));
|
||||
const auto Dst = GetVReg(Node);
|
||||
|
||||
switch (OpSize) {
|
||||
@@ -611,9 +769,7 @@ DEF_OP(StoreContextIndexed) {
|
||||
case 2:
|
||||
case 4:
|
||||
case 8: {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Op->Stride);
|
||||
mul(ARMEmitter::Size::i64Bit, TMP1, Index, TMP1);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, TMP1.R());
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, Index, FEXCore::ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Op->Stride));
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
@@ -652,9 +808,7 @@ DEF_OP(StoreContextIndexed) {
|
||||
case 8:
|
||||
case 16:
|
||||
case 32: {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Op->Stride);
|
||||
mul(ARMEmitter::Size::i64Bit, TMP1, Index, TMP1);
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, TMP1.R());
|
||||
add(ARMEmitter::Size::i64Bit, TMP1, STATE, Index, FEXCore::ARMEmitter::ShiftType::LSL, FEXCore::ilog2(Op->Stride));
|
||||
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
@@ -897,12 +1051,20 @@ DEF_OP(FillRegister) {
|
||||
DEF_OP(LoadFlag) {
|
||||
auto Op = IROp->C<IR::IROp_LoadFlag>();
|
||||
auto Dst = GetReg(Node);
|
||||
ldrb(Dst, STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
|
||||
if (Op->Flag == 24 /* NZCV */)
|
||||
ldr(Dst.W(), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
else
|
||||
ldrb(Dst, STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
}
|
||||
|
||||
DEF_OP(StoreFlag) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
strb(GetReg(Op->Value.ID()), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
|
||||
if (Op->Flag == 24 /* NZCV */)
|
||||
str(GetReg(Op->Value.ID()).W(), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
else
|
||||
strb(GetReg(Op->Value.ID()), STATE, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag);
|
||||
}
|
||||
|
||||
FEXCore::ARMEmitter::ExtendedMemOperand Arm64JITCore::GenerateMemOperand(uint8_t AccessSize,
|
||||
@@ -922,9 +1084,9 @@ FEXCore::ARMEmitter::ExtendedMemOperand Arm64JITCore::GenerateMemOperand(uint8_t
|
||||
} else {
|
||||
auto RegOffset = GetReg(Offset.ID());
|
||||
switch(OffsetType.Val) {
|
||||
case IR::MEM_OFFSET_SXTX.Val: return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::SXTX, (int)std::log2(OffsetScale) );
|
||||
case IR::MEM_OFFSET_UXTW.Val: return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::UXTW, (int)std::log2(OffsetScale) );
|
||||
case IR::MEM_OFFSET_SXTW.Val: return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::SXTW, (int)std::log2(OffsetScale) );
|
||||
case IR::MEM_OFFSET_SXTX.Val: return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::SXTX, FEXCore::ilog2(OffsetScale) );
|
||||
case IR::MEM_OFFSET_UXTW.Val: return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::UXTW, FEXCore::ilog2(OffsetScale) );
|
||||
case IR::MEM_OFFSET_SXTW.Val: return ARMEmitter::ExtendedMemOperand(Base.X(), RegOffset.X(), ARMEmitter::ExtendedType::SXTW, FEXCore::ilog2(OffsetScale) );
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled GenerateMemOperand OffsetType: {}", OffsetType.Val); break;
|
||||
}
|
||||
}
|
||||
@@ -1077,8 +1239,6 @@ DEF_OP(LoadMemTSO) {
|
||||
ldapurb(Dst, MemReg, Offset);
|
||||
}
|
||||
else {
|
||||
// Aligned
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
ldapurh(Dst, MemReg, Offset);
|
||||
@@ -1093,19 +1253,17 @@ DEF_OP(LoadMemTSO) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
}
|
||||
else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
if (OpSize == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
ldaprb(Dst.W(), MemReg);
|
||||
}
|
||||
else {
|
||||
// Aligned
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
ldaprh(Dst.W(), MemReg);
|
||||
@@ -1120,6 +1278,7 @@ DEF_OP(LoadMemTSO) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
}
|
||||
@@ -1130,8 +1289,6 @@ DEF_OP(LoadMemTSO) {
|
||||
ldarb(Dst, MemReg);
|
||||
}
|
||||
else {
|
||||
// Aligned
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
ldarh(Dst, MemReg);
|
||||
@@ -1146,11 +1303,11 @@ DEF_OP(LoadMemTSO) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
}
|
||||
}
|
||||
else {
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
switch (OpSize) {
|
||||
@@ -1178,7 +1335,8 @@ DEF_OP(LoadMemTSO) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled LoadMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
// Half-barrier.
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISHLD);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1363,6 +1521,7 @@ DEF_OP(StoreMemTSO) {
|
||||
stlurb(Src, MemReg, Offset);
|
||||
}
|
||||
else {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
@@ -1378,7 +1537,6 @@ DEF_OP(StoreMemTSO) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
nop();
|
||||
}
|
||||
}
|
||||
else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
@@ -1389,6 +1547,7 @@ DEF_OP(StoreMemTSO) {
|
||||
stlrb(Src, MemReg);
|
||||
}
|
||||
else {
|
||||
// Half-barrier once back-patched.
|
||||
nop();
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
@@ -1404,10 +1563,10 @@ DEF_OP(StoreMemTSO) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
nop();
|
||||
}
|
||||
}
|
||||
else {
|
||||
// Half-Barrier.
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
const auto Src = GetVReg(Op->Value.ID());
|
||||
const auto MemSrc = GenerateMemOperand(OpSize, MemReg, Op->Offset, Op->OffsetType, Op->OffsetScale);
|
||||
@@ -1436,7 +1595,6 @@ DEF_OP(StoreMemTSO) {
|
||||
LOGMAN_MSG_A_FMT("Unhandled StoreMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1860,32 +2018,78 @@ DEF_OP(MemCpy) {
|
||||
// Destination already set to the final pointer.
|
||||
}
|
||||
|
||||
|
||||
|
||||
DEF_OP(ParanoidLoadMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_LoadMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Addr = GetReg(Op->Addr.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
LOGMAN_MSG_A_FMT("ParanoidLoadMemTSO: No offset allowed");
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
(void)IsInlineConstant(Op->Offset, &Offset);
|
||||
}
|
||||
|
||||
if (OpSize == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
const auto Dst = GetReg(Node);
|
||||
ldapurb(Dst, MemReg, Offset);
|
||||
}
|
||||
else {
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
ldapurh(Dst, MemReg, Offset);
|
||||
break;
|
||||
case 4:
|
||||
ldapur(Dst.W(), MemReg, Offset);
|
||||
break;
|
||||
case 8:
|
||||
ldapur(Dst.X(), MemReg, Offset);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
else if (CTX->HostFeatures.SupportsRCPC && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
if (OpSize == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
ldaprb(Dst.W(), MemReg);
|
||||
}
|
||||
else {
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
ldaprh(Dst.W(), MemReg);
|
||||
break;
|
||||
case 4:
|
||||
ldapr(Dst.W(), MemReg);
|
||||
break;
|
||||
case 8:
|
||||
ldapr(Dst.X(), MemReg);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Dst = GetReg(Node);
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
ldarb(Dst, Addr);
|
||||
ldarb(Dst, MemReg);
|
||||
break;
|
||||
case 2:
|
||||
ldarh(Dst, Addr);
|
||||
ldarh(Dst, MemReg);
|
||||
break;
|
||||
case 4:
|
||||
ldar(Dst.W(), Addr);
|
||||
ldar(Dst.W(), MemReg);
|
||||
break;
|
||||
case 8:
|
||||
ldar(Dst.X(), Addr);
|
||||
ldar(Dst.X(), MemReg);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled ParanoidLoadMemTSO size: {}", OpSize);
|
||||
@@ -1896,31 +2100,30 @@ DEF_OP(ParanoidLoadMemTSO) {
|
||||
const auto Dst = GetVReg(Node);
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
ldarb(TMP1, Addr);
|
||||
ins(ARMEmitter::SubRegSize::i8Bit, Dst, 0, TMP1);
|
||||
ldarb(TMP1, MemReg);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
|
||||
break;
|
||||
case 2:
|
||||
ldarh(TMP1, Addr);
|
||||
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 0, TMP1);
|
||||
ldarh(TMP1, MemReg);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
|
||||
break;
|
||||
case 4:
|
||||
ldar(TMP1.W(), Addr);
|
||||
ins(ARMEmitter::SubRegSize::i32Bit, Dst, 0, TMP1);
|
||||
ldar(TMP1.W(), MemReg);
|
||||
fmov(ARMEmitter::Size::i32Bit, Dst.S(), TMP1.W());
|
||||
break;
|
||||
case 8:
|
||||
ldar(TMP1, Addr);
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, TMP1);
|
||||
ldar(TMP1, MemReg);
|
||||
fmov(ARMEmitter::Size::i64Bit, Dst.D(), TMP1);
|
||||
break;
|
||||
case 16:
|
||||
nop();
|
||||
ldaxp(ARMEmitter::Size::i64Bit, TMP1, TMP2, Addr);
|
||||
ldaxp(ARMEmitter::Size::i64Bit, TMP1, TMP2, MemReg);
|
||||
clrex();
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, TMP1);
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 1, TMP2);
|
||||
break;
|
||||
case 32:
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), PRED_TMP_32B.Zeroing(), Addr);
|
||||
ld1b<ARMEmitter::SubRegSize::i8Bit>(Dst.Z(), PRED_TMP_32B.Zeroing(), MemReg);
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
break;
|
||||
default:
|
||||
@@ -1934,26 +2137,50 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
const auto Op = IROp->C<IR::IROp_StoreMemTSO>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Addr = GetReg(Op->Addr.ID());
|
||||
const auto MemReg = GetReg(Op->Addr.ID());
|
||||
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
LOGMAN_MSG_A_FMT("ParanoidStoreMemTSO: No offset allowed");
|
||||
if (CTX->HostFeatures.SupportsTSOImm9 && Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
uint64_t Offset = 0;
|
||||
if (!Op->Offset.IsInvalid()) {
|
||||
(void)IsInlineConstant(Op->Offset, &Offset);
|
||||
}
|
||||
|
||||
if (OpSize == 1) {
|
||||
// 8bit load is always aligned to natural alignment
|
||||
stlurb(Src, MemReg, Offset);
|
||||
}
|
||||
else {
|
||||
switch (OpSize) {
|
||||
case 2:
|
||||
stlurh(Src, MemReg, Offset);
|
||||
break;
|
||||
case 4:
|
||||
stlur(Src.W(), MemReg, Offset);
|
||||
break;
|
||||
case 8:
|
||||
stlur(Src.X(), MemReg, Offset);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
else if (Op->Class == FEXCore::IR::GPRClass) {
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
stlrb(Src, Addr);
|
||||
stlrb(Src, MemReg);
|
||||
break;
|
||||
case 2:
|
||||
stlrh(Src, Addr);
|
||||
stlrh(Src, MemReg);
|
||||
break;
|
||||
case 4:
|
||||
stlr(Src.W(), Addr);
|
||||
stlr(Src.W(), MemReg);
|
||||
break;
|
||||
case 8:
|
||||
stlr(Src.X(), Addr);
|
||||
stlr(Src.X(), MemReg);
|
||||
break;
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled ParanoidStoreMemTSO size: {}", OpSize);
|
||||
@@ -1966,19 +2193,19 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
switch (OpSize) {
|
||||
case 1:
|
||||
umov<ARMEmitter::SubRegSize::i8Bit>(TMP1, Src, 0);
|
||||
stlrb(TMP1, Addr);
|
||||
stlrb(TMP1, MemReg);
|
||||
break;
|
||||
case 2:
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(TMP1, Src, 0);
|
||||
stlrh(TMP1, Addr);
|
||||
stlrh(TMP1, MemReg);
|
||||
break;
|
||||
case 4:
|
||||
umov<ARMEmitter::SubRegSize::i32Bit>(TMP1, Src, 0);
|
||||
stlr(TMP1.W(), Addr);
|
||||
stlr(TMP1.W(), MemReg);
|
||||
break;
|
||||
case 8:
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(TMP1, Src, 0);
|
||||
stlr(TMP1, Addr);
|
||||
stlr(TMP1, MemReg);
|
||||
break;
|
||||
case 16: {
|
||||
// Move vector to GPRs
|
||||
@@ -1988,14 +2215,14 @@ DEF_OP(ParanoidStoreMemTSO) {
|
||||
Bind(&B);
|
||||
|
||||
// ldaxp must not have both the destination registers be the same
|
||||
ldaxp(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::zr, TMP3, Addr); // <- Can hit SIGBUS. Overwritten with DMB
|
||||
stlxp(ARMEmitter::Size::i64Bit, TMP3, TMP1, TMP2, Addr); // <- Can also hit SIGBUS
|
||||
ldaxp(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::zr, TMP3, MemReg); // <- Can hit SIGBUS. Overwritten with DMB
|
||||
stlxp(ARMEmitter::Size::i64Bit, TMP3, TMP1, TMP2, MemReg); // <- Can also hit SIGBUS
|
||||
cbnz(ARMEmitter::Size::i64Bit, TMP3, &B); // < Overwritten with DMB
|
||||
break;
|
||||
}
|
||||
case 32: {
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(Src.Z(), PRED_TMP_32B, Addr, 0);
|
||||
st1b<ARMEmitter::SubRegSize::i8Bit>(Src.Z(), PRED_TMP_32B, MemReg, 0);
|
||||
dmb(FEXCore::ARMEmitter::BarrierScope::ISH);
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -12,6 +12,8 @@ $end_info$
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
#include "FEXCore/Debug/InternalThreadState.h"
|
||||
|
||||
#include <FEXCore/Core/SignalDelegator.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
|
||||
@@ -37,7 +39,6 @@ DEF_OP(Fence) {
|
||||
}
|
||||
}
|
||||
|
||||
#ifndef _WIN32
|
||||
DEF_OP(Break) {
|
||||
auto Op = IROp->C<IR::IROp_Break>();
|
||||
|
||||
@@ -59,15 +60,15 @@ DEF_OP(Break) {
|
||||
str(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
|
||||
|
||||
switch (Op->Reason.Signal) {
|
||||
case SIGILL:
|
||||
case Core::FAULT_SIGILL:
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGILL));
|
||||
br(TMP1);
|
||||
break;
|
||||
case SIGTRAP:
|
||||
case Core::FAULT_SIGTRAP:
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
|
||||
br(TMP1);
|
||||
break;
|
||||
case SIGSEGV:
|
||||
case Core::FAULT_SIGSEGV:
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGSEGV));
|
||||
br(TMP1);
|
||||
break;
|
||||
@@ -77,11 +78,6 @@ DEF_OP(Break) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
#else
|
||||
DEF_OP(Break) {
|
||||
ERROR_AND_DIE_FMT("Unsupported");
|
||||
}
|
||||
#endif
|
||||
|
||||
DEF_OP(GetRoundingMode) {
|
||||
auto Dst = GetReg(Node);
|
||||
@@ -148,7 +144,7 @@ DEF_OP(Print) {
|
||||
auto Op = IROp->C<IR::IROp_Print>();
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
SpillStaticRegs();
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
if (IsGPR(Op->Value.ID())) {
|
||||
mov(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GetReg(Op->Value.ID()));
|
||||
@@ -174,7 +170,7 @@ DEF_OP(ProcessorID) {
|
||||
// Ordering is incredibly important here
|
||||
// We must spill any overlapping registers first THEN claim we are in a syscall without invalidating state at all
|
||||
// Only spill the registers that intersect with our usage
|
||||
SpillStaticRegs(false, SpillMask);
|
||||
SpillStaticRegs(TMP1, false, SpillMask);
|
||||
|
||||
// Now that we are spilled, store in the state that we are in a syscall
|
||||
// Still without overwriting registers that matter
|
||||
|
||||
+59
-26
@@ -8,6 +8,8 @@ $end_info$
|
||||
#include "Interface/Core/ArchHelpers/CodeEmitter/Registers.h"
|
||||
#include "Interface/Core/JIT/Arm64/JITClass.h"
|
||||
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
DEF_OP(VectorZero) {
|
||||
@@ -2293,40 +2295,47 @@ DEF_OP(VInsElement) {
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const uint32_t ElementSize = Op->Header.ElementSize;
|
||||
|
||||
const auto DestIdx = Op->DestIdx;
|
||||
const auto SrcIdx = Op->SrcIdx;
|
||||
const uint32_t DestIdx = Op->DestIdx;
|
||||
const uint32_t SrcIdx = Op->SrcIdx;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto SrcVector = GetVReg(Op->SrcVector.ID());
|
||||
auto Reg = GetVReg(Op->DestVector.ID());
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 1 || ElementSize == 2 || ElementSize == 4 || ElementSize == 8 || ElementSize == 16, "Invalid size");
|
||||
const auto SubRegSize =
|
||||
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i128Bit;
|
||||
|
||||
// We're going to use this to create our predicate register literal.
|
||||
// On an SVE 256-bit capable system, the predicate register will be
|
||||
// 32-bit in size. We want to set up only the element corresponding
|
||||
// to the destination index, since we're going to copy over the equivalent
|
||||
// indexed element from the source vector.
|
||||
auto Data = [ElementSize, DestIdx]() -> uint32_t {
|
||||
const auto Data = [ElementSize, DestIdx]() -> uint32_t {
|
||||
const auto Log2ElementSize = FEXCore::ilog2(ElementSize);
|
||||
|
||||
[[maybe_unused]] const auto MaxIndex = (32U >> Log2ElementSize) - 1;
|
||||
LOGMAN_THROW_AA_FMT(DestIdx <= MaxIndex, "DestIdx ({}) out of range. Must be within [0, {}]",
|
||||
DestIdx, MaxIndex);
|
||||
|
||||
const auto ShiftAmount = DestIdx << Log2ElementSize;
|
||||
|
||||
switch (ElementSize) {
|
||||
case 1:
|
||||
LOGMAN_THROW_AA_FMT(DestIdx <= 31, "DestIdx out of range: {}", DestIdx);
|
||||
return 1U << DestIdx;
|
||||
case 2:
|
||||
LOGMAN_THROW_AA_FMT(DestIdx <= 15, "DestIdx out of range: {}", DestIdx);
|
||||
return 1U << (DestIdx * 2);
|
||||
case 4:
|
||||
LOGMAN_THROW_AA_FMT(DestIdx <= 7, "DestIdx out of range: {}", DestIdx);
|
||||
return 1U << (DestIdx * 4);
|
||||
case 8:
|
||||
LOGMAN_THROW_AA_FMT(DestIdx <= 3, "DestIdx out of range: {}", DestIdx);
|
||||
return 1U << (DestIdx * 8);
|
||||
return 1U << ShiftAmount;
|
||||
case 16:
|
||||
LOGMAN_THROW_AA_FMT(DestIdx <= 1, "DestIdx out of range: {}", DestIdx);
|
||||
// Predicates can't be subdivided into the Q format, so we can just set up
|
||||
// the predicate to select the two adjacent doublewords.
|
||||
return 0x101U << (DestIdx * 16);
|
||||
return 0x101U << ShiftAmount;
|
||||
default:
|
||||
FEX_UNREACHABLE;
|
||||
return UINT32_MAX;
|
||||
@@ -2339,13 +2348,6 @@ DEF_OP(VInsElement) {
|
||||
adr(TMP1, &DataLocation);
|
||||
ldr(Predicate, TMP1);
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 1 || ElementSize == 2 || ElementSize == 4 || ElementSize == 8 || ElementSize == 16, "Invalid size");
|
||||
const auto SubRegSize =
|
||||
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i128Bit;
|
||||
|
||||
// Broadcast our source value across a temporary,
|
||||
// then combine with the destination.
|
||||
dup(SubRegSize, VTMP2.Z(), SrcVector.Z(), SrcIdx);
|
||||
@@ -2365,11 +2367,6 @@ DEF_OP(VInsElement) {
|
||||
Bind(&PastConstant);
|
||||
}
|
||||
else {
|
||||
if (Dst.Idx() != Reg.Idx()) {
|
||||
mov(VTMP1.Q(), Reg.Q());
|
||||
Reg = VTMP1;
|
||||
}
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 1 || ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize =
|
||||
ElementSize == 1 ? ARMEmitter::SubRegSize::i8Bit :
|
||||
@@ -2377,6 +2374,11 @@ DEF_OP(VInsElement) {
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (Dst.Idx() != Reg.Idx()) {
|
||||
mov(VTMP1.Q(), Reg.Q());
|
||||
Reg = VTMP1;
|
||||
}
|
||||
|
||||
ins(SubRegSize, Reg.Q(), DestIdx, SrcVector.Q(), SrcIdx);
|
||||
|
||||
if (Dst.Idx() != Reg.Idx()) {
|
||||
@@ -3025,6 +3027,37 @@ DEF_OP(VUABDL) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VUABDL2) {
|
||||
const auto Op = IROp->C<IR::IROp_VUABDL2>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize =
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ElementSize == 8 ? ARMEmitter::SubRegSize::i64Bit : ARMEmitter::SubRegSize::i8Bit;
|
||||
|
||||
if (HostSupportsSVE && Is256Bit) {
|
||||
// To mimic the behavior of AdvSIMD UABDL, we need to get the
|
||||
// absolute difference of the even elements (UADBLB), get the
|
||||
// absolute difference of the odd elemenets (UABDLT), then
|
||||
// interleave the results in both vectors together.
|
||||
|
||||
uabdlb(SubRegSize, VTMP1.Z(), Vector1.Z(), Vector2.Z());
|
||||
uabdlt(SubRegSize, VTMP2.Z(), Vector1.Z(), Vector2.Z());
|
||||
zip2(SubRegSize, Dst.Z(), VTMP1.Z(), VTMP2.Z());
|
||||
} else {
|
||||
uabdl2(SubRegSize, Dst.Q(), Vector1.Q(), Vector2.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VTBL1) {
|
||||
const auto Op = IROp->C<IR::IROp_VTBL1>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -161,6 +161,30 @@ DEF_OP(Neg) {
|
||||
neg(Dst);
|
||||
}
|
||||
|
||||
DEF_OP(Abs) {
|
||||
auto Op = IROp->C<IR::IROp_Abs>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
Xbyak::Reg Src;
|
||||
Xbyak::Reg Dst;
|
||||
switch (OpSize) {
|
||||
case 4:
|
||||
Src = GetSrc<RA_32>(Op->Src.ID());
|
||||
Dst = GetDst<RA_32>(Node);
|
||||
break;
|
||||
case 8:
|
||||
Src = GetSrc<RA_64>(Op->Src.ID());
|
||||
Dst = GetDst<RA_64>(Node);
|
||||
break;
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled Abs size: {}", OpSize);
|
||||
break;
|
||||
}
|
||||
mov(Dst, Src);
|
||||
mov(TMP1, Src);
|
||||
neg(Dst);
|
||||
cmovs(Dst, TMP1);
|
||||
}
|
||||
|
||||
DEF_OP(Mul) {
|
||||
auto Op = IROp->C<IR::IROp_Mul>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
@@ -1116,7 +1140,28 @@ DEF_OP(Select) {
|
||||
} else {
|
||||
cmp(GRCMP(Op->Cmp1.ID()), GRCMP(Op->Cmp2.ID()));
|
||||
}
|
||||
} else if (IsFPR(Op->Cmp1.ID())) {
|
||||
}
|
||||
else if (IsGPRPair(Op->Cmp1.ID())) {
|
||||
if (Op->CompareSize == 4) {
|
||||
const auto Src1 = GetSrcPair<RA_32>(Op->Cmp1.ID());
|
||||
const auto Src2 = GetSrcPair<RA_32>(Op->Cmp2.ID());
|
||||
mov (TMP1.cvt32(), Src1.first);
|
||||
mov (TMP2.cvt32(), Src1.second);
|
||||
xor_(TMP1.cvt32(), Src2.first);
|
||||
xor_(TMP2.cvt32(), Src2.second);
|
||||
or_(TMP1.cvt32(), TMP2.cvt32());
|
||||
}
|
||||
else {
|
||||
const auto Src1 = GetSrcPair<RA_64>(Op->Cmp1.ID());
|
||||
const auto Src2 = GetSrcPair<RA_64>(Op->Cmp2.ID());
|
||||
mov (TMP1, Src1.first);
|
||||
mov (TMP2, Src1.second);
|
||||
xor_(TMP1, Src2.first);
|
||||
xor_(TMP2, Src2.second);
|
||||
or_(TMP1, TMP2);
|
||||
}
|
||||
}
|
||||
else if (IsFPR(Op->Cmp1.ID())) {
|
||||
if (Op->CompareSize == 4)
|
||||
ucomiss(GetSrc(Op->Cmp1.ID()), GetSrc(Op->Cmp2.ID()));
|
||||
else
|
||||
@@ -1298,6 +1343,7 @@ void X86JITCore::RegisterALUHandlers() {
|
||||
REGISTER_OP(ADD, Add);
|
||||
REGISTER_OP(SUB, Sub);
|
||||
REGISTER_OP(NEG, Neg);
|
||||
REGISTER_OP(ABS, Abs);
|
||||
REGISTER_OP(MUL, Mul);
|
||||
REGISTER_OP(UMUL, UMul);
|
||||
REGISTER_OP(DIV, Div);
|
||||
|
||||
@@ -146,6 +146,7 @@ DEF_OP(Syscall) {
|
||||
auto Op = IROp->C<IR::IROp_Syscall>();
|
||||
// XXX: This is very terrible, but I don't care for right now
|
||||
|
||||
FEXCore::IR::SyscallFlags Flags = Op->Flags;
|
||||
auto NumPush = RA64.size();
|
||||
|
||||
for (auto &Reg : RA64)
|
||||
@@ -186,7 +187,11 @@ DEF_OP(Syscall) {
|
||||
for (uint32_t i = RA64.size(); i > 0; --i)
|
||||
pop(RA64[i - 1]);
|
||||
|
||||
mov (GetDst<RA_64>(Node), rax);
|
||||
if ((Flags & FEXCore::IR::SyscallFlags::NORETURNEDRESULT) != FEXCore::IR::SyscallFlags::NORETURNEDRESULT) {
|
||||
// Move result to its destination register.
|
||||
// Only if `NORETURNEDRESULT` wasn't set, otherwise we might overwrite the CPUState refilled with `FillStaticRegs`
|
||||
mov (GetDst<RA_64>(Node), rax);
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Thunk) {
|
||||
@@ -302,6 +307,42 @@ DEF_OP(CPUID) {
|
||||
mov(Dst.second, rdx);
|
||||
}
|
||||
|
||||
DEF_OP(XGETBV) {
|
||||
auto Op = IROp->C<IR::IROp_XGetBV>();
|
||||
|
||||
for (auto &Reg : RA64)
|
||||
push(Reg);
|
||||
|
||||
// CPUID ABI
|
||||
// this: rdi
|
||||
// Function: rsi
|
||||
//
|
||||
// Result: RAX, RDX. 4xi32
|
||||
|
||||
// rsi can be in the source registers, so copy argument to edx first
|
||||
mov (esi, GetSrc<RA_32>(Op->Function.ID()));
|
||||
mov (rdi, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.CPUIDObj)]);
|
||||
|
||||
auto NumPush = RA64.size();
|
||||
|
||||
if (NumPush & 1)
|
||||
sub(rsp, 8); // Align
|
||||
|
||||
// {rdi, rsi, rdx}
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.XCRFunction)]);
|
||||
|
||||
if (NumPush & 1)
|
||||
add(rsp, 8); // Align
|
||||
|
||||
for (uint32_t i = RA64.size(); i > 0; --i)
|
||||
pop(RA64[i - 1]);
|
||||
|
||||
auto Dst = GetSrcPair<RA_64>(Node);
|
||||
mov(Dst.first.cvt32(), eax);
|
||||
mov(Dst.second, rax);
|
||||
shr(Dst.second, 32);
|
||||
}
|
||||
|
||||
#undef DEF_OP
|
||||
void X86JITCore::RegisterBranchHandlers() {
|
||||
#define REGISTER_OP(op, x) OpHandlers[FEXCore::IR::IROps::OP_##op] = &X86JITCore::Op_##x
|
||||
@@ -314,6 +355,7 @@ void X86JITCore::RegisterBranchHandlers() {
|
||||
REGISTER_OP(VALIDATECODE, ValidateCode);
|
||||
REGISTER_OP(THREADREMOVECODEENTRY, ThreadRemoveCodeEntry);
|
||||
REGISTER_OP(CPUID, CPUID);
|
||||
REGISTER_OP(XGETBV, XGETBV);
|
||||
#undef REGISTER_OP
|
||||
}
|
||||
}
|
||||
+34
-7
@@ -308,16 +308,13 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
PushRegs();
|
||||
|
||||
const auto Op = IROp->C<IR::IROp_VPCMPESTRX>();
|
||||
const auto Is64Bit = Op->GPRSize == 8;
|
||||
const auto Control = Op->Control;
|
||||
|
||||
const auto LHS = GetSrc(Op->LHS.ID());
|
||||
const auto RHS = GetSrc(Op->RHS.ID());
|
||||
const auto SrcRAX = GetSrc<RA_64>(Op->RAX.ID());
|
||||
const auto SrcRDX = GetSrc<RA_64>(Op->RDX.ID());
|
||||
|
||||
// Encode the size check into the 8th bit to save a parameter
|
||||
const auto Control = Op->Control | (uint16_t(Is64Bit) << 8);
|
||||
|
||||
mov(rdi, SrcRAX);
|
||||
mov(rsi, SrcRDX);
|
||||
|
||||
@@ -387,7 +384,7 @@ static uint64_t X86JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame,
|
||||
}
|
||||
|
||||
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
|
||||
Context::ContextImpl::ThreadAddBlockLink(Thread, GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [record, LinkerAddress]{
|
||||
// undo the link
|
||||
record[0] = LinkerAddress;
|
||||
});
|
||||
@@ -444,6 +441,11 @@ X86JITCore::X86JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::Intern
|
||||
Common.CPUIDFunction = PMF.GetConvertedPointer();
|
||||
}
|
||||
|
||||
{
|
||||
FEXCore::Utils::MemberFunctionToPointerCast PMF(&FEXCore::CPUIDEmu::RunXCRFunction);
|
||||
Common.XCRFunction = PMF.GetConvertedPointer();
|
||||
}
|
||||
|
||||
Common.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
|
||||
Common.SyscallHandlerFunc = reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall);
|
||||
Common.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Context::ContextImpl::ThreadExitFunctionLink<X86JITCore_ExitFunctionLink>);
|
||||
@@ -482,11 +484,15 @@ IR::PhysicalRegister X86JITCore::GetPhys(IR::NodeID Node) const {
|
||||
}
|
||||
|
||||
bool X86JITCore::IsFPR(IR::NodeID Node) const {
|
||||
return RAData->GetNodeRegister(Node).Class == IR::FPRClass.Val;
|
||||
return RAData->GetNodeRegister(Node).Class == IR::FPRClass;
|
||||
}
|
||||
|
||||
bool X86JITCore::IsGPR(IR::NodeID Node) const {
|
||||
return RAData->GetNodeRegister(Node).Class == IR::GPRClass.Val;
|
||||
return RAData->GetNodeRegister(Node).Class == IR::GPRClass;
|
||||
}
|
||||
|
||||
bool X86JITCore::IsGPRPair(IR::NodeID Node) const {
|
||||
return RAData->GetNodeRegister(Node).Class == IR::GPRPairClass;
|
||||
}
|
||||
|
||||
template<uint8_t RAType>
|
||||
@@ -815,12 +821,33 @@ CPUBackend::CompiledCode X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]
|
||||
auto JITBlockTail = getCurr<JITCodeTail*>();
|
||||
setSize(getSize() + sizeof(JITCodeTail));
|
||||
|
||||
auto JITRIPEntriesLocation = getCurr<uint8_t *>();
|
||||
auto JITRIPEntries = getCurr<JITRIPReconstructEntries*>();
|
||||
|
||||
setSize(getSize() + sizeof(JITRIPReconstructEntries) * DebugData->GuestOpcodes.size());
|
||||
|
||||
// Put the block's RIP entry in the tail.
|
||||
// This will be used for RIP reconstruction in the future.
|
||||
// TODO: This needs to be a data RIP relocation once code caching works.
|
||||
// Current relocation code doesn't support this feature yet.
|
||||
JITBlockTail->RIP = Entry;
|
||||
|
||||
{
|
||||
// Store the RIP entries.
|
||||
JITBlockTail->NumberOfRIPEntries = DebugData->GuestOpcodes.size();
|
||||
JITBlockTail->OffsetToRIPEntries = JITRIPEntriesLocation - JITBlockTailLocation;
|
||||
uintptr_t CurrentRIPOffset = 0;
|
||||
uint64_t CurrentPCOffset = 0;
|
||||
for (size_t i = 0; i < DebugData->GuestOpcodes.size(); i++) {
|
||||
const auto &GuestOpcode = DebugData->GuestOpcodes[i];
|
||||
auto &RIPEntry = JITRIPEntries[i];
|
||||
RIPEntry.HostPCOffset = GuestOpcode.HostEntryOffset - CurrentPCOffset;
|
||||
RIPEntry.GuestRIPOffset = GuestOpcode.GuestEntryOffset - CurrentRIPOffset;
|
||||
CurrentPCOffset = GuestOpcode.HostEntryOffset;
|
||||
CurrentRIPOffset = GuestOpcode.GuestEntryOffset;
|
||||
}
|
||||
}
|
||||
|
||||
CodeHeader->OffsetToBlockTail = JITBlockTailLocation - CodeData.BlockBegin;
|
||||
|
||||
CodeData.Size = getCurr<uint8_t*>() - CodeData.BlockBegin;
|
||||
|
||||
@@ -169,6 +169,7 @@ private:
|
||||
|
||||
[[nodiscard]] bool IsFPR(IR::NodeID Node) const;
|
||||
[[nodiscard]] bool IsGPR(IR::NodeID Node) const;
|
||||
[[nodiscard]] bool IsGPRPair(IR::NodeID Node) const;
|
||||
|
||||
template<uint8_t RAType>
|
||||
[[nodiscard]] Xbyak::Reg GetSrc(IR::NodeID Node) const;
|
||||
@@ -245,6 +246,7 @@ private:
|
||||
DEF_OP(Add);
|
||||
DEF_OP(Sub);
|
||||
DEF_OP(Neg);
|
||||
DEF_OP(Abs);
|
||||
DEF_OP(Mul);
|
||||
DEF_OP(UMul);
|
||||
DEF_OP(Div);
|
||||
@@ -313,6 +315,7 @@ private:
|
||||
DEF_OP(ValidateCode);
|
||||
DEF_OP(ThreadRemoveCodeEntry);
|
||||
DEF_OP(CPUID);
|
||||
DEF_OP(XGETBV);
|
||||
|
||||
///< Conversion ops
|
||||
DEF_OP(VInsGPR);
|
||||
@@ -452,6 +455,7 @@ private:
|
||||
DEF_OP(VUMull2);
|
||||
DEF_OP(VSMull2);
|
||||
DEF_OP(VUABDL);
|
||||
DEF_OP(VUABDL2);
|
||||
DEF_OP(VTBL1);
|
||||
DEF_OP(VRev64);
|
||||
|
||||
|
||||
@@ -601,14 +601,22 @@ DEF_OP(LoadFlag) {
|
||||
auto Op = IROp->C<IR::IROp_LoadFlag>();
|
||||
|
||||
auto Dst = GetDst<RA_64>(Node);
|
||||
movzx(Dst, byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)]);
|
||||
|
||||
if (Op->Flag == 24 /* NZCV */)
|
||||
mov(Dst.cvt32(), dword [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)]);
|
||||
else
|
||||
movzx(Dst, byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)]);
|
||||
}
|
||||
|
||||
DEF_OP(StoreFlag) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
|
||||
mov (rax, GetSrc<RA_64>(Op->Value.ID()));
|
||||
mov(byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)], al);
|
||||
|
||||
if (Op->Flag == 24 /* NZCV */)
|
||||
mov(dword [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)], eax);
|
||||
else
|
||||
mov(byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)], al);
|
||||
}
|
||||
|
||||
Xbyak::RegExp X86JITCore::GenerateModRM(Xbyak::Reg Base, IR::OrderedNodeWrapper Offset, IR::MemOffsetType OffsetType, uint8_t OffsetScale) const {
|
||||
|
||||
@@ -4367,6 +4367,93 @@ DEF_OP(VUABDL) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VUABDL2) {
|
||||
const auto Op = IROp->C<IR::IROp_VUABDL2>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
const auto Dst = GetDst(Node);
|
||||
const auto Vector1 = GetSrc(Op->Vector1.ID());
|
||||
const auto Vector2 = GetSrc(Op->Vector2.ID());
|
||||
|
||||
switch (ElementSize) {
|
||||
case 2: {
|
||||
vpxor(xmm15, xmm15, xmm15);
|
||||
vpxor(xmm14, xmm14, xmm14);
|
||||
|
||||
if (Is256Bit) {
|
||||
vextracti128(xmm15, ToYMM(Vector1), 1);
|
||||
vextracti128(xmm14, ToYMM(Vector2), 1);
|
||||
|
||||
vpxor(xmm12, xmm12, xmm12);
|
||||
vpxor(xmm13, xmm13, xmm13);
|
||||
vpxor(Dst, Dst, Dst);
|
||||
|
||||
// Bottom half
|
||||
vpunpcklbw(xmm13, xmm15, xmm13);
|
||||
vpunpcklbw(xmm12, xmm14, xmm12);
|
||||
|
||||
// Top half
|
||||
vpunpckhbw(xmm15, xmm15, Dst);
|
||||
vpunpckhbw(xmm14, xmm14, Dst);
|
||||
|
||||
// Reinsert
|
||||
vinserti128(ymm13, ymm13, xmm15, 1);
|
||||
vinserti128(ymm12, ymm12, xmm14, 1);
|
||||
|
||||
vpsubw(ToYMM(Dst), ymm12, ymm13);
|
||||
vpabsw(ToYMM(Dst), ToYMM(Dst));
|
||||
} else {
|
||||
vpunpckhbw(xmm15, Vector1, xmm15);
|
||||
vpunpckhbw(xmm14, Vector2, xmm14);
|
||||
vpsubw(Dst, xmm14, xmm15);
|
||||
vpabsw(Dst, Dst);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
vpxor(xmm15, xmm15, xmm15);
|
||||
vpxor(xmm14, xmm14, xmm14);
|
||||
|
||||
if (Is256Bit) {
|
||||
vextracti128(xmm15, ToYMM(Vector1), 1);
|
||||
vextracti128(xmm14, ToYMM(Vector2), 1);
|
||||
|
||||
vpxor(xmm12, xmm12, xmm12);
|
||||
vpxor(xmm13, xmm13, xmm13);
|
||||
vpxor(Dst, Dst, Dst);
|
||||
|
||||
// Bottom half
|
||||
vpunpcklwd(xmm13, xmm15, xmm13);
|
||||
vpunpcklwd(xmm12, xmm14, xmm12);
|
||||
|
||||
// Top half
|
||||
vpunpckhwd(xmm15, xmm15, Dst);
|
||||
vpunpckhwd(xmm14, xmm14, Dst);
|
||||
|
||||
// Reinsert
|
||||
vinserti128(ymm13, ymm13, xmm15, 1);
|
||||
vinserti128(ymm12, ymm12, xmm14, 1);
|
||||
|
||||
vpsubd(ToYMM(Dst), ymm12, ymm13);
|
||||
vpabsd(ToYMM(Dst), ToYMM(Dst));
|
||||
} else {
|
||||
vpunpckhwd(xmm15, Vector1, xmm15);
|
||||
vpunpckhwd(xmm14, Vector2, xmm14);
|
||||
vpsubd(Dst, xmm14, xmm15);
|
||||
vpabsd(Dst, Dst);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Element Size: {}", ElementSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
DEF_OP(VTBL1) {
|
||||
const auto Op = IROp->C<IR::IROp_VTBL1>();
|
||||
const auto OpSize = IROp->Size;
|
||||
@@ -4609,6 +4696,7 @@ void X86JITCore::RegisterVectorHandlers() {
|
||||
REGISTER_OP(VUMULL2, VUMull2);
|
||||
REGISTER_OP(VSMULL2, VSMull2);
|
||||
REGISTER_OP(VUABDL, VUABDL);
|
||||
REGISTER_OP(VUABDL2, VUABDL2);
|
||||
REGISTER_OP(VTBL1, VTBL1);
|
||||
REGISTER_OP(VREV64, VRev64);
|
||||
#undef REGISTER_OP
|
||||
|
||||
+6
-1
@@ -22,6 +22,9 @@ namespace FEXCore::CodeSerialize {
|
||||
// Multiblock enabled
|
||||
unsigned MultiBlock : 1;
|
||||
|
||||
// Hardware TSO enabled
|
||||
unsigned HardwareTSOEnabled : 1;
|
||||
|
||||
// TSO enabled
|
||||
unsigned TSOEnabled : 1;
|
||||
|
||||
@@ -48,13 +51,14 @@ namespace FEXCore::CodeSerialize {
|
||||
|
||||
// Padding to remove uninitialized data warning from asan
|
||||
// Shows remaining amount of bits available for config
|
||||
unsigned _Pad : 18;
|
||||
unsigned _Pad : 17;
|
||||
|
||||
bool operator==(CodeObjectSerializationConfig const &other) const {
|
||||
return Cookie == other.Cookie &&
|
||||
MaxInstPerBlock == other.MaxInstPerBlock &&
|
||||
Arch == other.Arch &&
|
||||
MultiBlock == other.MultiBlock &&
|
||||
HardwareTSOEnabled == other.HardwareTSOEnabled &&
|
||||
TSOEnabled == other.TSOEnabled &&
|
||||
ABILocalFlags == other.ABILocalFlags &&
|
||||
ABINoPF == other.ABINoPF &&
|
||||
@@ -71,6 +75,7 @@ namespace FEXCore::CodeSerialize {
|
||||
Hash <<= 32; Hash |= other.MaxInstPerBlock;
|
||||
Hash <<= 1; Hash |= other.Arch;
|
||||
Hash <<= 1; Hash |= other.MultiBlock;
|
||||
Hash <<= 1; Hash |= other.HardwareTSOEnabled;
|
||||
Hash <<= 1; Hash |= other.TSOEnabled;
|
||||
Hash <<= 1; Hash |= other.ABILocalFlags;
|
||||
Hash <<= 1; Hash |= other.ABINoPF;
|
||||
|
||||
+275
-171
File diff suppressed because it is too large.
Load diff
+264
-68
@@ -28,13 +28,6 @@ class OpDispatchBuilder final : public IREmitter {
|
||||
friend class FEXCore::IR::Pass;
|
||||
friend class FEXCore::IR::PassManager;
|
||||
|
||||
enum class SelectionFlag {
|
||||
Nothing, // must rely on x86 flags
|
||||
CMP, // flags were set by a CMP between flagsOpDest/flagsOpDestSigned and flagsOpSrc/flagsOpSrcSigned with flagsOpSize size
|
||||
AND, // flags were set by an AND/TEST, flagsOpDest contains the resulting value of flagsOpSize size
|
||||
FCMP, // flags were set by a ucomis* / comis*
|
||||
};
|
||||
|
||||
public:
|
||||
enum class FlagsGenerationType : uint8_t {
|
||||
TYPE_NONE,
|
||||
@@ -49,6 +42,7 @@ public:
|
||||
TYPE_LSHLI,
|
||||
TYPE_LSHR,
|
||||
TYPE_LSHRI,
|
||||
TYPE_LSHRDI,
|
||||
TYPE_ASHR,
|
||||
TYPE_ASHRI,
|
||||
TYPE_ROR,
|
||||
@@ -68,25 +62,6 @@ public:
|
||||
TYPE_RDRAND,
|
||||
};
|
||||
|
||||
SelectionFlag flagsOp{};
|
||||
uint8_t flagsOpSize{};
|
||||
OrderedNode* flagsOpDest{};
|
||||
OrderedNode* flagsOpSrc{};
|
||||
OrderedNode* flagsOpDestSigned{};
|
||||
OrderedNode* flagsOpSrcSigned{};
|
||||
|
||||
FEXCore::Context::ContextImpl *CTX{};
|
||||
|
||||
// Used during new op bringup
|
||||
bool ShouldDump {false};
|
||||
|
||||
struct JumpTargetInfo {
|
||||
OrderedNode* BlockEntry;
|
||||
bool HaveEmitted;
|
||||
};
|
||||
|
||||
fextl::map<uint64_t, JumpTargetInfo> JumpTargets;
|
||||
|
||||
OrderedNode* GetNewJumpBlock(uint64_t RIP) {
|
||||
auto it = JumpTargets.find(RIP);
|
||||
LOGMAN_THROW_A_FMT(it != JumpTargets.end(), "Couldn't find block generated for 0x{:x}", RIP);
|
||||
@@ -108,6 +83,10 @@ public:
|
||||
|
||||
void StartNewBlock() {
|
||||
flagsOp = SelectionFlag::Nothing;
|
||||
|
||||
// If we loaded flags but didn't change them, invalidate the cached copy and move on.
|
||||
// Changes get stored out by CalculateDeferredFlags.
|
||||
CachedNZCV = nullptr;
|
||||
}
|
||||
|
||||
bool FinishOp(uint64_t NextRIP, bool LastOp) {
|
||||
@@ -149,6 +128,31 @@ public:
|
||||
return false;
|
||||
}
|
||||
|
||||
static bool CanHaveSideEffects(FEXCore::X86Tables::X86InstInfo const* TableInfo, FEXCore::X86Tables::DecodedOp Op) {
|
||||
if (TableInfo && TableInfo->Flags & X86Tables::InstFlags::FLAGS_DEBUG_MEM_ACCESS) {
|
||||
// If it is marked as having memory access then always say it has a side-effect.
|
||||
// Not always true but better to be safe.
|
||||
return true;
|
||||
}
|
||||
|
||||
auto CanHaveSideEffects = false;
|
||||
|
||||
auto HasPotentialMemoryAccess = [](X86Tables::DecodedOperand const &Operand) -> bool {
|
||||
if (Operand.IsNone()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// This isn't guaranteed that all of these types will access memory, but be safe.
|
||||
return Operand.IsGPRDirect() || Operand.IsGPRIndirect() || Operand.IsRIPRelative() || Operand.IsSIB();
|
||||
};
|
||||
|
||||
CanHaveSideEffects |= HasPotentialMemoryAccess(Op->Dest);
|
||||
CanHaveSideEffects |= HasPotentialMemoryAccess(Op->Src[0]);
|
||||
CanHaveSideEffects |= HasPotentialMemoryAccess(Op->Src[1]);
|
||||
CanHaveSideEffects |= HasPotentialMemoryAccess(Op->Src[2]);
|
||||
return CanHaveSideEffects;
|
||||
}
|
||||
|
||||
OpDispatchBuilder(FEXCore::Context::ContextImpl *ctx);
|
||||
OpDispatchBuilder(FEXCore::Utils::IntrusivePooledAllocator &Allocator);
|
||||
|
||||
@@ -157,6 +161,12 @@ public:
|
||||
bool HadDecodeFailure() const { return DecodeFailure; }
|
||||
bool NeedsBlockEnder() const { return NeedsBlockEnd; }
|
||||
|
||||
void ResetHandledLock() { HandledLock = false; }
|
||||
bool HasHandledLock() const { return HandledLock; }
|
||||
|
||||
void SetDumpIR(bool DumpIR) { ShouldDump = DumpIR; }
|
||||
bool ShouldDumpIR() const { return ShouldDump; }
|
||||
|
||||
void BeginFunction(uint64_t RIP, fextl::vector<FEXCore::Frontend::Decoder::DecodedBlocks> const *Blocks);
|
||||
void Finalize();
|
||||
|
||||
@@ -219,6 +229,7 @@ public:
|
||||
void MOVOffsetOp(OpcodeArgs);
|
||||
void CMOVOp(OpcodeArgs);
|
||||
void CPUIDOp(OpcodeArgs);
|
||||
void XGetBVOp(OpcodeArgs);
|
||||
template<bool SHL1Bit>
|
||||
void SHLOp(OpcodeArgs);
|
||||
void SHLImmediateOp(OpcodeArgs);
|
||||
@@ -491,7 +502,9 @@ public:
|
||||
void VPALIGNROp(OpcodeArgs);
|
||||
|
||||
void VPCMPESTRIOp(OpcodeArgs);
|
||||
void VPCMPESTRMOp(OpcodeArgs);
|
||||
void VPCMPISTRIOp(OpcodeArgs);
|
||||
void VPCMPISTRMOp(OpcodeArgs);
|
||||
|
||||
void VPERM2Op(OpcodeArgs);
|
||||
void VPERMDOp(OpcodeArgs);
|
||||
@@ -696,6 +709,8 @@ public:
|
||||
void FXSaveOp(OpcodeArgs);
|
||||
void FXRStoreOp(OpcodeArgs);
|
||||
|
||||
void XSaveOp(OpcodeArgs);
|
||||
|
||||
void PAlignrOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void UCOMISxOp(OpcodeArgs);
|
||||
@@ -745,11 +760,9 @@ public:
|
||||
void PHADDS(OpcodeArgs);
|
||||
void PHSUBS(OpcodeArgs);
|
||||
|
||||
template<uint8_t FenceType>
|
||||
void FenceOp(OpcodeArgs);
|
||||
|
||||
void CLWB(OpcodeArgs);
|
||||
void CLFLUSHOPT(OpcodeArgs);
|
||||
void LoadFenceOrXRSTOR(OpcodeArgs);
|
||||
void MemFenceOrXSAVEOPT(OpcodeArgs);
|
||||
void StoreFenceOrCLFlush(OpcodeArgs);
|
||||
void CLZeroOp(OpcodeArgs);
|
||||
@@ -799,16 +812,54 @@ public:
|
||||
void InvalidOp(OpcodeArgs);
|
||||
|
||||
void SetPackedRFLAG(bool Lower8, OrderedNode *Src);
|
||||
OrderedNode *GetPackedRFLAG(bool Lower8);
|
||||
OrderedNode *GetPackedRFLAG(uint32_t FlagsMask = ~0U);
|
||||
|
||||
void SetMultiblock(bool _Multiblock) { Multiblock = _Multiblock; }
|
||||
|
||||
bool HandledLock = false;
|
||||
private:
|
||||
enum class SelectionFlag {
|
||||
Nothing, // must rely on x86 flags
|
||||
CMP, // flags were set by a CMP between flagsOpDest/flagsOpDestSigned and flagsOpSrc/flagsOpSrcSigned with flagsOpSize size
|
||||
AND, // flags were set by an AND/TEST, flagsOpDest contains the resulting value of flagsOpSize size
|
||||
FCMP, // flags were set by a ucomis* / comis*
|
||||
};
|
||||
|
||||
struct JumpTargetInfo {
|
||||
OrderedNode* BlockEntry;
|
||||
bool HaveEmitted;
|
||||
};
|
||||
|
||||
FEXCore::Context::ContextImpl *CTX{};
|
||||
|
||||
SelectionFlag flagsOp{};
|
||||
uint8_t flagsOpSize{};
|
||||
OrderedNode* flagsOpDest{};
|
||||
OrderedNode* flagsOpSrc{};
|
||||
OrderedNode* flagsOpDestSigned{};
|
||||
OrderedNode* flagsOpSrcSigned{};
|
||||
|
||||
static bool IsNZCV(unsigned BitOffset) {
|
||||
switch (BitOffset) {
|
||||
case FEXCore::X86State::RFLAG_CF_LOC:
|
||||
case FEXCore::X86State::RFLAG_ZF_LOC:
|
||||
case FEXCore::X86State::RFLAG_SF_LOC:
|
||||
case FEXCore::X86State::RFLAG_OF_LOC:
|
||||
return true;
|
||||
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode* CachedNZCV = {};
|
||||
uint32_t PossiblySetNZCVBits = 0;
|
||||
|
||||
fextl::map<uint64_t, JumpTargetInfo> JumpTargets;
|
||||
bool HandledLock{false};
|
||||
bool DecodeFailure{false};
|
||||
bool NeedsBlockEnd{false};
|
||||
FEXCore::IR::IROp_IRHeader *Current_Header{};
|
||||
OrderedNode *Current_HeaderNode{};
|
||||
// Used during new op bringup
|
||||
bool ShouldDump{false};
|
||||
|
||||
void ALUOpImpl(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp, bool RequiresMask);
|
||||
|
||||
@@ -862,7 +913,7 @@ private:
|
||||
const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm);
|
||||
|
||||
void PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit);
|
||||
void PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask);
|
||||
|
||||
OrderedNode* PHADDSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1,
|
||||
const X86Tables::DecodedOperand& Src2);
|
||||
@@ -950,6 +1001,23 @@ private:
|
||||
|
||||
OrderedNode* Vector_CVT_Int_To_FloatImpl(OpcodeArgs, size_t SrcElementSize, bool Widen);
|
||||
|
||||
void XSaveOpImpl(OpcodeArgs);
|
||||
void SaveX87State(OpcodeArgs, OrderedNode *MemBase);
|
||||
void SaveSSEState(OrderedNode *MemBase);
|
||||
void SaveMXCSRState(OrderedNode *MemBase);
|
||||
void SaveAVXState(OrderedNode *MemBase);
|
||||
|
||||
void XRstorOpImpl(OpcodeArgs);
|
||||
void RestoreX87State(OrderedNode *MemBase);
|
||||
void RestoreSSEState(OrderedNode *MemBase);
|
||||
void RestoreMXCSRState(OrderedNode *MXCSR);
|
||||
void RestoreAVXState(OrderedNode *MemBase);
|
||||
void DefaultX87State(OpcodeArgs);
|
||||
void DefaultSSEState();
|
||||
void DefaultAVXState();
|
||||
|
||||
OrderedNode *GetMXCSR();
|
||||
|
||||
#undef OpcodeArgs
|
||||
|
||||
OrderedNode *AppendSegmentOffset(OrderedNode *Value, uint32_t Flags, uint32_t DefaultPrefix = 0, bool Override = false);
|
||||
@@ -993,19 +1061,123 @@ private:
|
||||
[[nodiscard]] uint32_t GetDstBitSize(X86Tables::DecodedOp Op) const;
|
||||
[[nodiscard]] uint32_t GetSrcBitSize(X86Tables::DecodedOp Op) const;
|
||||
|
||||
static inline constexpr unsigned IndexNZCV(unsigned BitOffset) {
|
||||
switch (BitOffset) {
|
||||
case FEXCore::X86State::RFLAG_OF_LOC: return 28;
|
||||
case FEXCore::X86State::RFLAG_CF_LOC: return 29;
|
||||
case FEXCore::X86State::RFLAG_ZF_LOC: return 30;
|
||||
case FEXCore::X86State::RFLAG_SF_LOC: return 31;
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode *GetNZCV() {
|
||||
if (!CachedNZCV) {
|
||||
CachedNZCV = _LoadFlag(FEXCore::X86State::RFLAG_NZCV_LOC);
|
||||
|
||||
// We don't know what's set
|
||||
PossiblySetNZCVBits = ~0;
|
||||
}
|
||||
|
||||
return CachedNZCV;
|
||||
}
|
||||
|
||||
void SetNZCV(OrderedNode *Value) {
|
||||
CachedNZCV = Value;
|
||||
}
|
||||
|
||||
void ZeroNZCV() {
|
||||
CachedNZCV = _Constant(0);
|
||||
PossiblySetNZCVBits = 0;
|
||||
}
|
||||
|
||||
void ZeroCV() {
|
||||
// Get old NZCV before we mess with PossiblySetNZCVBits
|
||||
auto OldNZCV = GetNZCV();
|
||||
|
||||
// Mask out the NZ bits, clearing CV. Even if the code sets CV after, this can end up faster
|
||||
// moves by allowing orlshl to be used instead of bfi.
|
||||
PossiblySetNZCVBits = (1u << IndexNZCV(FEXCore::X86State::RFLAG_SF_LOC)) |
|
||||
(1u << IndexNZCV(FEXCore::X86State::RFLAG_ZF_LOC));
|
||||
SetNZCV(_And(OldNZCV, _Constant(PossiblySetNZCVBits)));
|
||||
}
|
||||
|
||||
void SetN_ZeroZCV(unsigned SrcSize, OrderedNode *Res) {
|
||||
static_assert(IndexNZCV(FEXCore::X86State::RFLAG_SF_LOC) == 31);
|
||||
|
||||
unsigned NBit = 31;
|
||||
unsigned SignBit = (SrcSize * 8) - 1;
|
||||
|
||||
OrderedNode *Shifted;
|
||||
|
||||
// Shift the sign bit into the N bit
|
||||
if (SignBit > NBit)
|
||||
Shifted = _Ashr(Res, _Constant(SignBit - NBit));
|
||||
else if (SignBit < NBit)
|
||||
Shifted = _Lshl(Res, _Constant(NBit - SignBit));
|
||||
else
|
||||
Shifted = Res;
|
||||
|
||||
// Mask off just the N bit, which now equals the sign bit
|
||||
CachedNZCV = _And(Shifted, _Constant(1u << NBit));
|
||||
PossiblySetNZCVBits = (1u << NBit);
|
||||
}
|
||||
|
||||
void SetNZ_ZeroCV(unsigned SrcSize, OrderedNode *Res) {
|
||||
// The TestNZ opcode does this operation natively for 32-bit or 64-bit.
|
||||
// Otherwise we can implement the functionality ourselves with some bit math.
|
||||
if (CTX->BackendFeatures.SupportsFlags && SrcSize >= 4) {
|
||||
CachedNZCV = _TestNZ(SrcSize, Res);
|
||||
PossiblySetNZCVBits = (1u << 31) | (1u << 30);
|
||||
} else {
|
||||
// N
|
||||
SetN_ZeroZCV(SrcSize, Res);
|
||||
|
||||
// Z
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
auto SelectOp = _Select(FEXCore::IR::COND_EQ, Res, Zero, One, Zero);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(SelectOp);
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode *InsertNZCV(OrderedNode *NZCV, unsigned BitOffset, OrderedNode *Value) {
|
||||
unsigned Bit = IndexNZCV(BitOffset);
|
||||
|
||||
uint32_t SetBits = PossiblySetNZCVBits;
|
||||
PossiblySetNZCVBits |= (1u << Bit);
|
||||
|
||||
if (SetBits == 0)
|
||||
return _Lshl(Value, _Constant(Bit));
|
||||
else if (CTX->BackendFeatures.SupportsShiftedBitwise && (SetBits & (1u << Bit)) == 0)
|
||||
return _Orlshl(NZCV, Value, Bit);
|
||||
else
|
||||
return _Bfi(4, 1, Bit, NZCV, Value);
|
||||
}
|
||||
|
||||
template<unsigned BitOffset>
|
||||
void SetRFLAG(OrderedNode *Value) {
|
||||
flagsOp = SelectionFlag::Nothing;
|
||||
_StoreFlag(_Bfe(1, 0, Value), BitOffset);
|
||||
SetRFLAG(Value, BitOffset);
|
||||
}
|
||||
|
||||
void SetRFLAG(OrderedNode *Value, unsigned BitOffset) {
|
||||
flagsOp = SelectionFlag::Nothing;
|
||||
_StoreFlag(_Bfe(1, 0, Value), BitOffset);
|
||||
|
||||
if (IsNZCV(BitOffset))
|
||||
SetNZCV(InsertNZCV(GetNZCV(), BitOffset, Value));
|
||||
else
|
||||
_StoreFlag(Value, BitOffset);
|
||||
}
|
||||
|
||||
OrderedNode *GetRFLAG(unsigned BitOffset) {
|
||||
return _LoadFlag(BitOffset);
|
||||
if (IsNZCV(BitOffset)) {
|
||||
if (!CachedNZCV || (PossiblySetNZCVBits & (1u << IndexNZCV(BitOffset))))
|
||||
return _Bfe(1, 1, IndexNZCV(BitOffset), GetNZCV());
|
||||
else
|
||||
return _Constant(0);
|
||||
} else {
|
||||
return _LoadFlag(BitOffset);
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode *SelectCC(uint8_t OP, OrderedNode *TrueValue, OrderedNode *FalseValue);
|
||||
@@ -1110,34 +1282,41 @@ private:
|
||||
/**
|
||||
* @name These functions are used by the deferred flag handling while it is calculating and storing flags in to RFLAGs.
|
||||
* @{ */
|
||||
void CalculcateFlags_ADC(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF);
|
||||
void CalculcateFlags_SBB(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF);
|
||||
void CalculcateFlags_SUB(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true);
|
||||
void CalculcateFlags_ADD(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true);
|
||||
void CalculcateFlags_MUL(uint8_t SrcSize, OrderedNode *Res, OrderedNode *High);
|
||||
void CalculcateFlags_UMUL(OrderedNode *High);
|
||||
void CalculcateFlags_Logical(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_ShiftLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_ShiftLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculcateFlags_ShiftRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_ShiftRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculcateFlags_SignShiftRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_SignShiftRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculcateFlags_RotateRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_RotateLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_RotateRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculcateFlags_RotateLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculcateFlags_FCMP(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculcateFlags_BEXTR(OrderedNode *Src);
|
||||
void CalculcateFlags_BLSI(uint8_t SrcSize, OrderedNode *Src);
|
||||
void CalculcateFlags_BLSMSK(OrderedNode *Src);
|
||||
void CalculcateFlags_BLSR(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src);
|
||||
void CalculcateFlags_POPCOUNT(OrderedNode *Src);
|
||||
void CalculcateFlags_BZHI(uint8_t SrcSize, OrderedNode *Result, OrderedNode *Src);
|
||||
void CalculcateFlags_TZCNT(OrderedNode *Src);
|
||||
void CalculcateFlags_LZCNT(uint8_t SrcSize, OrderedNode *Src);
|
||||
void CalculcateFlags_BITSELECT(OrderedNode *Src);
|
||||
void CalculcateFlags_RDRAND(OrderedNode *Src);
|
||||
OrderedNode *LoadPF();
|
||||
void CalculatePFUncheckedABI(OrderedNode *Res, OrderedNode *condition = nullptr);
|
||||
void CalculatePF(OrderedNode *Res, OrderedNode *condition = nullptr);
|
||||
|
||||
void CalculateOF_Add(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculateFlags_ADC(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF);
|
||||
void CalculateFlags_SBB(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF);
|
||||
void CalculateFlags_SUB(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true);
|
||||
void CalculateFlags_ADD(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF = true);
|
||||
void CalculateFlags_MUL(uint8_t SrcSize, OrderedNode *Res, OrderedNode *High);
|
||||
void CalculateFlags_UMUL(OrderedNode *High);
|
||||
void CalculateFlags_Logical(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculateFlags_ShiftLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculateFlags_ShiftRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRightDoubleImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculateFlags_SignShiftRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculateFlags_RotateRight(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculateFlags_RotateLeft(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculateFlags_RotateRightImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculateFlags_RotateLeftImmediate(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift);
|
||||
void CalculateFlags_FCMP(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
|
||||
void CalculateFlags_BEXTR(OrderedNode *Src);
|
||||
void CalculateFlags_BLSI(uint8_t SrcSize, OrderedNode *Src);
|
||||
void CalculateFlags_BLSMSK(OrderedNode *Src);
|
||||
void CalculateFlags_BLSR(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src);
|
||||
void CalculateFlags_POPCOUNT(OrderedNode *Src);
|
||||
void CalculateFlags_BZHI(uint8_t SrcSize, OrderedNode *Result, OrderedNode *Src);
|
||||
void CalculateFlags_TZCNT(OrderedNode *Src);
|
||||
void CalculateFlags_LZCNT(uint8_t SrcSize, OrderedNode *Src);
|
||||
void CalculateFlags_BITSELECT(OrderedNode *Src);
|
||||
void CalculateFlags_RDRAND(OrderedNode *Src);
|
||||
/** @} */
|
||||
|
||||
/**
|
||||
@@ -1350,6 +1529,23 @@ private:
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_ShiftRightDoubleImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) {
|
||||
// No flags changed if shift is zero.
|
||||
if (Shift == 0) return;
|
||||
|
||||
CurrentDeferredFlags = DeferredFlagData {
|
||||
.Type = FlagsGenerationType::TYPE_LSHRDI,
|
||||
.SrcSize = GetSrcSize(Op),
|
||||
.Res = Res,
|
||||
.Sources = {
|
||||
.OneSrcImmediate = {
|
||||
.Src1 = Src1,
|
||||
.Imm = Shift,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
void GenerateFlags_RotateRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
// Doesn't set all the flags, needs to calculate.
|
||||
CalculateDeferredFlags();
|
||||
@@ -1555,14 +1751,14 @@ private:
|
||||
uint64_t Entry;
|
||||
|
||||
OrderedNode* _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *Addr, OrderedNode *Value, uint8_t Align = 1) {
|
||||
if (CTX->IsTSOEnabled())
|
||||
if (CTX->IsAtomicTSOEnabled())
|
||||
return _StoreMemTSO(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
else
|
||||
return _StoreMem(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
OrderedNode* _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, OrderedNode *ssa0, uint8_t Align = 1) {
|
||||
if (CTX->IsTSOEnabled())
|
||||
if (CTX->IsAtomicTSOEnabled())
|
||||
return _LoadMemTSO(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
else
|
||||
return _LoadMem(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
|
||||
+407
-550
File diff suppressed because it is too large.
Load diff
+342
-158
@@ -204,8 +204,8 @@ void OpDispatchBuilder::VMOVSLDUPOp(OpcodeArgs) {
|
||||
void OpDispatchBuilder::MOVSSOp(OpcodeArgs) {
|
||||
if (Op->Dest.IsGPR() && Op->Src[0].IsGPR()) {
|
||||
// MOVSS xmm1, xmm2
|
||||
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, 16, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 4, Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
auto Result = _VInsElement(16, 4, 0, 0, Dest, Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
@@ -217,7 +217,7 @@ void OpDispatchBuilder::MOVSSOp(OpcodeArgs) {
|
||||
}
|
||||
else {
|
||||
// MOVSS mem32, xmm1
|
||||
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 4, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, 4, -1);
|
||||
}
|
||||
}
|
||||
@@ -2257,40 +2257,26 @@ template
|
||||
void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<8, true>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::MASKMOVOp(OpcodeArgs) {
|
||||
// Until we get correct PHI nodes this is required to be a loop unroll
|
||||
const auto Size = uint32_t{GetSrcSize(Op)} * 8;
|
||||
const auto Size = GetSrcSize(Op);
|
||||
|
||||
OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *MaskSrc = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
// Mask only cares about the top bit of each byte
|
||||
MaskSrc = _VCMPLTZ(Size, 1, MaskSrc);
|
||||
|
||||
// Vector that will overwrite byte elements.
|
||||
OrderedNode *VectorSrc = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
|
||||
// RDI source
|
||||
auto MemDest = LoadGPRRegister(X86State::REG_RDI);
|
||||
|
||||
const size_t NumElements = Size / 64;
|
||||
for (size_t Element = 0; Element < NumElements; ++Element) {
|
||||
// Extract the current element
|
||||
auto SrcElement = _VExtractToGPR(GetSrcSize(Op), 8, Src, Element);
|
||||
auto DestElement = _VExtractToGPR(GetSrcSize(Op), 8, Dest, Element);
|
||||
// DS prefix by default.
|
||||
MemDest = AppendSegmentOffset(MemDest, Op->Flags, FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX);
|
||||
|
||||
constexpr size_t NumSelectBits = 64 / 8;
|
||||
for (size_t Select = 0; Select < NumSelectBits; ++Select) {
|
||||
auto SelectMask = _Bfe(1, 8 * Select + 7, SrcElement);
|
||||
auto CondJump = _CondJump(SelectMask, {COND_EQ});
|
||||
auto StoreBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
SetFalseJumpTarget(CondJump, StoreBlock);
|
||||
SetCurrentCodeBlock(StoreBlock);
|
||||
{
|
||||
auto DestByte = _Bfe(8, 8 * Select, DestElement);
|
||||
auto MemLocation = _Add(MemDest, _Constant(Element * 8 + Select));
|
||||
// MASKMOVDQU/MASKMOVQ is explicitly weakly-ordered on its store
|
||||
_StoreMem(GPRClass, 1, MemLocation, DestByte, 1);
|
||||
}
|
||||
auto Jump = _Jump();
|
||||
auto NextJumpTarget = CreateNewCodeBlockAfter(StoreBlock);
|
||||
SetJumpTarget(Jump, NextJumpTarget);
|
||||
SetTrueJumpTarget(CondJump, NextJumpTarget);
|
||||
SetCurrentCodeBlock(NextJumpTarget);
|
||||
}
|
||||
}
|
||||
OrderedNode *XMMReg = _LoadMem(FPRClass, Size, MemDest, 1);
|
||||
|
||||
// If the Mask element high bit is set then overwrite the element with the source, else keep the memory variant
|
||||
XMMReg = _VBSL(Size, MaskSrc, VectorSrc, XMMReg);
|
||||
_StoreMem(FPRClass, Size, MemDest, XMMReg, 1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VMASKMOVOpImpl(OpcodeArgs, size_t ElementSize, size_t DataSize, bool IsStore,
|
||||
@@ -2455,6 +2441,76 @@ void OpDispatchBuilder::FXSaveOp(OpcodeArgs) {
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false);
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
SaveX87State(Op, Mem);
|
||||
SaveSSEState(Mem);
|
||||
SaveMXCSRState(Mem);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::XSaveOp(OpcodeArgs) {
|
||||
XSaveOpImpl(Op);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::XSaveOpImpl(OpcodeArgs) {
|
||||
const auto XSaveBase = [this, Op] {
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false);
|
||||
return AppendSegmentOffset(Mem, Op->Flags);
|
||||
};
|
||||
|
||||
// NOTE: Mask should be EAX and EDX concatenated, but we only need to test
|
||||
// for features that are in the lower 32 bits, so EAX only is sufficient.
|
||||
OrderedNode *Mask = LoadGPRRegister(X86State::REG_RAX);
|
||||
OrderedNode *Base = XSaveBase();
|
||||
|
||||
const auto StoreIfFlagSet = [&](uint32_t BitIndex, auto fn, uint32_t FieldSize = 1){
|
||||
OrderedNode *BitFlag = _Bfe(FieldSize, BitIndex, Mask);
|
||||
auto CondJump = _CondJump(BitFlag, {COND_NEQ});
|
||||
|
||||
auto StoreBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
SetTrueJumpTarget(CondJump, StoreBlock);
|
||||
SetCurrentCodeBlock(StoreBlock);
|
||||
{
|
||||
fn();
|
||||
}
|
||||
auto Jump = _Jump();
|
||||
auto NextJumpTarget = CreateNewCodeBlockAfter(StoreBlock);
|
||||
SetJumpTarget(Jump, NextJumpTarget);
|
||||
SetFalseJumpTarget(CondJump, NextJumpTarget);
|
||||
SetCurrentCodeBlock(NextJumpTarget);
|
||||
};
|
||||
|
||||
// x87
|
||||
{
|
||||
StoreIfFlagSet(0, [this, Op, Base] { SaveX87State(Op, Base); });
|
||||
}
|
||||
// SSE
|
||||
{
|
||||
StoreIfFlagSet(1, [this, Base] { SaveSSEState(Base); });
|
||||
}
|
||||
// AVX
|
||||
if (CTX->HostFeatures.SupportsAVX)
|
||||
{
|
||||
StoreIfFlagSet(2, [this, Base] { SaveAVXState(Base); });
|
||||
}
|
||||
|
||||
// We need to save MXCSR and MXCSR_MASK if either SSE or AVX are requested to be saved
|
||||
{
|
||||
StoreIfFlagSet(1, [this, Base] { SaveMXCSRState(Base); }, 2);
|
||||
}
|
||||
|
||||
// Update XSTATE_BV region of the XSAVE header
|
||||
{
|
||||
OrderedNode *HeaderOffset = _Add(Base, _Constant(512));
|
||||
|
||||
// NOTE: We currently only support the first 3 bits (x87, SSE, and AVX)
|
||||
OrderedNode *RequestedFeatures = _Bfe(3, 0, Mask);
|
||||
|
||||
// XSTATE_BV section of the header is 8 bytes in size, but we only really
|
||||
// care about setting at most 3 bits in the first byte. We zero out the rest.
|
||||
_StoreMem(GPRClass, 8, HeaderOffset, RequestedFeatures);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SaveX87State(OpcodeArgs, OrderedNode *MemBase) {
|
||||
// Saves 512bytes to the memory location provided
|
||||
// Header changes depending on if REX.W is set or not
|
||||
if (Op->Flags & X86Tables::DecodeFlags::FLAG_REX_WIDENING) {
|
||||
@@ -2472,12 +2528,12 @@ void OpDispatchBuilder::FXSaveOp(OpcodeArgs) {
|
||||
|
||||
{
|
||||
auto FCW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreMem(GPRClass, 2, Mem, FCW, 2);
|
||||
_StoreMem(GPRClass, 2, MemBase, FCW, 2);
|
||||
}
|
||||
|
||||
{
|
||||
// We must construct the FSW from our various bits
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(2));
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(2));
|
||||
OrderedNode *FSW = _Constant(0);
|
||||
auto Top = GetX87Top();
|
||||
FSW = _Or(FSW, _Lshl(Top, _Constant(11)));
|
||||
@@ -2496,7 +2552,7 @@ void OpDispatchBuilder::FXSaveOp(OpcodeArgs) {
|
||||
|
||||
{
|
||||
// FTW
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(4));
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(4));
|
||||
auto FTW = _LoadContext(2, GPRClass, offsetof(FEXCore::Core::CPUState, FTW));
|
||||
_StoreMem(GPRClass, 2, MemLocation, FTW, 2);
|
||||
}
|
||||
@@ -2545,33 +2601,142 @@ void OpDispatchBuilder::FXSaveOp(OpcodeArgs) {
|
||||
// MXCSR_MASK: Mask for writes to the MXCSR register
|
||||
// If OSFXSR bit in CR4 is not set than FXSAVE /may/ not save the XMM registers
|
||||
// This is implementation dependent
|
||||
for (unsigned i = 0; i < 8; ++i) {
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
OrderedNode *MMReg = _LoadContext(16, FPRClass, offsetof(FEXCore::Core::CPUState, mm[i]));
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 32));
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(i * 16 + 32));
|
||||
|
||||
_StoreMem(FPRClass, 16, MemLocation, MMReg, 16);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SaveSSEState(OrderedNode *MemBase) {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (unsigned i = 0; i < NumRegs; ++i) {
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
OrderedNode *XMMReg = LoadXMMRegister(i);
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 160));
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(i * 16 + 160));
|
||||
|
||||
_StoreMem(FPRClass, 16, MemLocation, XMMReg, 16);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SaveMXCSRState(OrderedNode *MemBase) {
|
||||
OrderedNode *MXCSR = GetMXCSR();
|
||||
OrderedNode *MXCSRLocation = _Add(MemBase, _Constant(24));
|
||||
_StoreMem(GPRClass, 4, MXCSRLocation, MXCSR, 4);
|
||||
|
||||
// Store the mask for all bits.
|
||||
OrderedNode *MXCSRMaskLocation = _Add(MXCSRLocation, _Constant(4));
|
||||
_StoreMem(GPRClass, 4, MXCSRMaskLocation, _Constant(0xFFFF), 4);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SaveAVXState(OrderedNode *MemBase) {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
OrderedNode *Upper = _VDupElement(32, 16, LoadXMMRegister(i), 1);
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(i * 16 + 576));
|
||||
|
||||
_StoreMem(FPRClass, 16, MemLocation, Upper, 16);
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::GetMXCSR() {
|
||||
// Default MXCSR Value
|
||||
OrderedNode *MXCSR = _Constant(0x1F80);
|
||||
OrderedNode *RoundingMode = _GetRoundingMode();
|
||||
return _Bfi(4, 3, 13, MXCSR, RoundingMode);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FXRStoreOp(OpcodeArgs) {
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1, false);
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
RestoreX87State(Mem);
|
||||
RestoreSSEState(Mem);
|
||||
|
||||
OrderedNode *MXCSRLocation = _Add(Mem, _Constant(24));
|
||||
OrderedNode *MXCSR = _LoadMem(GPRClass, 4, MXCSRLocation, 4);
|
||||
RestoreMXCSRState(MXCSR);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::XRstorOpImpl(OpcodeArgs) {
|
||||
const auto XSaveBase = [this, Op] {
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false);
|
||||
return AppendSegmentOffset(Mem, Op->Flags);
|
||||
};
|
||||
|
||||
// Set up base address for the XSAVE region to restore from, and also read the
|
||||
// XSTATE_BV bit flags out of the XSTATE header.
|
||||
OrderedNode *Base = XSaveBase();
|
||||
OrderedNode *Mask = _LoadMem(GPRClass, 8, _Add(Base, _Constant(512)), 8);
|
||||
|
||||
// If a bit in our XSTATE_BV is set, then we restore from that region of the XSAVE area,
|
||||
// otherwise, if not set, then we need to set the relevant data the bit corresponds to
|
||||
// to it's defined initial configuration.
|
||||
const auto RestoreIfFlagSetOrDefault = [&](uint32_t BitIndex, auto restore_fn, auto default_fn, uint32_t FieldSize = 1){
|
||||
OrderedNode *BitFlag = _Bfe(FieldSize, BitIndex, Mask);
|
||||
auto CondJump = _CondJump(BitFlag, {COND_NEQ});
|
||||
|
||||
auto RestoreBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
SetTrueJumpTarget(CondJump, RestoreBlock);
|
||||
SetCurrentCodeBlock(RestoreBlock);
|
||||
{
|
||||
restore_fn();
|
||||
}
|
||||
auto RestoreExitJump = _Jump();
|
||||
auto DefaultBlock = CreateNewCodeBlockAfter(RestoreBlock);
|
||||
auto ExitBlock = CreateNewCodeBlockAfter(DefaultBlock);
|
||||
SetJumpTarget(RestoreExitJump, ExitBlock);
|
||||
SetFalseJumpTarget(CondJump, DefaultBlock);
|
||||
SetCurrentCodeBlock(DefaultBlock);
|
||||
{
|
||||
default_fn();
|
||||
}
|
||||
auto DefaultExitJump = _Jump();
|
||||
SetJumpTarget(DefaultExitJump, ExitBlock);
|
||||
SetCurrentCodeBlock(ExitBlock);
|
||||
};
|
||||
|
||||
// x87
|
||||
{
|
||||
RestoreIfFlagSetOrDefault(0,
|
||||
[this, Base] { RestoreX87State(Base); },
|
||||
[this, Op] { DefaultX87State(Op); });
|
||||
}
|
||||
// SSE
|
||||
{
|
||||
RestoreIfFlagSetOrDefault(1,
|
||||
[this, Base] { RestoreSSEState(Base); },
|
||||
[this] { DefaultSSEState(); });
|
||||
}
|
||||
// AVX
|
||||
if (CTX->HostFeatures.SupportsAVX)
|
||||
{
|
||||
RestoreIfFlagSetOrDefault(2,
|
||||
[this, Base] { RestoreAVXState(Base); },
|
||||
[this] { DefaultAVXState(); });
|
||||
}
|
||||
|
||||
{
|
||||
// We need to restore the MXCSR if either SSE or AVX are requested to be saved
|
||||
RestoreIfFlagSetOrDefault(1,
|
||||
[this, Base] {
|
||||
OrderedNode *MXCSRLocation = _Add(Base, _Constant(24));
|
||||
OrderedNode *MXCSR = _LoadMem(GPRClass, 4, MXCSRLocation, 4);
|
||||
RestoreMXCSRState(MXCSR);
|
||||
},
|
||||
[] { /* Intentionally do nothing*/ }, 2);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RestoreX87State(OrderedNode *MemBase) {
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, MemBase, 2);
|
||||
_F80LoadFCW(NewFCW);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
{
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(2));
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(2));
|
||||
auto NewFSW = _LoadMem(GPRClass, 2, MemLocation, 2);
|
||||
|
||||
// Strip out the FSW information
|
||||
@@ -2591,26 +2756,78 @@ void OpDispatchBuilder::FXRStoreOp(OpcodeArgs) {
|
||||
|
||||
{
|
||||
// FTW
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(4));
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(4));
|
||||
auto NewFTW = _LoadMem(GPRClass, 2, MemLocation, 2);
|
||||
_StoreContext(2, GPRClass, NewFTW, offsetof(FEXCore::Core::CPUState, FTW));
|
||||
}
|
||||
|
||||
for (unsigned i = 0; i < 8; ++i) {
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 32));
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(i * 16 + 32));
|
||||
auto MMReg = _LoadMem(FPRClass, 16, MemLocation, 16);
|
||||
_StoreContext(16, FPRClass, MMReg, offsetof(FEXCore::Core::CPUState, mm[i]));
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RestoreSSEState(OrderedNode *MemBase) {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (unsigned i = 0; i < NumRegs; ++i) {
|
||||
OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 160));
|
||||
auto XMMReg = _LoadMem(FPRClass, 16, MemLocation, 16);
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(i * 16 + 160));
|
||||
OrderedNode *XMMReg = _LoadMem(FPRClass, 16, MemLocation, 16);
|
||||
StoreXMMRegister(i, XMMReg);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RestoreMXCSRState(OrderedNode *MXCSR) {
|
||||
// We only support the rounding mode and FTZ bit being set
|
||||
OrderedNode *RoundingMode = _Bfe(4, 3, 13, MXCSR);
|
||||
_SetRoundingMode(RoundingMode);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::RestoreAVXState(OrderedNode *MemBase) {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
OrderedNode *XMMReg = LoadXMMRegister(i);
|
||||
OrderedNode *MemLocation = _Add(MemBase, _Constant(i * 16 + 576));
|
||||
OrderedNode *YMMHReg = _LoadMem(FPRClass, 16, MemLocation, 16);
|
||||
OrderedNode *YMM = _VInsElement(32, 16, 1, 0, XMMReg, YMMHReg);
|
||||
StoreXMMRegister(i, YMM);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::DefaultX87State(OpcodeArgs) {
|
||||
// We can piggy-back on FNINIT's implementation, since
|
||||
// it performs the same behavior as required by XRSTOR for resetting flags
|
||||
FNINIT(Op);
|
||||
|
||||
// On top of resetting the flags to a default state, we also need to clear
|
||||
// all of the ST0-7/MM0-7 registers to zero.
|
||||
OrderedNode *ZeroVector = _VectorZero(Core::CPUState::MM_REG_SIZE);
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
_StoreContext(16, FPRClass, ZeroVector, offsetof(FEXCore::Core::CPUState, mm[i]));
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::DefaultSSEState() {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
OrderedNode *ZeroVector = _VectorZero(Core::CPUState::XMM_SSE_REG_SIZE);
|
||||
for (uint32_t i = 0; i < NumRegs; ++i) {
|
||||
StoreXMMRegister(i, ZeroVector);
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::DefaultAVXState() {
|
||||
const auto NumRegs = CTX->Config.Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; i++) {
|
||||
OrderedNode* Reg = LoadXMMRegister(i);
|
||||
OrderedNode* Dst = _VMov(16, Reg);
|
||||
StoreXMMRegister(i, Dst);
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::PALIGNROpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1,
|
||||
const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm) {
|
||||
@@ -2676,18 +2893,11 @@ void OpDispatchBuilder::UCOMISxOp<8>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::LDMXCSR(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
// We only support the rounding mode and FTZ bit being set
|
||||
OrderedNode *RoundingMode = _Bfe(4, 3, 13, Dest);
|
||||
_SetRoundingMode(RoundingMode);
|
||||
RestoreMXCSRState(Dest);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::STMXCSR(OpcodeArgs) {
|
||||
// Default MXCSR
|
||||
OrderedNode *MXCSR = _Constant(32, 0x1F80);
|
||||
OrderedNode *RoundingMode = _GetRoundingMode();
|
||||
MXCSR = _Bfi(4, 3, 13, MXCSR, RoundingMode);
|
||||
|
||||
StoreResult(GPRClass, Op, MXCSR, -1);
|
||||
StoreResult(GPRClass, Op, GetMXCSR(), -1);
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::PACKUSOpImpl(OpcodeArgs, size_t ElementSize,
|
||||
@@ -3417,36 +3627,19 @@ void OpDispatchBuilder::VPHSUBOp<2>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VPHSUBOp<4>(OpcodeArgs);
|
||||
|
||||
OrderedNode* OpDispatchBuilder::PHADDSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1,
|
||||
const X86Tables::DecodedOperand& Src2) {
|
||||
OrderedNode* OpDispatchBuilder::PHADDSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op) {
|
||||
const auto Size = GetSrcSize(Op);
|
||||
const uint8_t ElementSize = 2;
|
||||
|
||||
OrderedNode *Src1Node = LoadSource(FPRClass, Op, Src1, Op->Flags, -1);
|
||||
OrderedNode *Src2Node = LoadSource(FPRClass, Op, Src2, Op->Flags, -1);
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Src1Op, Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Src2Op, Op->Flags, -1);
|
||||
|
||||
if (Size == 8) {
|
||||
// Implementation is more efficient for 8byte registers
|
||||
OrderedNode *Src1_Larger = _VSXTL(Size * 2, 2, Src1Node);
|
||||
OrderedNode *Src2_Larger = _VSXTL(Size * 2, 2, Src2Node);
|
||||
|
||||
OrderedNode *AddRes = _VAddP(Size * 2, 4, Src1_Larger, Src2_Larger);
|
||||
|
||||
// Saturate back down to the result
|
||||
return _VSQXTN(Size * 2, 4, AddRes);
|
||||
}
|
||||
|
||||
OrderedNode *Src1_Larger = _VSXTL(Size, 2, Src1Node);
|
||||
OrderedNode *Src1_Larger_H = _VSXTL2(Size, 2, Src1Node);
|
||||
|
||||
OrderedNode *Src2_Larger = _VSXTL(Size, 2, Src2Node);
|
||||
OrderedNode *Src2_Larger_H = _VSXTL2(Size, 2, Src2Node);
|
||||
|
||||
OrderedNode *AddRes_L = _VAddP(Size, 4, Src1_Larger, Src1_Larger_H);
|
||||
OrderedNode *AddRes_H = _VAddP(Size, 4, Src2_Larger, Src2_Larger_H);
|
||||
auto Even = _VUnZip(Size, ElementSize, Src1, Src2);
|
||||
auto Odd = _VUnZip2(Size, ElementSize, Src1, Src2);
|
||||
|
||||
// Saturate back down to the result
|
||||
OrderedNode *Res = _VSQXTN(Size, 4, AddRes_L);
|
||||
return _VSQXTN2(Size, 4, Res, AddRes_H);
|
||||
return _VSQAdd(Size, ElementSize, Even, Odd);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PHADDS(OpcodeArgs) {
|
||||
@@ -3473,51 +3666,15 @@ OrderedNode* OpDispatchBuilder::PHSUBSOpImpl(OpcodeArgs, const X86Tables::Decode
|
||||
const X86Tables::DecodedOperand& Src2Op) {
|
||||
const auto Size = GetSrcSize(Op);
|
||||
const uint8_t ElementSize = 2;
|
||||
const uint8_t NumElements = Size / ElementSize;
|
||||
|
||||
OrderedNode *Src1 = LoadSource(FPRClass, Op, Src1Op, Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Src2Op, Op->Flags, -1);
|
||||
|
||||
// This is a bit complicated since AArch64 doesn't support a pairwise subtract
|
||||
OrderedNode *Src1_Neg = _VNeg(Size, ElementSize, Src1);
|
||||
OrderedNode *Src2_Neg = _VNeg(Size, ElementSize, Src2);
|
||||
|
||||
// Now we need to swizzle the values
|
||||
OrderedNode *Swizzle_Src1 = Src1;
|
||||
OrderedNode *Swizzle_Src2 = Src2;
|
||||
|
||||
// Odd elements turn in to negated elements
|
||||
for (size_t i = 1; i < NumElements; i += 2) {
|
||||
Swizzle_Src1 = _VInsElement(Size, ElementSize, i, i, Swizzle_Src1, Src1_Neg);
|
||||
Swizzle_Src2 = _VInsElement(Size, ElementSize, i, i, Swizzle_Src2, Src2_Neg);
|
||||
}
|
||||
|
||||
Src1 = Swizzle_Src1;
|
||||
Src2 = Swizzle_Src2;
|
||||
|
||||
if (Size == 8) {
|
||||
// Implementation is more efficient for 8byte registers
|
||||
OrderedNode *Src1_Larger = _VSXTL(Size * 2, 2, Src1);
|
||||
OrderedNode *Src2_Larger = _VSXTL(Size * 2, 2, Src2);
|
||||
|
||||
OrderedNode *AddRes = _VAddP(Size * 2, 4, Src1_Larger, Src2_Larger);
|
||||
|
||||
// Saturate back down to the result
|
||||
return _VSQXTN(Size * 2, 4, AddRes);
|
||||
}
|
||||
|
||||
OrderedNode *Src1_Larger = _VSXTL(Size, 2, Src1);
|
||||
OrderedNode *Src1_Larger_H = _VSXTL2(Size, 2, Src1);
|
||||
|
||||
OrderedNode *Src2_Larger = _VSXTL(Size, 2, Src2);
|
||||
OrderedNode *Src2_Larger_H = _VSXTL2(Size, 2, Src2);
|
||||
|
||||
OrderedNode *AddRes_L = _VAddP(Size, 4, Src1_Larger, Src1_Larger_H);
|
||||
OrderedNode *AddRes_H = _VAddP(Size, 4, Src2_Larger, Src2_Larger_H);
|
||||
auto Even = _VUnZip(Size, ElementSize, Src1, Src2);
|
||||
auto Odd = _VUnZip2(Size, ElementSize, Src1, Src2);
|
||||
|
||||
// Saturate back down to the result
|
||||
OrderedNode *Res = _VSQXTN(Size, 4, AddRes_L);
|
||||
return _VSQXTN2(Size, 4, Res, AddRes_H);
|
||||
return _VSQSub(Size, ElementSize, Even, Odd);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PHSUBS(OpcodeArgs) {
|
||||
@@ -3554,33 +3711,19 @@ OrderedNode* OpDispatchBuilder::PSADBWOpImpl(OpcodeArgs,
|
||||
OrderedNode *Src2 = LoadSource(FPRClass, Op, Src2Op, Op->Flags, -1);
|
||||
|
||||
if (Size == 8) {
|
||||
OrderedNode *Src1_Low = _VUXTL(Size*2, 1, Src1);
|
||||
OrderedNode *Src2_Low = _VUXTL(Size*2, 1, Src2);
|
||||
|
||||
OrderedNode *SubResult = _VSub(Size*2, 2, Src1_Low, Src2_Low);
|
||||
OrderedNode *AbsResult = _VAbs(Size*2, 2, SubResult);
|
||||
auto AbsResult = _VUABDL(Size * 2, 1, Src1, Src2);
|
||||
|
||||
// Now vector-wide add the results for each
|
||||
return _VAddV(Size * 2, 2, AbsResult);
|
||||
}
|
||||
|
||||
|
||||
OrderedNode *Src1_Low = _VUXTL(Size, 1, Src1);
|
||||
OrderedNode *Src1_High = _VUXTL2(Size, 1, Src1);
|
||||
|
||||
OrderedNode *Src2_Low = _VUXTL(Size, 1, Src2);
|
||||
OrderedNode *Src2_High = _VUXTL2(Size, 1, Src2);
|
||||
|
||||
OrderedNode *SubResult_Low = _VSub(Size, 2, Src1_Low, Src2_Low);
|
||||
OrderedNode *SubResult_High = _VSub(Size, 2, Src1_High, Src2_High);
|
||||
|
||||
OrderedNode *AbsResult_Low = _VAbs(Size, 2, SubResult_Low);
|
||||
OrderedNode *AbsResult_High = _VAbs(Size, 2, SubResult_High);
|
||||
auto AbsResult_Low = _VUABDL(Size, 1, Src1, Src2);
|
||||
auto AbsResult_High = _VUABDL2(Size, 1, Src1, Src2);
|
||||
|
||||
OrderedNode *Result_Low = _VAddV(16, 2, AbsResult_Low);
|
||||
OrderedNode *Result_High = _VAddV(16, 2, AbsResult_High);
|
||||
auto Low = _VZip(Size, 8, Result_Low, Result_High);
|
||||
|
||||
OrderedNode *Low = _VInsElement(Size, 8, 1, 0, Result_Low, Result_High);
|
||||
if (Is128Bit) {
|
||||
return Low;
|
||||
}
|
||||
@@ -4011,11 +4154,8 @@ OrderedNode* OpDispatchBuilder::PHMINPOSUWOpImpl(OpcodeArgs) {
|
||||
Element, MinGPR, Indexes[i - 1], Pos);
|
||||
}
|
||||
|
||||
// Insert the minimum in to bits [15:0]
|
||||
OrderedNode *Result = _VMov(2, Min);
|
||||
|
||||
// Insert position in to bits [18:16]
|
||||
return _VInsGPR(16, 2, 1, Result, Pos);
|
||||
return _VInsGPR(16, 2, 1, Min, Pos);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PHMINPOSUWOp(OpcodeArgs) {
|
||||
@@ -4591,9 +4731,9 @@ void OpDispatchBuilder::VPERMILRegOp<4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VPERMILRegOp<8>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit) {
|
||||
void OpDispatchBuilder::PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask) {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src[1] needs to be a literal");
|
||||
const auto Control = Op->Src[1].Data.Literal.Value;
|
||||
const uint16_t Control = Op->Src[1].Data.Literal.Value;
|
||||
|
||||
// SSE4.2 string instructions modify flags, so invalidate
|
||||
// any previously deferred flags.
|
||||
@@ -4611,32 +4751,73 @@ void OpDispatchBuilder::PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit) {
|
||||
OrderedNode *IntermediateResult{};
|
||||
if (IsExplicit) {
|
||||
// Will be 4 in the absence of a REX.W bit and 8 in the presence of a REX.W bit.
|
||||
//
|
||||
// While the control bit immediate for the instruction itself is only ever 8 bits
|
||||
// in size, we use it as a 16-bit value so that we can use the 8th bit to signify
|
||||
// whether or not RAX and RDX should be interpreted as a 64-bit value.
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto Is64Bit = SrcSize == 8;
|
||||
const auto NewControl = uint16_t(Control | (uint16_t(Is64Bit) << 8));
|
||||
|
||||
OrderedNode *SrcRAX = LoadGPRRegister(X86State::REG_RAX);
|
||||
OrderedNode *SrcRDX = LoadGPRRegister(X86State::REG_RDX);
|
||||
|
||||
IntermediateResult = _VPCMPESTRX(SrcSize, Src1, Src2, SrcRAX, SrcRDX, Control);
|
||||
IntermediateResult = _VPCMPESTRX(Src1, Src2, SrcRAX, SrcRDX, NewControl);
|
||||
} else {
|
||||
IntermediateResult = _VPCMPISTRX(Src1, Src2, Control);
|
||||
}
|
||||
OrderedNode *ResultNoFlags = _And(IntermediateResult, _Constant(0xFFFF));
|
||||
|
||||
// For the indexed variant of the instructions, if control[6] is set, then we
|
||||
// store the index of the most significant bit into ECX. If it's not set,
|
||||
// then we store the least significant bit.
|
||||
OrderedNode *ZeroConst = _Constant(0);
|
||||
|
||||
const auto ECXResult = [&]() -> OrderedNode* {
|
||||
if (IsMask) {
|
||||
// For the masked variant of the instructions, if control[6] is set, then we
|
||||
// need to expand the intermediate result into a byte or word mask (depending
|
||||
// on data size specified in control[1]) along the entire length of XMM0,
|
||||
// where set bits in the intermediate result set the corresponding entry
|
||||
// in XMM0 to all 1s and unset bits set the corresponding entry to all 0s.
|
||||
//
|
||||
// If control[6] is not set, then we just store the intermediate result as-is
|
||||
// into the least significant bits of XMM0 and zero extend it.
|
||||
const auto IsExpandedMask = (Control & 0b0100'0000) != 0;
|
||||
|
||||
if (IsExpandedMask) {
|
||||
// We need to iterate over the intermediate result and
|
||||
// expand the mask into XMM0 elements.
|
||||
const auto ElementSize = 1U << (Control & 1);
|
||||
const auto NumElements = 16U >> (Control & 1);
|
||||
|
||||
OrderedNode *Result = _VectorZero(Core::CPUState::XMM_SSE_REG_SIZE);
|
||||
for (uint32_t i = 0; i < NumElements; i++) {
|
||||
OrderedNode *SignBit = _Sbfe(1, i, IntermediateResult);
|
||||
Result = _VInsGPR(Core::CPUState::XMM_SSE_REG_SIZE, ElementSize, i, Result, SignBit);
|
||||
}
|
||||
StoreXMMRegister(0, Result);
|
||||
} else {
|
||||
// We insert the intermediate result as-is.
|
||||
StoreXMMRegister(0, _VCastFromGPR(16, 2, IntermediateResult));
|
||||
}
|
||||
} else {
|
||||
// For the indexed variant of the instructions, if control[6] is set, then we
|
||||
// store the index of the most significant bit into ECX. If it's not set,
|
||||
// then we store the least significant bit.
|
||||
const auto UseMSBIndex = (Control & 0b0100'0000) != 0;
|
||||
|
||||
OrderedNode *ResultNoFlags = _Bfe(16, 0, IntermediateResult);
|
||||
|
||||
OrderedNode *IfZero = _Constant(16 >> (Control & 1));
|
||||
OrderedNode *IfNotZero = UseMSBIndex ? _FindMSB(ResultNoFlags)
|
||||
: _FindLSB(ResultNoFlags);
|
||||
|
||||
return _Select(IR::COND_EQ, ResultNoFlags, ZeroConst,
|
||||
IfZero, IfNotZero);
|
||||
}();
|
||||
OrderedNode *Result = _Select(IR::COND_EQ, ResultNoFlags, ZeroConst,
|
||||
IfZero, IfNotZero);
|
||||
|
||||
const uint8_t GPRSize = CTX->GetGPRSize();
|
||||
if (GPRSize == 8) {
|
||||
// If being stored to an 8-byte register, zero extend the 4-byte result.
|
||||
Result = _Bfe(8, 32, 0, Result);
|
||||
}
|
||||
StoreGPRRegister(X86State::REG_RCX, Result);
|
||||
}
|
||||
|
||||
// Set all of the necessary flags.
|
||||
// We use the top 16-bits of the result to store the flags
|
||||
@@ -4656,16 +4837,19 @@ void OpDispatchBuilder::PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit) {
|
||||
|
||||
SetRFLAG<X86State::RFLAG_AF_LOC>(ZeroConst);
|
||||
SetRFLAG<X86State::RFLAG_PF_LOC>(ZeroConst);
|
||||
|
||||
// ... and we're done!
|
||||
StoreGPRRegister(X86State::REG_RCX, ECXResult, 4);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VPCMPESTRIOp(OpcodeArgs) {
|
||||
PCMPXSTRXOpImpl(Op, true);
|
||||
PCMPXSTRXOpImpl(Op, true, false);
|
||||
}
|
||||
void OpDispatchBuilder::VPCMPESTRMOp(OpcodeArgs) {
|
||||
PCMPXSTRXOpImpl(Op, true, true);
|
||||
}
|
||||
void OpDispatchBuilder::VPCMPISTRIOp(OpcodeArgs) {
|
||||
PCMPXSTRXOpImpl(Op, false);
|
||||
PCMPXSTRXOpImpl(Op, false, false);
|
||||
}
|
||||
void OpDispatchBuilder::VPCMPISTRMOp(OpcodeArgs) {
|
||||
PCMPXSTRXOpImpl(Op, false, true);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
+15
-16
@@ -171,12 +171,14 @@ void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
auto zero = _Constant(0);
|
||||
|
||||
// Sign extend to 64bits
|
||||
if (read_width != 8)
|
||||
if (read_width != 8) {
|
||||
data = _Sext(read_width * 8, data);
|
||||
}
|
||||
|
||||
// Extract sign and make interger absolute
|
||||
auto sign = _Select(COND_SLT, data, zero, _Constant(0x8000), zero);
|
||||
auto absolute = _Select(COND_SLT, data, zero, _Sub(zero, data), data);
|
||||
|
||||
auto absolute = _Abs(data);
|
||||
|
||||
// left justify the absolute interger
|
||||
auto shift = _Sub(_Constant(63), _FindMSB(absolute));
|
||||
@@ -621,18 +623,19 @@ void OpDispatchBuilder::FXTRACT(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FNINIT(OpcodeArgs) {
|
||||
auto Zero = _Constant(0);
|
||||
// Init FCW to 0x037F
|
||||
auto NewFCW = _Constant(16, 0x037F);
|
||||
_F80LoadFCW(NewFCW);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
// Init FSW to 0
|
||||
SetX87Top(_Constant(0));
|
||||
SetX87Top(Zero);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(Zero);
|
||||
|
||||
// Tags all get set to 0b11
|
||||
_StoreContext(2, GPRClass, _Constant(0xFFFF), offsetof(FEXCore::Core::CPUState, FTW));
|
||||
@@ -1278,7 +1281,7 @@ void OpDispatchBuilder::X87FXAM(OpcodeArgs) {
|
||||
OrderedNode *Result = _VExtractToGPR(16, 8, a, 1);
|
||||
|
||||
// Extract the sign bit
|
||||
Result = _Lshr(Result, _Constant(15));
|
||||
Result = _Bfe(1, 15, Result);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Result);
|
||||
|
||||
// Claim this is a normal number
|
||||
@@ -1354,28 +1357,24 @@ void OpDispatchBuilder::X87FCMOV(OpcodeArgs) {
|
||||
break;
|
||||
}
|
||||
|
||||
auto MaskConst = _Constant(FLAGMask);
|
||||
auto RFLAG = GetPackedRFLAG(FLAGMask);
|
||||
|
||||
auto RFLAG = GetPackedRFLAG(false);
|
||||
|
||||
auto AndOp = _And(RFLAG, MaskConst);
|
||||
switch (Type) {
|
||||
case COMPARE_ZERO: {
|
||||
SrcCond = _Select(FEXCore::IR::COND_EQ,
|
||||
AndOp, ZeroConst, OneConst, ZeroConst);
|
||||
RFLAG, ZeroConst, OneConst, ZeroConst);
|
||||
break;
|
||||
}
|
||||
case COMPARE_NOTZERO: {
|
||||
SrcCond = _Select(FEXCore::IR::COND_EQ,
|
||||
AndOp, ZeroConst, ZeroConst, OneConst);
|
||||
RFLAG, ZeroConst, ZeroConst, OneConst);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
SrcCond = _Sbfe(1, 0, SrcCond);
|
||||
|
||||
OrderedNode *VecCond = _VCastFromGPR(16, 8, SrcCond);
|
||||
VecCond = _VInsGPR(16, 8, 1, VecCond, SrcCond);
|
||||
OrderedNode *VecCond = _VDupFromGPR(16, 8, SrcCond);
|
||||
|
||||
auto top = GetX87Top();
|
||||
OrderedNode* arg;
|
||||
|
||||
@@ -50,12 +50,12 @@ void OpDispatchBuilder::FNINITF64(OpcodeArgs) {
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
// Init FSW to 0
|
||||
SetX87Top(_Constant(0));
|
||||
SetX87Top(Zero);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(_Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(Zero);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(Zero);
|
||||
|
||||
// Tags all get set to 0b11
|
||||
_StoreContext(2, GPRClass, _Constant(0xFFFF), offsetof(FEXCore::Core::CPUState, FTW));
|
||||
@@ -1119,7 +1119,7 @@ void OpDispatchBuilder::X87FXAMF64(OpcodeArgs) {
|
||||
OrderedNode *Result = _VExtractToGPR(8, 8, a, 0);
|
||||
|
||||
// Extract the sign bit
|
||||
Result = _Lshr(Result, _Constant(63));
|
||||
Result = _Bfe(1, 63, Result);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Result);
|
||||
|
||||
// Claim this is a normal number
|
||||
|
||||
@@ -6,26 +6,6 @@
|
||||
#include <signal.h>
|
||||
|
||||
namespace FEXCore {
|
||||
struct ThreadState {
|
||||
FEXCore::Core::InternalThreadState *Thread{};
|
||||
};
|
||||
|
||||
thread_local ThreadState ThreadData{};
|
||||
|
||||
FEXCore::Core::InternalThreadState *SignalDelegator::GetTLSThread() {
|
||||
return ThreadData.Thread;
|
||||
}
|
||||
|
||||
void SignalDelegator::RegisterTLSState(FEXCore::Core::InternalThreadState *Thread) {
|
||||
ThreadData.Thread = Thread;
|
||||
RegisterFrontendTLSState(Thread);
|
||||
}
|
||||
|
||||
void SignalDelegator::UninstallTLSState(FEXCore::Core::InternalThreadState *Thread) {
|
||||
UninstallFrontendTLSState(Thread);
|
||||
ThreadData.Thread = nullptr;
|
||||
}
|
||||
|
||||
void SignalDelegator::RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
|
||||
SetHostSignalHandler(Signal, Func, Required);
|
||||
FrontendRegisterHostSignalHandler(Signal, Func, Required);
|
||||
|
||||
@@ -146,10 +146,10 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0x9E, 1, X86InstInfo{"SAHF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{0x9F, 1, X86InstInfo{"LAHF", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{0xA4, 1, X86InstInfo{"MOVSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA5, 1, X86InstInfo{"MOVS", TYPE_INST, FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA6, 1, X86InstInfo{"CMPSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA7, 1, X86InstInfo{"CMPS", TYPE_INST, FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA4, 1, X86InstInfo{"MOVSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA5, 1, X86InstInfo{"MOVS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA6, 1, X86InstInfo{"CMPSB", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
{0xA7, 1, X86InstInfo{"CMPS", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS | FLAGS_SUPPORTS_REP, 0, nullptr}},
|
||||
|
||||
{0xA8, 1, X86InstInfo{"TEST", TYPE_INST, GenFlagsSameSize(SIZE_8BIT) | FLAGS_SF_DST_RAX , 1, nullptr}},
|
||||
{0xA9, 1, X86InstInfo{"TEST", TYPE_INST, FLAGS_SF_DST_RAX | FLAGS_SRC_SEXT | FLAGS_DISPLACE_SIZE_DIV_2, 4, nullptr}},
|
||||
@@ -169,7 +169,7 @@ void InitializeBaseTables(Context::OperatingMode Mode) {
|
||||
{0xC9, 1, X86InstInfo{"LEAVE", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_DEBUG_MEM_ACCESS , 0, nullptr}},
|
||||
{0xCA, 2, X86InstInfo{"RETF", TYPE_PRIV, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SETS_RIP | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
{0xCC, 1, X86InstInfo{"INT3", TYPE_INST, FLAGS_DEBUG, 0, nullptr}},
|
||||
{0xCD, 1, X86InstInfo{"INT", TYPE_INST, FLAGS_DEBUG , 1, nullptr}},
|
||||
{0xCD, 1, X86InstInfo{"INT", TYPE_INST, DEFAULT_SYSCALL_FLAGS, 1, nullptr}},
|
||||
{0xCF, 1, X86InstInfo{"IRET", TYPE_INST, FLAGS_SETS_RIP | FLAGS_BLOCK_END, 0, nullptr}},
|
||||
|
||||
{0xD7, 1, X86InstInfo{"XLAT", TYPE_INST, FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
|
||||
@@ -43,9 +43,9 @@ void InitializeH0F3ATables(Context::OperatingMode Mode) {
|
||||
{OPD(0, PF_3A_66, 0x42), 1, X86InstInfo{"MPSADBW", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x44), 1, X86InstInfo{"PCLMULQDQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(0, PF_3A_66, 0x60), 1, X86InstInfo{"PCMPESTRM", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x60), 1, X86InstInfo{"PCMPESTRM", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x61), 1, X86InstInfo{"PCMPESTRI", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x62), 1, X86InstInfo{"PCMPISTRM", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x62), 1, X86InstInfo{"PCMPISTRM", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(0, PF_3A_66, 0x63), 1, X86InstInfo{"PCMPISTRI", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(0, PF_3A_NONE, 0xCC), 1, X86InstInfo{"SHA1RNDS4", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
@@ -334,7 +334,7 @@ void InitializeSecondaryGroupTables() {
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 1), 1, X86InstInfo{"FXRSTOR", TYPE_INST, FLAGS_MODRM, 0, nullptr}}, // MMX/x87
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 2), 1, X86InstInfo{"LDMXCSR", TYPE_INST, GenFlagsSameSize(SIZE_32BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 3), 1, X86InstInfo{"STMXCSR", TYPE_INST, GenFlagsSameSize(SIZE_32BIT) | FLAGS_MODRM | FLAGS_SF_MOD_DST | FLAGS_SF_MOD_MEM_ONLY, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 4), 1, X86InstInfo{"XSAVE", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 4), 1, X86InstInfo{"XSAVE", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 5), 1, X86InstInfo{"LFENCE/XRSTOR", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 6), 1, X86InstInfo{"MFENCE/XSAVEOPT", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
{OPD(TYPE_GROUP_15, PF_NONE, 7), 1, X86InstInfo{"SFENCE/CLFLUSH", TYPE_INST, FLAGS_MODRM | FLAGS_SF_MOD_DST, 0, nullptr}},
|
||||
|
||||
@@ -48,7 +48,7 @@ void InitializeSecondaryModRMTables() {
|
||||
{((3 << 3) | 1), 1, X86InstInfo{"RDTSCP", TYPE_INST, FLAGS_NONE, 0, nullptr}},
|
||||
{((3 << 3) | 2), 1, X86InstInfo{"MONITORX", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{((3 << 3) | 3), 1, X86InstInfo{"MWAITX", TYPE_PRIV, FLAGS_NONE, 0, nullptr}},
|
||||
{((3 << 3) | 4), 1, X86InstInfo{"CLZERO", TYPE_INST, FLAGS_SF_SRC_RAX, 0, nullptr}},
|
||||
{((3 << 3) | 4), 1, X86InstInfo{"CLZERO", TYPE_INST, GenFlagsSameSize(SIZE_64BITDEF) | FLAGS_SF_SRC_RAX | FLAGS_DEBUG_MEM_ACCESS, 0, nullptr}},
|
||||
{((3 << 3) | 5), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{((3 << 3) | 6), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
{((3 << 3) | 7), 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
@@ -22,7 +22,7 @@ void InitializeSecondaryTables(Context::OperatingMode Mode) {
|
||||
{0x02, 1, X86InstInfo{"LAR", TYPE_UNDEC, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x03, 1, X86InstInfo{"LSL", TYPE_UNDEC, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x04, 1, X86InstInfo{"", TYPE_INVALID, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x05, 1, X86InstInfo{"SYSCALL", TYPE_INST, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x05, 1, X86InstInfo{"SYSCALL", TYPE_INST, DEFAULT_SYSCALL_FLAGS, 0, nullptr}},
|
||||
{0x06, 1, X86InstInfo{"CLTS", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x07, 1, X86InstInfo{"SYSRET", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
{0x08, 1, X86InstInfo{"INVD", TYPE_PRIV, FLAGS_NO_OVERLAY, 0, nullptr}},
|
||||
|
||||
@@ -342,10 +342,10 @@ void InitializeVEXTables() {
|
||||
{OPD(2, 0b01, 0x8C), 1, X86InstInfo{"VPMASKMOV", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x8E), 1, X86InstInfo{"VPMASKMOV", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_SF_MOD_MEM_ONLY | FLAGS_SF_MOD_DST | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(2, 0b01, 0x90), 1, X86InstInfo{"VPGATHERD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x91), 1, X86InstInfo{"VPGATHERQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x92), 1, X86InstInfo{"VPGATHERD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x93), 1, X86InstInfo{"VPGATHERQ", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x90), 1, X86InstInfo{"VPGATHERDD/Q", TYPE_UNDEC, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_2ND_SRC | FLAGS_VEX_VSIB | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x91), 1, X86InstInfo{"VPGATHERQD/Q", TYPE_UNDEC, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_2ND_SRC | FLAGS_VEX_VSIB | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x92), 1, X86InstInfo{"VGATHERDPS/D", TYPE_UNDEC, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_2ND_SRC | FLAGS_VEX_VSIB | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x93), 1, X86InstInfo{"VGATHERQPS/D", TYPE_UNDEC, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_2ND_SRC | FLAGS_VEX_VSIB | FLAGS_XMM_FLAGS, 0, nullptr}},
|
||||
|
||||
{OPD(2, 0b01, 0x96), 1, X86InstInfo{"VFMADDSUB132", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(2, 0b01, 0x97), 1, X86InstInfo{"VFMSUBADD132", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
@@ -456,9 +456,9 @@ void InitializeVEXTables() {
|
||||
{OPD(3, 0b01, 0x5E), 1, X86InstInfo{"VMFSUBADDPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x5F), 1, X86InstInfo{"VFMSUBADDPD", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
{OPD(3, 0b01, 0x60), 1, X86InstInfo{"VPCMPESTRM", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x60), 1, X86InstInfo{"VPCMPESTRM", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x61), 1, X86InstInfo{"VPCMPESTRI", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x62), 1, X86InstInfo{"VPCMPISTRM", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
{OPD(3, 0b01, 0x62), 1, X86InstInfo{"VPCMPISTRM", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
{OPD(3, 0b01, 0x63), 1, X86InstInfo{"VPCMPISTRI", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_XMM_FLAGS, 1, nullptr}},
|
||||
|
||||
{OPD(3, 0b01, 0x68), 1, X86InstInfo{"VFMADDPS", TYPE_UNDEC, FLAGS_NONE, 0, nullptr}},
|
||||
|
||||
@@ -25,7 +25,7 @@ constexpr uint32_t FLAG_ADDRESS_SIZE = (1 << 1);
|
||||
constexpr uint32_t FLAG_LOCK = (1 << 2);
|
||||
constexpr uint32_t FLAG_LEGACY_PREFIX = (1 << 3);
|
||||
constexpr uint32_t FLAG_REX_PREFIX = (1 << 4);
|
||||
// Hole where 1 << 5 is
|
||||
constexpr uint32_t FLAG_VSIB_BYTE = (1 << 5);
|
||||
// Hole where 1 << 6 is
|
||||
constexpr uint32_t FLAG_REX_WIDENING = (1 << 7);
|
||||
constexpr uint32_t FLAG_REX_XGPR_B = (1 << 8);
|
||||
@@ -138,9 +138,10 @@ struct DecodedOperand {
|
||||
}
|
||||
|
||||
union TypeUnion {
|
||||
struct {
|
||||
struct GPRType {
|
||||
bool HighBits;
|
||||
uint8_t GPR;
|
||||
auto operator<=>(const GPRType&) const = default;
|
||||
} GPR;
|
||||
|
||||
struct {
|
||||
@@ -155,9 +156,10 @@ struct DecodedOperand {
|
||||
} Value;
|
||||
} RIPLiteral;
|
||||
|
||||
struct {
|
||||
struct LiteralType {
|
||||
uint64_t Value;
|
||||
uint8_t Size;
|
||||
auto operator<=>(const LiteralType&) const = default;
|
||||
} Literal;
|
||||
|
||||
struct {
|
||||
@@ -347,6 +349,8 @@ constexpr InstFlagType FLAGS_VEX_1ST_SRC = (1ULL << 24);
|
||||
constexpr InstFlagType FLAGS_VEX_2ND_SRC = (1ULL << 25);
|
||||
// Whether or not the instruction has a VEX prefix for the destination
|
||||
constexpr InstFlagType FLAGS_VEX_DST = (1ULL << 26);
|
||||
// Whether or not the instruction has a VSIB byte
|
||||
constexpr InstFlagType FLAGS_VEX_VSIB = (1ULL << 27);
|
||||
|
||||
constexpr InstFlagType FLAGS_SIZE_DST_OFF = 58;
|
||||
constexpr InstFlagType FLAGS_SIZE_SRC_OFF = FLAGS_SIZE_DST_OFF + 3;
|
||||
@@ -361,6 +365,13 @@ constexpr InstFlagType SIZE_128BIT = 0b101;
|
||||
constexpr InstFlagType SIZE_256BIT = 0b110;
|
||||
constexpr InstFlagType SIZE_64BITDEF = 0b111; // Default mode is 64bit instead of typical 32bit
|
||||
|
||||
#ifndef _WIN32
|
||||
constexpr uint32_t DEFAULT_SYSCALL_FLAGS = FLAGS_NO_OVERLAY;
|
||||
#else
|
||||
// Syscall ends a block on WIN32 because the instruction can update the CPU's RIP.
|
||||
constexpr uint32_t DEFAULT_SYSCALL_FLAGS = FLAGS_NO_OVERLAY | FLAGS_BLOCK_END;
|
||||
#endif
|
||||
|
||||
constexpr InstFlagType GetSizeDstFlags(InstFlagType Flags) { return (Flags >> FLAGS_SIZE_DST_OFF) & SIZE_MASK; }
|
||||
constexpr InstFlagType GetSizeSrcFlags(InstFlagType Flags) { return (Flags >> FLAGS_SIZE_SRC_OFF) & SIZE_MASK; }
|
||||
|
||||
|
||||
+2
-1
@@ -53,8 +53,9 @@ static __attribute__((aligned(16), naked, section("HostToGuestTrampolineTemplate
|
||||
);
|
||||
#elif defined(_M_ARM_64)
|
||||
asm(
|
||||
// x11 is part of the custom ABI and needs to point to the TrampolineInstanceInfo.
|
||||
"ldr x16, 0f \n"
|
||||
"adr x11, 0f \n"
|
||||
"ldr x16, [x11] \n"
|
||||
"br x16 \n"
|
||||
// Manually align to the next 8-byte boundary
|
||||
// NOTE: GCC over-aligns to a full page when using .align directives on ARM (last tested on GCC 11.2)
|
||||
|
||||
+6
-6
@@ -206,7 +206,7 @@ namespace FEXCore::IR {
|
||||
FEXCore::Allocator::YesIKnowImNotSupposedToUseTheGlibcAllocator glibc;
|
||||
|
||||
AOTIRCaptureCacheWriteoutLock.lock();
|
||||
std::function<void()> fn = std::move(AOTIRCaptureCacheWriteoutQueue.front());
|
||||
WriteOutFn fn = std::move(AOTIRCaptureCacheWriteoutQueue.front());
|
||||
bool MaybeEmpty = false;
|
||||
AOTIRCaptureCacheWriteoutQueue.pop();
|
||||
MaybeEmpty = AOTIRCaptureCacheWriteoutQueue.size() == 0;
|
||||
@@ -225,7 +225,7 @@ namespace FEXCore::IR {
|
||||
LOGMAN_MSG_A_FMT("Must never get here");
|
||||
}
|
||||
|
||||
void AOTIRCaptureCache::AOTIRCaptureCacheWriteoutQueue_Append(const std::function<void()> &fn) {
|
||||
void AOTIRCaptureCache::AOTIRCaptureCacheWriteoutQueue_Append(const WriteOutFn &fn) {
|
||||
bool Flush = false;
|
||||
|
||||
{
|
||||
@@ -242,7 +242,7 @@ namespace FEXCore::IR {
|
||||
}
|
||||
}
|
||||
|
||||
void AOTIRCaptureCache::WriteFilesWithCode(std::function<void(const fextl::string& fileid, const fextl::string& filename)> Writer) {
|
||||
void AOTIRCaptureCache::WriteFilesWithCode(const Context::AOTIRCodeFileWriterFn &Writer) {
|
||||
std::shared_lock lk(AOTIRCacheLock);
|
||||
for( const auto &Entry: AOTIRCache) {
|
||||
if (Entry.second.ContainsCode) {
|
||||
@@ -251,8 +251,8 @@ namespace FEXCore::IR {
|
||||
}
|
||||
}
|
||||
|
||||
AOTIRCaptureCache::PreGenerateIRFetchResult AOTIRCaptureCache::PreGenerateIRFetch(uint64_t GuestRIP, FEXCore::IR::IRListView *IRList) {
|
||||
auto AOTIRCacheEntry = CTX->SyscallHandler->LookupAOTIRCacheEntry(GuestRIP);
|
||||
AOTIRCaptureCache::PreGenerateIRFetchResult AOTIRCaptureCache::PreGenerateIRFetch(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, FEXCore::IR::IRListView *IRList) {
|
||||
auto AOTIRCacheEntry = CTX->SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
|
||||
|
||||
PreGenerateIRFetchResult Result{};
|
||||
|
||||
@@ -306,7 +306,7 @@ namespace FEXCore::IR {
|
||||
// Both generated ir and LibraryJITName need a named region lookup
|
||||
if (GeneratedIR || CTX->Config.LibraryJITNaming() || CTX->Config.GDBSymbols()) {
|
||||
|
||||
auto AOTIRCacheEntry = CTX->SyscallHandler->LookupAOTIRCacheEntry(GuestRIP);
|
||||
auto AOTIRCacheEntry = CTX->SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
|
||||
|
||||
if (AOTIRCacheEntry.Entry) {
|
||||
if (DebugData && CTX->Config.LibraryJITNaming()) {
|
||||
|
||||
+14
-13
@@ -89,13 +89,14 @@ namespace FEXCore::IR {
|
||||
|
||||
class AOTIRCaptureCache final {
|
||||
public:
|
||||
using WriteOutFn = std::function<void()>;
|
||||
|
||||
AOTIRCaptureCache(FEXCore::Context::ContextImpl *ctx) : CTX {ctx} {}
|
||||
|
||||
void FinalizeAOTIRCache();
|
||||
void AOTIRCaptureCacheWriteoutQueue_Flush();
|
||||
void AOTIRCaptureCacheWriteoutQueue_Append(const std::function<void()> &fn);
|
||||
void WriteFilesWithCode(std::function<void(const fextl::string& fileid, const fextl::string& filename)> Writer);
|
||||
void AOTIRCaptureCacheWriteoutQueue_Append(const WriteOutFn &fn);
|
||||
void WriteFilesWithCode(const Context::AOTIRCodeFileWriterFn &Writer);
|
||||
|
||||
struct PreGenerateIRFetchResult {
|
||||
FEXCore::IR::IRListView *IRList {};
|
||||
@@ -105,7 +106,7 @@ namespace FEXCore::IR {
|
||||
uint64_t Length {};
|
||||
bool GeneratedIR {};
|
||||
};
|
||||
[[nodiscard]] PreGenerateIRFetchResult PreGenerateIRFetch(uint64_t GuestRIP, FEXCore::IR::IRListView *IRList);
|
||||
[[nodiscard]] PreGenerateIRFetchResult PreGenerateIRFetch(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP, FEXCore::IR::IRListView *IRList);
|
||||
|
||||
bool PostCompileCode(FEXCore::Core::InternalThreadState *Thread,
|
||||
void* CodePtr,
|
||||
@@ -121,16 +122,16 @@ namespace FEXCore::IR {
|
||||
void UnloadAOTIRCacheEntry(AOTIRCacheEntry *Entry);
|
||||
|
||||
// Callbacks
|
||||
void SetAOTIRLoader(std::function<int(const fextl::string&)> CacheReader) {
|
||||
AOTIRLoader = CacheReader;
|
||||
void SetAOTIRLoader(Context::AOTIRLoaderCBFn CacheReader) {
|
||||
AOTIRLoader = std::move(CacheReader);
|
||||
}
|
||||
|
||||
void SetAOTIRWriter(std::function<fextl::unique_ptr<FEXCore::Context::AOTIRWriter>(const fextl::string&)> CacheWriter) {
|
||||
AOTIRWriter = CacheWriter;
|
||||
void SetAOTIRWriter(Context::AOTIRWriterCBFn CacheWriter) {
|
||||
AOTIRWriter = std::move(CacheWriter);
|
||||
}
|
||||
|
||||
void SetAOTIRRenamer(std::function<void(const fextl::string&)> CacheRenamer) {
|
||||
AOTIRRenamer = CacheRenamer;
|
||||
void SetAOTIRRenamer(Context::AOTIRRenamerCBFn CacheRenamer) {
|
||||
AOTIRRenamer = std::move(CacheRenamer);
|
||||
}
|
||||
|
||||
private:
|
||||
@@ -140,13 +141,13 @@ namespace FEXCore::IR {
|
||||
std::shared_mutex AOTIRCaptureCacheWriteoutLock;
|
||||
std::atomic<bool> AOTIRCaptureCacheWriteoutFlusing;
|
||||
|
||||
fextl::queue<std::function<void()>> AOTIRCaptureCacheWriteoutQueue;
|
||||
fextl::queue<WriteOutFn> AOTIRCaptureCacheWriteoutQueue;
|
||||
|
||||
FEXCore::IR::AOTCacheType AOTIRCache;
|
||||
|
||||
std::function<int(const fextl::string&)> AOTIRLoader;
|
||||
std::function<fextl::unique_ptr<FEXCore::Context::AOTIRWriter>(const fextl::string&)> AOTIRWriter;
|
||||
std::function<void(const fextl::string&)> AOTIRRenamer;
|
||||
Context::AOTIRLoaderCBFn AOTIRLoader;
|
||||
Context::AOTIRWriterCBFn AOTIRWriter;
|
||||
Context::AOTIRRenamerCBFn AOTIRRenamer;
|
||||
fextl::unordered_map<fextl::string, FEXCore::IR::AOTIRCaptureCacheEntry> AOTIRCaptureCacheMap;
|
||||
};
|
||||
}
|
||||
+56
-13
@@ -297,6 +297,13 @@
|
||||
],
|
||||
"DestSize": "16",
|
||||
"NumElements": "2"
|
||||
},
|
||||
"GPRPair = XGetBV GPR:$Function": {
|
||||
"Desc": ["Calls in to the XCR handler function to return emulated XCR",
|
||||
"Returns a 64bit GPR pair that fits emulated EAX, EDX respectively"
|
||||
],
|
||||
"DestSize": "8",
|
||||
"NumElements": "2"
|
||||
}
|
||||
},
|
||||
"Moves": {
|
||||
@@ -710,11 +717,19 @@
|
||||
],
|
||||
"DestSize": "std::max<uint8_t>(4, GetOpSize(_Src))"
|
||||
},
|
||||
"GPR = Abs GPR:$Src": {
|
||||
"Desc": ["Integer 2's complement absolute value",
|
||||
"Dest = std::abs(Src)",
|
||||
"Will truncate to 64 or 32bits"
|
||||
],
|
||||
"DestSize": "std::max<uint8_t>(4, GetOpSize(_Src))"
|
||||
},
|
||||
"GPR = Not GPR:$Src": {
|
||||
"Desc": ["Integer binary not",
|
||||
"op:",
|
||||
"Dest = ~Src"
|
||||
]
|
||||
],
|
||||
"DestSize": "std::max<uint8_t>(4, GetOpSize(_Src))"
|
||||
},
|
||||
"GPR = Popcount GPR:$Src": {
|
||||
"Desc": ["Population count of source register",
|
||||
@@ -767,6 +782,14 @@
|
||||
"Desc": ["Integer binary or"
|
||||
]
|
||||
},
|
||||
"GPR = Orlshl GPR:$Src1, GPR:$Src2, u8:$BitShift": {
|
||||
"Desc": ["Integer binary or with logical shift left"
|
||||
]
|
||||
},
|
||||
"GPR = Orlshr GPR:$Src1, GPR:$Src2, u8:$BitShift": {
|
||||
"Desc": ["Integer binary or with logical shift right"
|
||||
]
|
||||
},
|
||||
"GPR = Xor GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer binary exclusive or"
|
||||
]
|
||||
@@ -779,20 +802,33 @@
|
||||
"Desc": ["Integer binary AND NOT. Performs the equivalent of Src1 & ~Src2"],
|
||||
"DestSize": "std::max<uint8_t>(4, GetOpSize(_Src1))"
|
||||
},
|
||||
"GPR = Lshl GPR:$Src1, GPR:$Src2": {
|
||||
"GPR = TestNZ u8:$Size, GPR:$Src1": {
|
||||
"Desc": ["Return NZCV for a GPR, setting N and Z accordingly and zeroing C and V"],
|
||||
"DestSize": "4"
|
||||
},
|
||||
"GPR = Lshl u8:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer logical shift left"
|
||||
],
|
||||
"DestSize": "std::max<uint8_t>(4, GetOpSize(_Src1))"
|
||||
"EmitValidation": [
|
||||
"Size >= 4"
|
||||
],
|
||||
"DestSize": "Size"
|
||||
},
|
||||
"GPR = Lshr GPR:$Src1, GPR:$Src2": {
|
||||
"GPR = Lshr u8:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer logical shift right"
|
||||
],
|
||||
"DestSize": "std::max<uint8_t>(4, GetOpSize(_Src1))"
|
||||
"EmitValidation": [
|
||||
"Size >= 4"
|
||||
],
|
||||
"DestSize": "Size"
|
||||
},
|
||||
"GPR = Ashr GPR:$Src1, GPR:$Src2": {
|
||||
"GPR = Ashr u8:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer arithmetic shift right"
|
||||
],
|
||||
"DestSize": "std::max<uint8_t>(4, GetOpSize(_Src1))"
|
||||
"EmitValidation": [
|
||||
"Size >= 4"
|
||||
],
|
||||
"DestSize": "Size"
|
||||
},
|
||||
"GPR = Ror GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer rotate right"
|
||||
@@ -861,12 +897,15 @@
|
||||
],
|
||||
"DestSize": "8"
|
||||
},
|
||||
"GPR = Select CondClass:$Cond, GPR:$Cmp1, GPR:$Cmp2, GPR:$TrueVal, GPR:$FalseVal, u8:$CompareSize": {
|
||||
"GPR = Select CondClass:$Cond, SSA:$Cmp1, SSA:$Cmp2, GPR:$TrueVal, GPR:$FalseVal, u8:$CompareSize": {
|
||||
"Desc": ["Ternary selection of GPRs",
|
||||
"op:",
|
||||
"Dest = Cmp1 <Cond> Cmp2 ? TrueVal : FalseVal"
|
||||
],
|
||||
"DestSize": "std::max<uint8_t>(4, std::max<uint8_t>(GetOpSize(_TrueVal), GetOpSize(_FalseVal)))"
|
||||
"DestSize": "std::max<uint8_t>(4, std::max<uint8_t>(GetOpSize(_TrueVal), GetOpSize(_FalseVal)))",
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Cmp1) == WalkFindRegClass($Cmp2)"
|
||||
]
|
||||
},
|
||||
"GPR = Extr GPR:$Upper, GPR:$Lower, u8:$LSB": {
|
||||
"Desc": ["Concats the two GPRs to create a value that is the size of the full two GPRs",
|
||||
@@ -1293,7 +1332,13 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
},
|
||||
|
||||
"FPR = VUABDL2 u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Desc": ["Unsigned Absolute Difference Long",
|
||||
"Using the high elements of the source vectors"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / (ElementSize << 1)"
|
||||
},
|
||||
"FPR = VUShl u8:#RegisterSize, u8:#ElementSize, FPR:$Vector, FPR:$ShiftVector": {
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
@@ -1397,7 +1442,7 @@
|
||||
"DestSize": "RegisterSize"
|
||||
},
|
||||
|
||||
"GPR = VPCMPESTRX u8:$GPRSize, FPR:$LHS, FPR:$RHS, GPR:$RAX, GPR:$RDX, u8:$Control": {
|
||||
"GPR = VPCMPESTRX FPR:$LHS, FPR:$RHS, GPR:$RAX, GPR:$RDX, u16:$Control": {
|
||||
"Desc": ["Performs intermediate behavior analogous to the x86 PCMPESTRI/PCMPESTRM instruction",
|
||||
"This will return the intermediate result of a PCMPESTR-type operation, but NOT the final",
|
||||
"result. This must be derived from the intermediate result",
|
||||
@@ -1406,7 +1451,6 @@
|
||||
"flags into the upper 16-bits of the 32-bit result, as these can also be derived over the",
|
||||
"course of creating the intermediate result"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "4"
|
||||
},
|
||||
"GPR = VPCMPISTRX FPR:$LHS, FPR:$RHS, u8:$Control": {
|
||||
@@ -1418,7 +1462,6 @@
|
||||
"flags into the upper 16-bits of the 32-bit result, as these can also be derived over the",
|
||||
"course of creating the intermediate result"
|
||||
],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "4"
|
||||
}
|
||||
},
|
||||
|
||||
+8
-8
@@ -103,7 +103,7 @@ static void PrintArg(fextl::stringstream *out, IRListView const* IR, OrderedNode
|
||||
if (ArgID.IsInvalid()) {
|
||||
*out << "%Invalid";
|
||||
} else {
|
||||
*out << "%ssa" << ArgID;
|
||||
*out << "%" << std::dec << ArgID;
|
||||
if (RAData) {
|
||||
auto PhyReg = RAData->GetNodeRegister(ArgID);
|
||||
|
||||
@@ -202,8 +202,8 @@ void Dump(fextl::stringstream *out, IRListView const* IR, IR::RegisterAllocation
|
||||
|
||||
++CurrentIndent;
|
||||
AddIndent();
|
||||
*out << "(%ssa0) " << "IRHeader ";
|
||||
*out << "%ssa" << HeaderOp->Blocks.ID() << ", ";
|
||||
*out << "(%0) " << "IRHeader ";
|
||||
*out << "%" << HeaderOp->Blocks.ID() << ", ";
|
||||
*out << "#" << std::dec << HeaderOp->BlockCount << std::endl;
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
|
||||
@@ -211,10 +211,10 @@ void Dump(fextl::stringstream *out, IRListView const* IR, IR::RegisterAllocation
|
||||
auto BlockIROp = BlockHeader->C<FEXCore::IR::IROp_CodeBlock>();
|
||||
|
||||
AddIndent();
|
||||
*out << "(%ssa" << IR->GetID(BlockNode) << ") " << "CodeBlock ";
|
||||
*out << "(%" << IR->GetID(BlockNode) << ") " << "CodeBlock ";
|
||||
|
||||
*out << "%ssa" << BlockIROp->Begin.ID() << ", ";
|
||||
*out << "%ssa" << BlockIROp->Last.ID() << std::endl;
|
||||
*out << "%" << BlockIROp->Begin.ID() << ", ";
|
||||
*out << "%" << BlockIROp->Last.ID() << std::endl;
|
||||
}
|
||||
|
||||
++CurrentIndent;
|
||||
@@ -244,7 +244,7 @@ void Dump(fextl::stringstream *out, IRListView const* IR, IR::RegisterAllocation
|
||||
NumElements /= ElementSize;
|
||||
}
|
||||
|
||||
*out << "%ssa" << std::dec << ID;
|
||||
*out << "%" << std::dec << ID;
|
||||
|
||||
if (RAData) {
|
||||
auto PhyReg = RAData->GetNodeRegister(ID);
|
||||
@@ -284,7 +284,7 @@ void Dump(fextl::stringstream *out, IRListView const* IR, IR::RegisterAllocation
|
||||
NumElements = IROp->Size / ElementSize;
|
||||
}
|
||||
|
||||
*out << "(%ssa" << std::dec << ID << ' ';
|
||||
*out << "(%" << std::dec << ID << ' ';
|
||||
*out << 'i' << std::dec << (ElementSize * 8);
|
||||
if (NumElements > 1) {
|
||||
*out << 'v' << std::dec << NumElements;
|
||||
|
||||
+1
-1
@@ -413,7 +413,7 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
}
|
||||
|
||||
// Check if we are pulling in some IR from the IR Printer
|
||||
// Prints (%ssa%d) at the start of lines without a definition
|
||||
// Prints (%%d) at the start of lines without a definition
|
||||
if (Line[0] == '(') {
|
||||
size_t DefinitionEnd = fextl::string::npos;
|
||||
if ((DefinitionEnd = Line.find_first_of(')', CurrentPos)) != fextl::string::npos) {
|
||||
|
||||
+146
-2
@@ -561,6 +561,44 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
}
|
||||
*/
|
||||
|
||||
case OP_LOADMEMTSO: {
|
||||
auto Op = IROp->CW<IR::IROp_LoadMemTSO>();
|
||||
auto AddressHeader = IREmit->GetOpHeader(Op->Addr);
|
||||
|
||||
if (Op->Class == FEXCore::IR::FPRClass && AddressHeader->Op == OP_ADD && AddressHeader->Size == 8) {
|
||||
// TODO: LRCPC3 supports a vector unscaled offset like LRCPC2.
|
||||
// Support once hardware is available to use this.
|
||||
auto [OffsetType, OffsetScale, Arg0, Arg1] = MemExtendedAddressing(IREmit, IROp->Size, AddressHeader);
|
||||
|
||||
Op->OffsetType = OffsetType;
|
||||
Op->OffsetScale = OffsetScale;
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Addr_Index, Arg0); // Addr
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Offset_Index, Arg1); // Offset
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_STOREMEMTSO: {
|
||||
auto Op = IROp->CW<IR::IROp_StoreMemTSO>();
|
||||
auto AddressHeader = IREmit->GetOpHeader(Op->Addr);
|
||||
|
||||
if (Op->Class == FEXCore::IR::FPRClass && AddressHeader->Op == OP_ADD && AddressHeader->Size == 8) {
|
||||
// TODO: LRCPC3 supports a vector unscaled offset like LRCPC2.
|
||||
// Support once hardware is available to use this.
|
||||
auto [OffsetType, OffsetScale, Arg0, Arg1] = MemExtendedAddressing(IREmit, IROp->Size, AddressHeader);
|
||||
|
||||
Op->OffsetType = OffsetType;
|
||||
Op->OffsetScale = OffsetScale;
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Addr_Index, Arg0); // Addr
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Offset_Index, Arg1); // Offset
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_LOADMEM: {
|
||||
auto Op = IROp->CW<IR::IROp_LoadMem>();
|
||||
auto AddressHeader = IREmit->GetOpHeader(Op->Addr);
|
||||
@@ -652,6 +690,20 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_TESTNZ: {
|
||||
auto Op = IROp->CW<IR::IROp_TestNZ>();
|
||||
uint64_t Constant1{};
|
||||
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1)) {
|
||||
bool N = Constant1 & (1ull << ((Op->Size * 8) - 1));
|
||||
bool Z = Constant1 == 0;
|
||||
uint32_t NZVC = (N ? (1u << 31) : 0) | (Z ? (1u << 30) : 0);
|
||||
|
||||
IREmit->ReplaceWithConstant(CodeNode, NZVC);
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_OR: {
|
||||
auto Op = IROp->CW<IR::IROp_Or>();
|
||||
uint64_t Constant1{};
|
||||
@@ -669,6 +721,32 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_ORLSHL: {
|
||||
auto Op = IROp->CW<IR::IROp_Orlshl>();
|
||||
uint64_t Constant1{};
|
||||
uint64_t Constant2{};
|
||||
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) &&
|
||||
IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
uint64_t NewConstant = Constant1 | (Constant2 << Op->BitShift);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_ORLSHR: {
|
||||
auto Op = IROp->CW<IR::IROp_Orlshr>();
|
||||
uint64_t Constant1{};
|
||||
uint64_t Constant2{};
|
||||
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) &&
|
||||
IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
uint64_t NewConstant = Constant1 | (Constant2 >> Op->BitShift);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_XOR: {
|
||||
auto Op = IROp->C<IR::IROp_Xor>();
|
||||
uint64_t Constant1{};
|
||||
@@ -694,7 +772,9 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) &&
|
||||
IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
uint64_t NewConstant = (Constant1 << Constant2) & getMask(Op);
|
||||
// Shifts mask the shift amount by 63 or 31 depending on operating size;
|
||||
uint64_t ShiftMask = IROp->Size == 8 ? 63 : 31;
|
||||
uint64_t NewConstant = (Constant1 << (Constant2 & ShiftMask)) & getMask(Op);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
Changed = true;
|
||||
}
|
||||
@@ -720,7 +800,9 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) &&
|
||||
IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) {
|
||||
uint64_t NewConstant = (Constant1 >> Constant2) & getMask(Op);
|
||||
// Shifts mask the shift amount by 63 or 31 depending on operating size;
|
||||
uint64_t ShiftMask = IROp->Size == 8 ? 63 : 31;
|
||||
uint64_t NewConstant = (Constant1 >> (Constant2 & ShiftMask)) & getMask(Op);
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
Changed = true;
|
||||
}
|
||||
@@ -775,6 +857,45 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_SBFE: {
|
||||
auto Op = IROp->C<IR::IROp_Bfe>();
|
||||
uint64_t Constant;
|
||||
if (IREmit->IsValueConstant(Op->Src, &Constant)) {
|
||||
// SBFE of a constant can be converted to a constant.
|
||||
uint64_t SourceMask = (1ULL << Op->Width) - 1;
|
||||
if (Op->Width == 64)
|
||||
SourceMask = ~0ULL;
|
||||
SourceMask <<= Op->lsb;
|
||||
|
||||
int64_t NewConstant = (Constant & SourceMask) >> Op->lsb;
|
||||
NewConstant <<= 64 - Op->Width;
|
||||
NewConstant >>= 64 - Op->Width;
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_BFI: {
|
||||
auto Op = IROp->C<IR::IROp_Bfi>();
|
||||
uint64_t ConstantDest{};
|
||||
uint64_t ConstantSrc{};
|
||||
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &ConstantDest) &&
|
||||
IREmit->IsValueConstant(Op->Header.Args[1], &ConstantSrc)) {
|
||||
|
||||
uint64_t SourceMask = (1ULL << Op->Width) - 1;
|
||||
if (Op->Width == 64)
|
||||
SourceMask = ~0ULL;
|
||||
|
||||
uint64_t NewConstant = ConstantDest & ~(SourceMask << Op->lsb);
|
||||
NewConstant |= (ConstantSrc & SourceMask) << Op->lsb;
|
||||
|
||||
IREmit->ReplaceWithConstant(CodeNode, NewConstant);
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_MUL: {
|
||||
auto Op = IROp->C<IR::IROp_Mul>();
|
||||
uint64_t Constant1{};
|
||||
@@ -797,7 +918,25 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_SELECT: {
|
||||
auto Op = IROp->C<IR::IROp_Select>();
|
||||
uint64_t Constant1{};
|
||||
uint64_t Constant2{};
|
||||
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) &&
|
||||
IREmit->IsValueConstant(Op->Header.Args[1], &Constant2) &&
|
||||
Op->Cond == COND_EQ) {
|
||||
|
||||
Constant1 &= getMask(Op);
|
||||
Constant2 &= getMask(Op);
|
||||
|
||||
bool is_true = Constant1 == Constant2;
|
||||
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(Op->Header.Args[is_true ? 2 : 3]));
|
||||
Changed = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_CONDJUMP: {
|
||||
auto Op = IROp->CW<IR::IROp_CondJump>();
|
||||
|
||||
@@ -807,6 +946,11 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current
|
||||
// Fold the select into the CondJump if possible. Could handle more complex cases, too.
|
||||
if (Op->Cond.Val == COND_NEQ && IREmit->IsValueConstant(Op->Cmp2, &Constant) && Constant == 0 && Select->Op == OP_SELECT) {
|
||||
|
||||
const auto SelectCmpClass = IREmit->WalkFindRegClass(Select->Args[0]);
|
||||
if (SelectCmpClass == GPRPairClass) {
|
||||
// If the comparison class is a GPRPair then don't fold the select since it isn't free.
|
||||
break;
|
||||
}
|
||||
uint64_t Constant1{};
|
||||
uint64_t Constant2{};
|
||||
|
||||
|
||||
+91
-55
@@ -69,7 +69,9 @@ namespace {
|
||||
FEXCore::IR::RegisterClassType AccessRegClass;
|
||||
uint32_t AccessOffset;
|
||||
uint8_t AccessSize;
|
||||
FEXCore::IR::OrderedNode *Node;
|
||||
///< The last value that was loaded or stored.
|
||||
FEXCore::IR::OrderedNode *ValueNode;
|
||||
///< With a store access, the store node that is doing the operation.
|
||||
FEXCore::IR::OrderedNode *StoreNode;
|
||||
};
|
||||
|
||||
@@ -314,6 +316,36 @@ namespace {
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
// _pad2
|
||||
ContextClassification->emplace_back(ContextMemberInfo {
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, _pad2),
|
||||
sizeof(FEXCore::Core::CPUState::_pad2),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
// DeferredSignalRefCount
|
||||
ContextClassification->emplace_back(ContextMemberInfo {
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount),
|
||||
sizeof(FEXCore::Core::CPUState::DeferredSignalRefCount),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
// DeferredSignalFaultAddress
|
||||
ContextClassification->emplace_back(ContextMemberInfo {
|
||||
ContextMemberClassification {
|
||||
offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress),
|
||||
sizeof(FEXCore::Core::CPUState::DeferredSignalFaultAddress),
|
||||
},
|
||||
ACCESS_NONE,
|
||||
FEXCore::IR::InvalidClass,
|
||||
});
|
||||
|
||||
|
||||
[[maybe_unused]] size_t ClassifiedStructSize{};
|
||||
ContextClassificationInfo->Lookup.reserve(sizeof(FEXCore::Core::CPUState));
|
||||
@@ -393,6 +425,10 @@ namespace {
|
||||
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
SetAccess(Offset++, ACCESS_NONE);
|
||||
|
||||
SetAccess(Offset++, ACCESS_INVALID);
|
||||
SetAccess(Offset++, ACCESS_INVALID);
|
||||
SetAccess(Offset++, ACCESS_INVALID);
|
||||
}
|
||||
|
||||
struct BlockInfo {
|
||||
@@ -430,7 +466,7 @@ ContextMemberInfo *RCLSE::FindMemberInfo(ContextInfo *ContextClassificationInfo,
|
||||
return ContextClassificationInfo->Lookup.at(Offset);
|
||||
}
|
||||
|
||||
ContextMemberInfo *RCLSE::RecordAccess(ContextMemberInfo *Info, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size, LastAccessType AccessType, FEXCore::IR::OrderedNode *Node, FEXCore::IR::OrderedNode *StoreNode) {
|
||||
ContextMemberInfo *RCLSE::RecordAccess(ContextMemberInfo *Info, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size, LastAccessType AccessType, FEXCore::IR::OrderedNode *ValueNode, FEXCore::IR::OrderedNode *StoreNode) {
|
||||
LOGMAN_THROW_AA_FMT((Offset + Size) <= (Info->Class.Offset + Info->Class.Size), "Access to context item went over member size");
|
||||
LOGMAN_THROW_AA_FMT(Info->Accessed != ACCESS_INVALID, "Tried to access invalid member");
|
||||
|
||||
@@ -446,15 +482,15 @@ ContextMemberInfo *RCLSE::RecordAccess(ContextMemberInfo *Info, FEXCore::IR::Reg
|
||||
Info->AccessRegClass = RegClass;
|
||||
Info->AccessOffset = Offset;
|
||||
Info->AccessSize = Size;
|
||||
Info->Node = Node;
|
||||
Info->ValueNode = ValueNode;
|
||||
if (StoreNode != nullptr)
|
||||
Info->StoreNode = StoreNode;
|
||||
return Info;
|
||||
}
|
||||
|
||||
ContextMemberInfo *RCLSE::RecordAccess(ContextInfo *ClassifiedInfo, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size, LastAccessType AccessType, FEXCore::IR::OrderedNode *Node, FEXCore::IR::OrderedNode *StoreNode) {
|
||||
ContextMemberInfo *RCLSE::RecordAccess(ContextInfo *ClassifiedInfo, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size, LastAccessType AccessType, FEXCore::IR::OrderedNode *ValueNode, FEXCore::IR::OrderedNode *StoreNode) {
|
||||
ContextMemberInfo *Info = FindMemberInfo(ClassifiedInfo, Offset, Size);
|
||||
return RecordAccess(Info, RegClass, Offset, Size, AccessType, Node, StoreNode);
|
||||
return RecordAccess(Info, RegClass, Offset, Size, AccessType, ValueNode, StoreNode);
|
||||
}
|
||||
|
||||
void RCLSE::CalculateControlFlowInfo(FEXCore::IR::IREmitter *IREmit) {
|
||||
@@ -512,32 +548,32 @@ void RCLSE::CalculateControlFlowInfo(FEXCore::IR::IREmitter *IREmit) {
|
||||
* @brief This pass removes redundant pairs of storecontext and loadcontext ops
|
||||
*
|
||||
* eg.
|
||||
* %ssa26 i128 = LoadMem %ssa25 i64, 0x10
|
||||
* (%%ssa27) StoreContext %ssa26 i128, 0x10, 0xb0
|
||||
* %ssa28 i128 = LoadContext 0x10, 0x90
|
||||
* %ssa29 i128 = LoadContext 0x10, 0xb0
|
||||
* %26 i128 = LoadMem %25 i64, 0x10
|
||||
* (%%27) StoreContext %26 i128, 0x10, 0xb0
|
||||
* %28 i128 = LoadContext 0x10, 0x90
|
||||
* %29 i128 = LoadContext 0x10, 0xb0
|
||||
* Converts to
|
||||
* %ssa26 i128 = LoadMem %ssa25 i64, 0x10
|
||||
* (%%ssa27) StoreContext %ssa26 i128, 0x10, 0xb0
|
||||
* %ssa28 i128 = LoadContext 0x10, 0x90
|
||||
* %26 i128 = LoadMem %25 i64, 0x10
|
||||
* (%%27) StoreContext %26 i128, 0x10, 0xb0
|
||||
* %28 i128 = LoadContext 0x10, 0x90
|
||||
*
|
||||
* eg.
|
||||
* %ssa6 i128 = LoadContext 0x10, 0x90
|
||||
* %ssa7 i128 = LoadContext 0x10, 0x90
|
||||
* %ssa8 i128 = VXor %ssa7 i128, %ssa6 i128
|
||||
* %6 i128 = LoadContext 0x10, 0x90
|
||||
* %7 i128 = LoadContext 0x10, 0x90
|
||||
* %8 i128 = VXor %7 i128, %6 i128
|
||||
* Converts to
|
||||
* %ssa6 i128 = LoadContext 0x10, 0x90
|
||||
* %ssa7 i128 = VXor %ssa6 i128, %ssa6 i128
|
||||
* %6 i128 = LoadContext 0x10, 0x90
|
||||
* %7 i128 = VXor %6 i128, %6 i128
|
||||
*
|
||||
* eg.
|
||||
* (%%ssa189) StoreContext %ssa188 i128, 0x10, 0xa0
|
||||
* %ssa190 i128 = LoadContext 0x10, 0x90
|
||||
* %ssa192 i128 = VAdd %ssa188 i128, %ssa190 i128, 0x10, 0x4
|
||||
* (%%ssa193) StoreContext %ssa192 i128, 0x10, 0xa0
|
||||
* (%%189) StoreContext %188 i128, 0x10, 0xa0
|
||||
* %190 i128 = LoadContext 0x10, 0x90
|
||||
* %192 i128 = VAdd %188 i128, %190 i128, 0x10, 0x4
|
||||
* (%%193) StoreContext %192 i128, 0x10, 0xa0
|
||||
* Converts to
|
||||
* %ssa173 i128 = LoadContext 0x10, 0x90
|
||||
* %ssa175 i128 = VAdd %ssa172 i128, %ssa173 i128, 0x10, 0x4
|
||||
* (%%ssa176) StoreContext %ssa175 i128, 0x10, 0xa0
|
||||
* %173 i128 = LoadContext 0x10, 0x90
|
||||
* %175 i128 = VAdd %172 i128, %173 i128, 0x10, 0x4
|
||||
* (%%176) StoreContext %175 i128, 0x10, 0xa0
|
||||
|
||||
*/
|
||||
bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
@@ -590,7 +626,7 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
uint32_t LastOffset = Info->AccessOffset;
|
||||
uint8_t LastSize = Info->AccessSize;
|
||||
LastAccessType LastAccess = Info->Accessed;
|
||||
OrderedNode *LastNode = Info->Node;
|
||||
OrderedNode *LastValueNode = Info->ValueNode;
|
||||
OrderedNode *LastStoreNode = Info->StoreNode;
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, CodeNode);
|
||||
|
||||
@@ -603,7 +639,7 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
if (LastClass == GPRClass) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
|
||||
uint8_t TruncateSize = IREmit->GetOpSize(LastNode);
|
||||
uint8_t TruncateSize = IREmit->GetOpSize(LastValueNode);
|
||||
|
||||
// Did store context do an implicit truncation?
|
||||
if (IREmit->GetOpSize(LastStoreNode) < TruncateSize)
|
||||
@@ -613,20 +649,20 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
if (IROp->Size < TruncateSize)
|
||||
TruncateSize = IROp->Size;
|
||||
|
||||
if (TruncateSize != IREmit->GetOpSize(LastNode)) {
|
||||
if (TruncateSize != IREmit->GetOpSize(LastValueNode)) {
|
||||
// We need to insert an explict truncation
|
||||
LastNode = IREmit->_Bfe(Info->AccessSize, TruncateSize * 8, 0, LastNode);
|
||||
LastValueNode = IREmit->_Bfe(Info->AccessSize, TruncateSize * 8, 0, LastValueNode);
|
||||
}
|
||||
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastNode);
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
} else if (LastClass == FPRClass) {
|
||||
if (LastSize == IROp->Size && LastSize == IREmit->GetOpSize(LastNode)) {
|
||||
if (LastSize == IROp->Size && LastSize == IREmit->GetOpSize(LastValueNode)) {
|
||||
if (IsFullAccess(Info->Accessed)) {
|
||||
// LoadCtx matches StoreCtx and Node Size
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastNode);
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
}
|
||||
else {
|
||||
@@ -634,37 +670,37 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
// the vector element
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
// zext to size
|
||||
LastNode = IREmit->_VMov(IROp->Size, LastNode);
|
||||
LastValueNode = IREmit->_VMov(IROp->Size, LastValueNode);
|
||||
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastNode);
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
}
|
||||
} else if (LastSize >= IROp->Size &&
|
||||
IROp->Size == IREmit->GetOpSize(LastNode)) {
|
||||
IROp->Size == IREmit->GetOpSize(LastValueNode)) {
|
||||
// LoadCtx is <= StoreCtx and Node is LoadCtx
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastNode);
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
} else if (LastSize >= IROp->Size &&
|
||||
IROp->Size < IREmit->GetOpSize(LastNode)) {
|
||||
IROp->Size < IREmit->GetOpSize(LastValueNode)) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
// trucate to size
|
||||
LastNode = IREmit->_VMov(IROp->Size, LastNode);
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastNode);
|
||||
LastValueNode = IREmit->_VMov(IROp->Size, LastValueNode);
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
} else if (LastSize >= IROp->Size &&
|
||||
IROp->Size > IREmit->GetOpSize(LastNode)) {
|
||||
IROp->Size > IREmit->GetOpSize(LastValueNode)) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
// zext to size
|
||||
LastNode = IREmit->_VMov(IROp->Size, LastNode);
|
||||
LastValueNode = IREmit->_VMov(IROp->Size, LastValueNode);
|
||||
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastNode);
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
} else {
|
||||
//fmt::print("RCLSE: Not GPR class, missed, {}, lastS: {}, S: {}, Node S: {}\n", LastClass, LastSize, IROp->Size, IREmit->GetOpSize(LastNode));
|
||||
//fmt::print("RCLSE: Not GPR class, missed, {}, lastS: {}, S: {}, Node S: {}\n", LastClass, LastSize, IROp->Size, IREmit->GetOpSize(LastValueNode));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -674,8 +710,8 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
LastOffset == Op->Offset &&
|
||||
LastSize == IROp->Size) {
|
||||
// Did we read and then read again?
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastNode);
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
@@ -719,18 +755,18 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
auto Op = IROp->CW<IR::IROp_LoadFlag>();
|
||||
auto Info = FindMemberInfo(&LocalInfo, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag, 1);
|
||||
LastAccessType LastAccess = Info->Accessed;
|
||||
OrderedNode *LastNode = Info->Node;
|
||||
OrderedNode *LastValueNode = Info->ValueNode;
|
||||
|
||||
if (IsWriteAccess(LastAccess)) { // 1 byte so always a full write
|
||||
// If the last store matches this load value then we can replace the loaded value with the previous valid one
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, LastNode);
|
||||
RecordAccess(Info, FEXCore::IR::GPRClass, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag, 1, ACCESS_READ, LastNode);
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, LastValueNode);
|
||||
RecordAccess(Info, FEXCore::IR::GPRClass, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag, 1, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
}
|
||||
else if (IsReadAccess(LastAccess)) {
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, LastNode);
|
||||
RecordAccess(Info, FEXCore::IR::GPRClass, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag, 1, ACCESS_READ, LastNode);
|
||||
IREmit->ReplaceAllUsesWith(CodeNode, LastValueNode);
|
||||
RecordAccess(Info, FEXCore::IR::GPRClass, offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag, 1, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -169,6 +169,28 @@ bool DeadStoreElimination::Run(IREmitter *IREmit) {
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
|
||||
auto ClassifyRegisterStore = [this](Info &BlockInfo, uint32_t Offset, uint8_t Size) {
|
||||
//// GPR ////
|
||||
if (IsFullGPR(Offset, Size))
|
||||
BlockInfo.gpr.writes |= GPRBit(Offset);
|
||||
else
|
||||
BlockInfo.gpr.reads |= GPRBit(Offset);
|
||||
|
||||
//// FPR ////
|
||||
if (IsTrackedWriteFPR(Offset, Size))
|
||||
BlockInfo.fpr.writes |= FPRBit(Offset, Size);
|
||||
else
|
||||
BlockInfo.fpr.reads |= FPRBit(Offset, Size);
|
||||
};
|
||||
|
||||
auto ClassifyRegisterLoad = [this](Info &BlockInfo, uint32_t Offset, uint8_t Size) {
|
||||
//// GPR ////
|
||||
BlockInfo.gpr.reads |= GPRBit(Offset);
|
||||
|
||||
//// FPR ////
|
||||
BlockInfo.fpr.reads |= FPRBit(Offset, Size);
|
||||
};
|
||||
|
||||
//// Flags ////
|
||||
if (IROp->Op == OP_STOREFLAG) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
@@ -188,43 +210,16 @@ bool DeadStoreElimination::Run(IREmitter *IREmit) {
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
BlockInfo.flag.reads |= 1UL << Op->Flag;
|
||||
} else if (IROp->Op == OP_STORECONTEXT) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
} else if (IROp->Op == OP_STOREREGISTER) {
|
||||
auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
//// GPR ////
|
||||
if (IsFullGPR(Op->Offset, IROp->Size))
|
||||
BlockInfo.gpr.writes |= GPRBit(Op->Offset);
|
||||
else
|
||||
BlockInfo.gpr.reads |= GPRBit(Op->Offset);
|
||||
|
||||
//// FPR ////
|
||||
if (IsTrackedWriteFPR(Op->Offset, IROp->Size))
|
||||
BlockInfo.fpr.writes |= FPRBit(Op->Offset, IROp->Size);
|
||||
else
|
||||
BlockInfo.fpr.reads |= FPRBit(Op->Offset, IROp->Size);
|
||||
} else if (IROp->Op == OP_STORECONTEXTINDEXED ||
|
||||
IROp->Op == OP_LOADCONTEXTINDEXED) {
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
//// GPR ////
|
||||
// We can't track through these
|
||||
BlockInfo.gpr.reads = -1;
|
||||
|
||||
//// FPR ////
|
||||
// We can't track through these
|
||||
BlockInfo.fpr.reads = -1;
|
||||
} else if (IROp->Op == OP_LOADCONTEXT) {
|
||||
auto Op = IROp->C<IR::IROp_LoadContext>();
|
||||
ClassifyRegisterStore(BlockInfo, Op->Offset, IROp->Size);
|
||||
} else if (IROp->Op == OP_LOADREGISTER) {
|
||||
auto Op = IROp->C<IR::IROp_LoadRegister>();
|
||||
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
//// GPR ////
|
||||
BlockInfo.gpr.reads |= GPRBit(Op->Offset);
|
||||
|
||||
//// FPR ////
|
||||
BlockInfo.fpr.reads |= FPRBit(Op->Offset, IROp->Size);
|
||||
ClassifyRegisterLoad(BlockInfo, Op->Offset, IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -321,6 +316,25 @@ bool DeadStoreElimination::Run(IREmitter *IREmit) {
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
|
||||
auto RemoveDeadRegisterStore = [this](FEXCore::IR::IREmitter *IREmit, FEXCore::IR::OrderedNode *CodeNode, Info &BlockInfo, uint32_t Offset, uint8_t Size) -> bool {
|
||||
bool Changed{};
|
||||
//// GPRs ////
|
||||
// If this OP_STOREREGISTER is never read, remove it
|
||||
if (BlockInfo.gpr.kill & GPRBit(Offset)) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Changed = true;
|
||||
}
|
||||
|
||||
//// FPRs ////
|
||||
// If this OP_STOREREGISTER is never read, remove it
|
||||
if ((BlockInfo.fpr.kill & FPRBit(Offset, Size)) == FPRBit(Offset, Size) && (FPRBit(Offset, Size) != 0)) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Changed = true;
|
||||
}
|
||||
|
||||
return Changed;
|
||||
};
|
||||
|
||||
//// Flags ////
|
||||
if (IROp->Op == OP_STOREFLAG) {
|
||||
auto Op = IROp->C<IR::IROp_StoreFlag>();
|
||||
@@ -332,24 +346,12 @@ bool DeadStoreElimination::Run(IREmitter *IREmit) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Changed = true;
|
||||
}
|
||||
} else if (IROp->Op == OP_STORECONTEXT) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
} else if (IROp->Op == OP_STOREREGISTER) {
|
||||
auto Op = IROp->C<IR::IROp_StoreRegister>();
|
||||
|
||||
auto& BlockInfo = InfoMap[BlockNode];
|
||||
|
||||
//// GPRs ////
|
||||
// If this OP_STORECONTEXT is never read, remove it
|
||||
if (BlockInfo.gpr.kill & GPRBit(Op->Offset)) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Changed = true;
|
||||
}
|
||||
|
||||
//// FPRs ////
|
||||
// If this OP_STORECONTEXT is never read, remove it
|
||||
if ((BlockInfo.fpr.kill & FPRBit(Op->Offset, IROp->Size)) == FPRBit(Op->Offset, IROp->Size) && (FPRBit(Op->Offset, IROp->Size) != 0)) {
|
||||
IREmit->Remove(CodeNode);
|
||||
Changed = true;
|
||||
}
|
||||
Changed |= RemoveDeadRegisterStore(IREmit, CodeNode, BlockInfo, Op->Offset, IROp->Size);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -183,7 +183,7 @@ bool IRCompaction::Run(IREmitter *IREmit) {
|
||||
|
||||
#ifndef NDEBUG
|
||||
LOGMAN_THROW_A_FMT(NewArg.Value != UINT32_MAX,
|
||||
"Tried remapping unfound node %ssa{}", OldArg);
|
||||
"Tried remapping unfound node %{}", OldArg);
|
||||
#endif
|
||||
|
||||
LocalIROp->Args[i].NodeOffset = NewArg.Value * sizeof(OrderedNode);
|
||||
|
||||
+14
-14
@@ -82,13 +82,13 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
HadError |= OpSize == 0;
|
||||
// Does the op have a destination of size 0?
|
||||
if (OpSize == 0) {
|
||||
Errors << "%ssa" << ID << ": Had destination but with no size" << std::endl;
|
||||
Errors << "%" << ID << ": Had destination but with no size" << std::endl;
|
||||
}
|
||||
|
||||
// Does the node have zero uses? Should have been DCE'd
|
||||
if (CodeNode->GetUses() == 0) {
|
||||
HadWarning |= true;
|
||||
Warnings << "%ssa" << ID << ": Destination created but had no uses" << std::endl;
|
||||
Warnings << "%" << ID << ": Destination created but had no uses" << std::endl;
|
||||
}
|
||||
|
||||
if (RAData) {
|
||||
@@ -101,20 +101,20 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
// If no register class was assigned
|
||||
if (AssignedClass == IR::InvalidClass) {
|
||||
HadError |= true;
|
||||
Errors << "%ssa" << ID << ": Had destination but with no register class assigned" << std::endl;
|
||||
Errors << "%" << ID << ": Had destination but with no register class assigned" << std::endl;
|
||||
}
|
||||
|
||||
// If no physical register was assigned
|
||||
if (PhyReg.Reg == IR::InvalidReg) {
|
||||
HadError |= true;
|
||||
Errors << "%ssa" << ID << ": Had destination but with no register assigned" << std::endl;
|
||||
Errors << "%" << ID << ": Had destination but with no register assigned" << std::endl;
|
||||
}
|
||||
|
||||
// Assigned class wasn't the expected class and it is a non-complex op
|
||||
if (AssignedClass != ExpectedClass &&
|
||||
ExpectedClass != IR::ComplexClass) {
|
||||
HadWarning |= true;
|
||||
Warnings << "%ssa" << ID << ": Destination had register class " << AssignedClass.Val << " When register class " << ExpectedClass.Val << " Was expected" << std::endl;
|
||||
Warnings << "%" << ID << ": Destination had register class " << AssignedClass.Val << " When register class " << ExpectedClass.Val << " Was expected" << std::endl;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -128,12 +128,12 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
// Was an argument defined after this node?
|
||||
if (ArgID >= ID) {
|
||||
HadError |= true;
|
||||
Errors << "%ssa" << ID << ": Arg[" << i << "] has definition after use at %ssa" << ArgID << std::endl;
|
||||
Errors << "%" << ID << ": Arg[" << i << "] has definition after use at %" << ArgID << std::endl;
|
||||
}
|
||||
|
||||
if (ArgID.IsValid() && !NodeIsLive.Get(ArgID.Value)) {
|
||||
HadError |= true;
|
||||
Errors << "%ssa" << ID << ": Arg[" << i << "] references dead %ssa" << ArgID << std::endl;
|
||||
Errors << "%" << ID << ": Arg[" << i << "] references dead %" << ArgID << std::endl;
|
||||
}
|
||||
|
||||
if (ArgID.IsValid()) {
|
||||
@@ -162,7 +162,7 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
|
||||
if (TrueTargetOp->Op != OP_CODEBLOCK) {
|
||||
HadError |= true;
|
||||
Errors << "CondJump %ssa" << ID << ": True Target Jumps to Op that isn't the begining of a block" << std::endl;
|
||||
Errors << "CondJump %" << ID << ": True Target Jumps to Op that isn't the begining of a block" << std::endl;
|
||||
}
|
||||
else {
|
||||
auto Block = OffsetToBlockMap.try_emplace(Op->TrueBlock.ID()).first;
|
||||
@@ -171,7 +171,7 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
|
||||
if (FalseTargetOp->Op != OP_CODEBLOCK) {
|
||||
HadError |= true;
|
||||
Errors << "CondJump %ssa" << ID << ": False Target Jumps to Op that isn't the begining of a block" << std::endl;
|
||||
Errors << "CondJump %" << ID << ": False Target Jumps to Op that isn't the begining of a block" << std::endl;
|
||||
}
|
||||
else {
|
||||
auto Block = OffsetToBlockMap.try_emplace(Op->FalseBlock.ID()).first;
|
||||
@@ -188,7 +188,7 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
FEXCore::IR::IROp_Header const *TargetOp = CurrentIR.GetOp<IROp_Header>(TargetNode);
|
||||
if (TargetOp->Op != OP_CODEBLOCK) {
|
||||
HadError |= true;
|
||||
Errors << "Jump %ssa" << ID << ": Jump to Op that isn't the begining of a block" << std::endl;
|
||||
Errors << "Jump %" << ID << ": Jump to Op that isn't the begining of a block" << std::endl;
|
||||
}
|
||||
else {
|
||||
auto Block = OffsetToBlockMap.try_emplace(Op->Header.Args[0].ID()).first;
|
||||
@@ -206,7 +206,7 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
size_t NumSuccessors = CurrentBlock->Successors.size();
|
||||
if (NumSuccessors > 2) {
|
||||
HadError |= true;
|
||||
Errors << "%ssa" << BlockID << " Has " << NumSuccessors << " successors which is too many" << std::endl;
|
||||
Errors << "%" << BlockID << " Has " << NumSuccessors << " successors which is too many" << std::endl;
|
||||
}
|
||||
|
||||
{
|
||||
@@ -222,7 +222,7 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
auto Op = GetOp(CodeCurrent);
|
||||
if (Op != IR::OP_ENDBLOCK) {
|
||||
HadError |= true;
|
||||
Errors << "%ssa" << BlockID << " Failed to end block with EndBlock" << std::endl;
|
||||
Errors << "%" << BlockID << " Failed to end block with EndBlock" << std::endl;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -233,7 +233,7 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
auto Op = GetOp(CodeCurrent);
|
||||
if (!IsBlockExit(Op)) {
|
||||
HadError |= true;
|
||||
Errors << "%ssa" << BlockID << " Didn't have a block exit IR op as its last instruction" << std::endl;
|
||||
Errors << "%" << BlockID << " Didn't have a block exit IR op as its last instruction" << std::endl;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -243,7 +243,7 @@ bool IRValidation::Run(IREmitter *IREmit) {
|
||||
auto [Node, IROp] = CurrentIR.at(IR::NodeID{i})();
|
||||
if (Node->NumUses != Uses[i] && IROp->Op != OP_CODEBLOCK && IROp->Op != OP_IRHEADER) {
|
||||
HadError |= true;
|
||||
Errors << "%ssa" << i << " Has " << Uses[i] << " Uses, but reports " << Node->NumUses << std::endl;
|
||||
Errors << "%" << i << " Has " << Uses[i] << " Uses, but reports " << Node->NumUses << std::endl;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -25,14 +25,9 @@ private:
|
||||
};
|
||||
|
||||
bool LongDivideEliminationPass::IsZeroOp(IREmitter *IREmit, OrderedNodeWrapper Arg) {
|
||||
auto IROp = IREmit->GetOpHeader(Arg);
|
||||
uint64_t Value;
|
||||
|
||||
// XOR based zero
|
||||
if (IROp->Op == OP_XOR) {
|
||||
return IROp->Args[0] == IROp->Args[1];
|
||||
}
|
||||
else if (IREmit->IsValueConstant(Arg, &Value)) {
|
||||
if (IREmit->IsValueConstant(Arg, &Value)) {
|
||||
// Zero constant based zero op
|
||||
return Value == 0;
|
||||
}
|
||||
@@ -87,8 +82,7 @@ bool LongDivideEliminationPass::Run(IREmitter *IREmit) {
|
||||
else if (IROp->Op == OP_LUDIV ||
|
||||
IROp->Op == OP_LUREM) {
|
||||
auto Op = IROp->C<IR::IROp_LUDiv>();
|
||||
// Check upper Op to see if it came from a xor zeroing op
|
||||
// XOR: Result = _Xor(Dest, Src);
|
||||
// Check upper Op to see if it came from a zeroing op
|
||||
// If it does then it we only need a 64bit UDIV
|
||||
if (IsZeroOp(IREmit, Op->Upper)) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
|
||||
@@ -47,7 +47,7 @@ bool PhiValidation::Run(IREmitter *IREmit) {
|
||||
// If we have found a non-phi IR op and then had a Phi or PhiValue value then this is a programming mistake
|
||||
// PHI values MUST be defined at the top of the block only
|
||||
HadError |= true;
|
||||
Errors << "Phi %ssa" << CurrentIR.GetID(CodeNode) << ": Was defined after non-phi operations. Which is invalid!" << std::endl;
|
||||
Errors << "Phi %" << CurrentIR.GetID(CodeNode) << ": Was defined after non-phi operations. Which is invalid!" << std::endl;
|
||||
}
|
||||
|
||||
// Check all the phi values to ensure they have the same type
|
||||
|
||||
@@ -297,28 +297,28 @@ bool RAValidation::Run(IREmitter *IREmit) {
|
||||
auto CurrentSSAAtReg = BlockRegState.Get(PhyReg);
|
||||
if (CurrentSSAAtReg == RegState::InvalidReg) {
|
||||
HadError |= true;
|
||||
Errors << fextl::fmt::format("%ssa{}: Arg[{}] unknown Reg: {}, class: {}\n", ID, i, PhyReg.Reg, PhyReg.Class);
|
||||
Errors << fextl::fmt::format("%{}: Arg[{}] unknown Reg: {}, class: {}\n", ID, i, PhyReg.Reg, PhyReg.Class);
|
||||
} else if (CurrentSSAAtReg == RegState::CorruptedPair) {
|
||||
HadError |= true;
|
||||
|
||||
auto Lower = BlockRegState.Get(PhysicalRegister(GPRClass, uint8_t(PhyReg.Reg*2) + 1));
|
||||
auto Upper = BlockRegState.Get(PhysicalRegister(GPRClass, PhyReg.Reg*2 + 1));
|
||||
|
||||
Errors << fextl::fmt::format("%ssa{}: Arg[{}] expects paired reg{} to contain %ssa{}, but it actually contains {{%ssa{}, %ssa{}}}\n",
|
||||
Errors << fextl::fmt::format("%{}: Arg[{}] expects paired reg{} to contain %{}, but it actually contains {{%{}, %{}}}\n",
|
||||
ID, i, PhyReg.Reg, ArgID, Lower, Upper);
|
||||
} else if (CurrentSSAAtReg == RegState::UninitializedValue) {
|
||||
HadError |= true;
|
||||
|
||||
Errors << fextl::fmt::format("%ssa{}: Arg[{}] expects reg{} to contain %ssa{}, but it is uninitialized\n",
|
||||
Errors << fextl::fmt::format("%{}: Arg[{}] expects reg{} to contain %{}, but it is uninitialized\n",
|
||||
ID, i, PhyReg.Reg, ArgID);
|
||||
} else if (CurrentSSAAtReg == RegState::ClobberedValue) {
|
||||
HadError |= true;
|
||||
|
||||
Errors << fextl::fmt::format("%ssa{}: Arg[{}] expects reg{} to contain %ssa{}, but contents vary depending on control flow\n",
|
||||
Errors << fextl::fmt::format("%{}: Arg[{}] expects reg{} to contain %{}, but contents vary depending on control flow\n",
|
||||
ID, i, PhyReg.Reg, ArgID);
|
||||
} else if (CurrentSSAAtReg != ArgID) {
|
||||
HadError |= true;
|
||||
Errors << fextl::fmt::format("%ssa{}: Arg[{}] expects reg{} to contain %ssa{}, but it actually contains %ssa{}\n",
|
||||
Errors << fextl::fmt::format("%{}: Arg[{}] expects reg{} to contain %{}, but it actually contains %{}\n",
|
||||
ID, i, PhyReg.Reg, ArgID, CurrentSSAAtReg);
|
||||
}
|
||||
};
|
||||
@@ -343,15 +343,15 @@ bool RAValidation::Run(IREmitter *IREmit) {
|
||||
|
||||
if (Value == RegState::UninitializedValue) {
|
||||
HadError |= true;
|
||||
Errors << fextl::fmt::format("%ssa{}: FillRegister expected %ssa{} in Slot {}, but was undefined in at least one control flow path\n",
|
||||
Errors << fextl::fmt::format("%{}: FillRegister expected %{} in Slot {}, but was undefined in at least one control flow path\n",
|
||||
ID, ExpectedValue, FillRegister->Slot);
|
||||
} else if (Value == RegState::ClobberedValue) {
|
||||
HadError |= true;
|
||||
Errors << fextl::fmt::format("%ssa{}: FillRegister expected %ssa{} in Slot {}, but contents vary depending on control flow\n",
|
||||
Errors << fextl::fmt::format("%{}: FillRegister expected %{} in Slot {}, but contents vary depending on control flow\n",
|
||||
ID, ExpectedValue, FillRegister->Slot);
|
||||
} else if (Value != ExpectedValue) {
|
||||
HadError |= true;
|
||||
Errors << fextl::fmt::format("%ssa{}: FillRegister expected %ssa{} in Slot {}, but it actually contains %ssa{}\n",
|
||||
Errors << fextl::fmt::format("%{}: FillRegister expected %{} in Slot {}, but it actually contains %{}\n",
|
||||
ID, ExpectedValue, FillRegister->Slot, Value);
|
||||
}
|
||||
break;
|
||||
|
||||
@@ -494,7 +494,7 @@ namespace {
|
||||
const auto ArgNode = Arg.ID();
|
||||
auto& ArgNodeLiveRange = LiveRanges[ArgNode.Value];
|
||||
LOGMAN_THROW_AA_FMT(ArgNodeLiveRange.Begin.Value != UINT32_MAX,
|
||||
"%ssa{} used by %ssa{} before defined?", ArgNode, Node);
|
||||
"%{} used by %{} before defined?", ArgNode, Node);
|
||||
|
||||
const auto ArgNodeBlockID = Graph->Nodes[ArgNode.Value].Head.BlockID;
|
||||
if (ArgNodeBlockID == BlockNodeID) {
|
||||
@@ -1311,7 +1311,7 @@ namespace {
|
||||
|
||||
if (!CurrentNodes.contains(InterferenceNode)) {
|
||||
InterferenceIdToSpill = InterferenceNode;
|
||||
LogMan::Msg::DFmt("Panic spilling %ssa{}, Live Range[{}, {})", InterferenceIdToSpill, InterferenceLiveRange->Begin, InterferenceLiveRange->End);
|
||||
LogMan::Msg::DFmt("Panic spilling %{}, Live Range[{}, {})", InterferenceIdToSpill, InterferenceLiveRange->Begin, InterferenceLiveRange->End);
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
@@ -1320,14 +1320,14 @@ namespace {
|
||||
|
||||
if (InterferenceIdToSpill.IsInvalid()) {
|
||||
int j = 0;
|
||||
LogMan::Msg::DFmt("node %ssa{}, was dumped in to virtual reg {}. Live Range[{}, {})",
|
||||
LogMan::Msg::DFmt("node %{}, was dumped in to virtual reg {}. Live Range[{}, {})",
|
||||
CurrentLocation, -1,
|
||||
OpLiveRange->Begin, OpLiveRange->End);
|
||||
|
||||
RegisterNode->Interferences.Iterate([&](IR::NodeID InterferenceNode) {
|
||||
auto *InterferenceLiveRange = &LiveRanges[InterferenceNode.Value];
|
||||
|
||||
LogMan::Msg::DFmt("\tInt{}: %ssa{} Remat: {} [{}, {})", j++, InterferenceNode, InterferenceLiveRange->RematCost, InterferenceLiveRange->Begin, InterferenceLiveRange->End);
|
||||
LogMan::Msg::DFmt("\tInt{}: %{} Remat: {} [{}, {})", j++, InterferenceNode, InterferenceLiveRange->RematCost, InterferenceLiveRange->Begin, InterferenceLiveRange->End);
|
||||
});
|
||||
}
|
||||
LOGMAN_THROW_A_FMT(InterferenceIdToSpill.IsValid(), "Couldn't find Node to spill");
|
||||
@@ -1395,7 +1395,7 @@ namespace {
|
||||
auto FirstUseLocation = FindFirstUse(IREmit, ConstantNode, NextIter, NodeIterator::Invalid());
|
||||
|
||||
LOGMAN_THROW_A_FMT(FirstUseLocation != IR::NodeIterator::Invalid(),
|
||||
"At %ssa{} Spilling Op %ssa{} but Failure to find op use",
|
||||
"At %{} Spilling Op %{} but Failure to find op use",
|
||||
Node, *InterferenceNode);
|
||||
|
||||
if (FirstUseLocation != IR::NodeIterator::Invalid()) {
|
||||
@@ -1454,7 +1454,7 @@ namespace {
|
||||
auto FirstUseLocation = FindFirstUse(IREmit, InterferenceOrderedNode, FirstIter, NodeIterator::Invalid());
|
||||
|
||||
LOGMAN_THROW_A_FMT(FirstUseLocation != NodeIterator::Invalid(),
|
||||
"At %ssa{} Spilling Op %ssa{} but Failure to find op use",
|
||||
"At %{} Spilling Op %{} but Failure to find op use",
|
||||
Node, *InterferenceNode);
|
||||
|
||||
if (FirstUseLocation != IR::NodeIterator::Invalid()) {
|
||||
|
||||
@@ -107,18 +107,18 @@ bool ValueDominanceValidation::Run(IREmitter *IREmit) {
|
||||
// then it must only be declared prior to this instruction
|
||||
// Eg: Valid
|
||||
// CodeBlock_1:
|
||||
// %ssa_1 = Load
|
||||
// %ssa_2 = Load
|
||||
// %ssa_3 = <Op> %ssa_1, %ssa_2
|
||||
// %_1 = Load
|
||||
// %_2 = Load
|
||||
// %_3 = <Op> %_1, %_2
|
||||
//
|
||||
// Eg: Invalid
|
||||
// CodeBlock_1:
|
||||
// %ssa_1 = Load
|
||||
// %ssa_2 = <Op> %ssa_1, %ssa_3
|
||||
// %ssa_3 = Load
|
||||
// %_1 = Load
|
||||
// %_2 = <Op> %_1, %_3
|
||||
// %_3 = Load
|
||||
if (Arg.ID() > CodeID) {
|
||||
HadError |= true;
|
||||
Errors << "Inst %ssa" << CodeID << ": Arg[" << i << "] %ssa" << Arg.ID() << " definition does not dominate this use!" << std::endl;
|
||||
Errors << "Inst %" << CodeID << ": Arg[" << i << "] %" << Arg.ID() << " definition does not dominate this use!" << std::endl;
|
||||
}
|
||||
}
|
||||
else if (Arg.ID() < BlockIROp->Begin.ID()) {
|
||||
@@ -127,21 +127,21 @@ bool ValueDominanceValidation::Run(IREmitter *IREmit) {
|
||||
|
||||
// Eg: Valid
|
||||
// CodeBlock_1:
|
||||
// %ssa_1 = Load
|
||||
// %ssa_2 = Load
|
||||
// %_1 = Load
|
||||
// %_2 = Load
|
||||
// Jump %CodeBlock_2
|
||||
//
|
||||
// CodeBlock_2:
|
||||
// %ssa_3 = <Op> %ssa_1, %ssa_2
|
||||
// %_3 = <Op> %_1, %_2
|
||||
//
|
||||
// Eg: Invalid
|
||||
// CodeBlock_1:
|
||||
// %ssa_1 = Load
|
||||
// %ssa_2 = Load
|
||||
// %_1 = Load
|
||||
// %_2 = Load
|
||||
// Jump %CodeBlock_3
|
||||
//
|
||||
// CodeBlock_2:
|
||||
// %ssa_3 = <Op> %ssa_1, %ssa_2
|
||||
// %_3 = <Op> %_1, %_2
|
||||
//
|
||||
// CodeBlock_3:
|
||||
// ...
|
||||
@@ -171,26 +171,26 @@ bool ValueDominanceValidation::Run(IREmitter *IREmit) {
|
||||
FoundPredDefine = true;
|
||||
break;
|
||||
}
|
||||
Errors << "\tChecking Pred %ssa" << CurrentIR.GetID(Pred) << std::endl;
|
||||
Errors << "\tChecking Pred %" << CurrentIR.GetID(Pred) << std::endl;
|
||||
}
|
||||
|
||||
if (!FoundPredDefine) {
|
||||
HadError |= true;
|
||||
Errors << "Inst %ssa" << CodeID << ": Arg[" << i << "] %ssa" << Arg.ID() << " definition does not dominate this use! But was defined before this block!" << std::endl;
|
||||
Errors << "Inst %" << CodeID << ": Arg[" << i << "] %" << Arg.ID() << " definition does not dominate this use! But was defined before this block!" << std::endl;
|
||||
}
|
||||
}
|
||||
else if (Arg.ID() > BlockIROp->Last.ID()) {
|
||||
// If this SSA argument is defined AFTER this block then it is just completely broken
|
||||
// Eg: Invalid
|
||||
// CodeBlock_1:
|
||||
// %ssa_1 = Load
|
||||
// %ssa_2 = <Op> %ssa_1, %ssa_3
|
||||
// %_1 = Load
|
||||
// %_2 = <Op> %_1, %_3
|
||||
// Jump %CodeBlock_2
|
||||
//
|
||||
// CodeBlock_2:
|
||||
// %ssa_3 = Load
|
||||
// %_3 = Load
|
||||
HadError |= true;
|
||||
Errors << "Inst %ssa" << CodeID << ": Arg[" << i << "] %ssa" << Arg.ID() << " definition does not dominate this use!" << std::endl;
|
||||
Errors << "Inst %" << CodeID << ": Arg[" << i << "] %" << Arg.ID() << " definition does not dominate this use!" << std::endl;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+12
@@ -340,5 +340,17 @@ namespace FEXCore::Allocator {
|
||||
::munmap(Region.Ptr, Region.Size);
|
||||
}
|
||||
}
|
||||
|
||||
void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) {
|
||||
if (Alloc64) {
|
||||
Alloc64->LockBeforeFork(Thread);
|
||||
}
|
||||
}
|
||||
|
||||
void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child) {
|
||||
if (Alloc64) {
|
||||
Alloc64->UnlockAfterFork(Thread, Child);
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
#pragma once
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace FEXCore::Allocator {
|
||||
void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread);
|
||||
void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child);
|
||||
}
|
||||
+28
-5
@@ -5,7 +5,7 @@
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/MathUtils.h>
|
||||
#include <FEXCore/fextl/sstream.h>
|
||||
#include <FEXHeaderUtils/ScopedSignalMask.h>
|
||||
#include <FEXCore/Utils/DeferredSignalMutex.h>
|
||||
#include <FEXHeaderUtils/Syscalls.h>
|
||||
#include <FEXHeaderUtils/TypeDefines.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
@@ -29,6 +29,16 @@
|
||||
|
||||
namespace Alloc::OSAllocator {
|
||||
|
||||
thread_local FEXCore::Core::InternalThreadState *TLSThread{};
|
||||
|
||||
void RegisterTLSData(FEXCore::Core::InternalThreadState *Thread) {
|
||||
TLSThread = Thread;
|
||||
}
|
||||
|
||||
void UninstallTLSData(FEXCore::Core::InternalThreadState *Thread) {
|
||||
TLSThread = nullptr;
|
||||
}
|
||||
|
||||
class OSAllocator_64Bit final : public Alloc::HostAllocator {
|
||||
public:
|
||||
OSAllocator_64Bit();
|
||||
@@ -39,6 +49,19 @@ namespace Alloc::OSAllocator {
|
||||
void *Mmap(void *addr, size_t length, int prot, int flags, int fd, off_t offset) override;
|
||||
int Munmap(void *addr, size_t length) override;
|
||||
|
||||
void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) override {
|
||||
AllocationMutex.lock();
|
||||
}
|
||||
|
||||
void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child) override {
|
||||
if (Child) {
|
||||
AllocationMutex.StealAndDropActiveLocks();
|
||||
}
|
||||
else {
|
||||
AllocationMutex.unlock();
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
// Upper bound is the maximum virtual address space of the host processor
|
||||
uintptr_t UPPER_BOUND = (1ULL << 57);
|
||||
@@ -129,7 +152,7 @@ namespace Alloc::OSAllocator {
|
||||
LiveRegionListType *LiveRegions{};
|
||||
|
||||
Alloc::ForwardOnlyIntrusiveArenaAllocator *ObjectAlloc{};
|
||||
std::mutex AllocationMutex{};
|
||||
FEXCore::ForkableUniqueMutex AllocationMutex;
|
||||
void DetermineVASize();
|
||||
|
||||
LiveVMARegion *MakeRegionActive(ReservedRegionListType::iterator ReservedIterator, uint64_t UsedSize) {
|
||||
@@ -248,7 +271,7 @@ void *OSAllocator_64Bit::Mmap(void *addr, size_t length, int prot, int flags, in
|
||||
size_t NumberOfPages = length / FHU::FEX_PAGE_SIZE;
|
||||
|
||||
// This needs a mutex to be thread safe
|
||||
FHU::ScopedSignalMaskWithMutex lk(AllocationMutex);
|
||||
FEXCore::ScopedPotentialDeferredSignalWithForkableMutex lk(AllocationMutex, TLSThread);
|
||||
|
||||
uint64_t AllocatedOffset{};
|
||||
LiveVMARegion *LiveRegion{};
|
||||
@@ -436,7 +459,7 @@ int OSAllocator_64Bit::Munmap(void *addr, size_t length) {
|
||||
}
|
||||
|
||||
// This needs a mutex to be thread safe
|
||||
FHU::ScopedSignalMaskWithMutex lk(AllocationMutex);
|
||||
FEXCore::ScopedPotentialDeferredSignalWithForkableMutex lk(AllocationMutex, TLSThread);
|
||||
|
||||
length = FEXCore::AlignUp(length, FHU::FEX_PAGE_SIZE);
|
||||
|
||||
@@ -561,7 +584,7 @@ OSAllocator_64Bit::OSAllocator_64Bit() {
|
||||
|
||||
OSAllocator_64Bit::~OSAllocator_64Bit() {
|
||||
// This needs a mutex to be thread safe
|
||||
FHU::ScopedSignalMaskWithMutex lk(AllocationMutex);
|
||||
FEXCore::ScopedPotentialDeferredSignalWithForkableMutex lk(AllocationMutex, TLSThread);
|
||||
|
||||
// Walk the pages and deallocate
|
||||
// First walk the live regions
|
||||
|
||||
@@ -6,6 +6,10 @@
|
||||
#include <cstdint>
|
||||
#include <sys/types.h>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
}
|
||||
|
||||
namespace Alloc {
|
||||
// HostAllocator is just a page pased slab allocator
|
||||
// Similar to mmap and munmap only mapping at the page level
|
||||
@@ -18,6 +22,9 @@ namespace Alloc {
|
||||
|
||||
virtual void *Mmap(void *addr, size_t length, int prot, int flags, int fd, off_t offset) { return nullptr; }
|
||||
virtual int Munmap(void *addr, size_t length) { return -1; }
|
||||
|
||||
virtual void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) {}
|
||||
virtual void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child) {}
|
||||
};
|
||||
|
||||
class GlobalAllocator {
|
||||
@@ -36,5 +43,7 @@ namespace Alloc {
|
||||
}
|
||||
|
||||
namespace Alloc::OSAllocator {
|
||||
void RegisterTLSData(FEXCore::Core::InternalThreadState *Thread);
|
||||
void UninstallTLSData(FEXCore::Core::InternalThreadState *Thread);
|
||||
fextl::unique_ptr<Alloc::HostAllocator> Create64BitAllocator();
|
||||
}
|
||||
+11
-15
@@ -2043,8 +2043,8 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
uint32_t Size = (Instr & 0xC000'0000) >> 30;
|
||||
uint32_t AddrReg = (Instr >> 5) & 0x1F;
|
||||
uint32_t DataReg = Instr & 0x1F;
|
||||
if ((Instr & 0x3F'FF'FC'00) == 0x08'DF'FC'00 || // LDAR*
|
||||
(Instr & 0x3F'FF'FC'00) == 0x38'BF'C0'00) { // LDAPR*
|
||||
if ((Instr & LDAXR_MASK) == LDAR_INST || // LDAR*
|
||||
(Instr & LDAXR_MASK) == LDAPR_INST) { // LDAPR*
|
||||
if (ParanoidTSO) {
|
||||
if (ArchHelpers::Arm64::HandleAtomicLoad(Instr, GPRs, 0)) {
|
||||
// Skip this instruction now
|
||||
@@ -2060,15 +2060,14 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
LDR |= Size << 30;
|
||||
LDR |= AddrReg << 5;
|
||||
LDR |= DataReg;
|
||||
PC[-1] = DMB;
|
||||
PC[0] = LDR;
|
||||
PC[1] = DMB;
|
||||
PC[1] = DMB_LD; // Back-patch the half-barrier.
|
||||
ClearICache(&PC[-1], 16);
|
||||
// Back up one instruction and have another go
|
||||
return std::make_pair(true, -4);
|
||||
// With the instruction modified, now execute again.
|
||||
return std::make_pair(true, 0);
|
||||
}
|
||||
}
|
||||
else if ( (Instr & 0x3F'FF'FC'00) == 0x08'9F'FC'00) { // STLR*
|
||||
else if ( (Instr & LDAXR_MASK) == STLR_INST) { // STLR*
|
||||
if (ParanoidTSO) {
|
||||
if (ArchHelpers::Arm64::HandleAtomicStore(Instr, GPRs, 0)) {
|
||||
// Skip this instruction now
|
||||
@@ -2084,9 +2083,8 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
STR |= Size << 30;
|
||||
STR |= AddrReg << 5;
|
||||
STR |= DataReg;
|
||||
PC[-1] = DMB;
|
||||
PC[-1] = DMB; // Back-patch the half-barrier.
|
||||
PC[0] = STR;
|
||||
PC[1] = DMB;
|
||||
ClearICache(&PC[-1], 16);
|
||||
// Back up one instruction and have another go
|
||||
return std::make_pair(true, -4);
|
||||
@@ -2111,12 +2109,11 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
LDUR |= AddrReg << 5;
|
||||
LDUR |= DataReg;
|
||||
LDUR |= Instr & (0b1'1111'1111 << 9);
|
||||
PC[-1] = DMB;
|
||||
PC[0] = LDUR;
|
||||
PC[1] = DMB;
|
||||
PC[1] = DMB_LD; // Back-patch the half-barrier.
|
||||
ClearICache(&PC[-1], 16);
|
||||
// Back up one instruction and have another go
|
||||
return std::make_pair(true, -4);
|
||||
// With the instruction modified, now execute again.
|
||||
return std::make_pair(true, 0);
|
||||
}
|
||||
}
|
||||
else if ((Instr & RCPC2_MASK) == STLUR_INST) { // STLUR*
|
||||
@@ -2138,9 +2135,8 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
|
||||
STUR |= AddrReg << 5;
|
||||
STUR |= DataReg;
|
||||
STUR |= Instr & (0b1'1111'1111 << 9);
|
||||
PC[-1] = DMB;
|
||||
PC[-1] = DMB; // Back-patch the half-barrier.
|
||||
PC[0] = STUR;
|
||||
PC[1] = DMB;
|
||||
ClearICache(&PC[-1], 16);
|
||||
// Back up one instruction and have another go
|
||||
return std::make_pair(true, -4);
|
||||
|
||||
@@ -22,6 +22,11 @@ bool HandleCASAL(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
bool HandleAtomicMemOp(void *_ucontext, void *_info, uint32_t Instr) {
|
||||
ERROR_AND_DIE_FMT("HandleAtomicMemOp Not Implemented");
|
||||
}
|
||||
|
||||
std::pair<bool, int32_t> HandleUnalignedAccess(bool ParanoidTSO, uintptr_t ProgramCounter, uint64_t *GPRs) {
|
||||
ERROR_AND_DIE_FMT("HandleAtomicMemOp Not Implemented");
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
}
|
||||
+10
-10
@@ -1,4 +1,5 @@
|
||||
#include <FEXCore/Config/Config.h>
|
||||
#include <FEXCore/Utils/File.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Telemetry.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
@@ -22,6 +23,7 @@ namespace FEXCore::Telemetry {
|
||||
"32bit CAS Tear",
|
||||
"64bit CAS Tear",
|
||||
"128bit CAS Tear",
|
||||
"Crash mask",
|
||||
};
|
||||
void Initialize() {
|
||||
auto DataDirectory = Config::GetDataDirectory();
|
||||
@@ -35,7 +37,6 @@ namespace FEXCore::Telemetry {
|
||||
}
|
||||
|
||||
void Shutdown(fextl::string const &ApplicationName) {
|
||||
#ifndef _WIN32
|
||||
auto DataDirectory = Config::GetDataDirectory();
|
||||
DataDirectory += "Telemetry/" + ApplicationName + ".telem";
|
||||
|
||||
@@ -45,23 +46,22 @@ namespace FEXCore::Telemetry {
|
||||
FHU::Filesystem::CopyFile(DataDirectory, Backup, FHU::Filesystem::CopyOptions::OVERWRITE_EXISTING);
|
||||
}
|
||||
|
||||
constexpr int USER_PERMS = S_IRWXU | S_IRWXG | S_IRWXO;
|
||||
int fd = open(DataDirectory.c_str(), O_CREAT | O_WRONLY | O_TRUNC | O_CLOEXEC, USER_PERMS);
|
||||
auto File = FEXCore::File::File(DataDirectory.c_str(),
|
||||
FEXCore::File::FileModes::WRITE |
|
||||
FEXCore::File::FileModes::CREATE |
|
||||
FEXCore::File::FileModes::TRUNCATE);
|
||||
|
||||
if (fd != -1) {
|
||||
if (File.IsValid()) {
|
||||
for (size_t i = 0; i < TelemetryType::TYPE_LAST; ++i) {
|
||||
auto &Name = TelemetryNames.at(i);
|
||||
auto &Data = TelemetryValues.at(i);
|
||||
auto Output = fextl::fmt::format("{}: {}\n", Name, *Data);
|
||||
write(fd, Output.c_str(), Output.size());
|
||||
fextl::fmt::print(File, "{}: {}\n", Name, *Data);
|
||||
}
|
||||
fsync(fd);
|
||||
close(fd);
|
||||
File.Flush();
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
Value &GetObject(TelemetryType Type) {
|
||||
Value &GetTelemetryValue(TelemetryType Type) {
|
||||
return TelemetryValues.at(Type);
|
||||
}
|
||||
#endif
|
||||
|
||||
Vendored
+10
-10
@@ -32,17 +32,17 @@ SSA is quite nice to work with when translating the x86-64 code to the IR, when
|
||||
* Read the python generation file to determine the extent of what it can do
|
||||
|
||||
## IR function considerations
|
||||
The first SSA node is a special case node that is considered invalid. This means %ssa0 will always be invalid for "null" node checks
|
||||
The first real SSA node also has to be a IRHeader node. This means it is safe to assume that %ssa1 will always be an IRHeader.
|
||||
The first SSA node is a special case node that is considered invalid. This means %0 will always be invalid for "null" node checks
|
||||
The first real SSA node also has to be a IRHeader node. This means it is safe to assume that %1 will always be an IRHeader.
|
||||
|
||||
|
||||
```(%%ssa1) IRHeader 0x41a9a0, %%ssa2, 5```
|
||||
```(%%1) IRHeader 0x41a9a0, %%2, 5```
|
||||
|
||||
The header provides information about that function like the entry point address.
|
||||
Additionally it also points to the first `CodeBlock` IROp
|
||||
|
||||
|
||||
```(%%ssa2) CodeBlock %%ssa7, %%ssa168, %%ssa3```
|
||||
```(%%2) CodeBlock %%7, %%168, %%3```
|
||||
|
||||
|
||||
* The `CodeBlock` Op is a jump target and must be treated as if it'll be jumped to from other blocks
|
||||
@@ -54,12 +54,12 @@ Additionally it also points to the first `CodeBlock` IROp
|
||||
### Example code block
|
||||
|
||||
```
|
||||
(%%ssa3) CodeBlock %%ssa169, %%ssa173, %%ssa4
|
||||
(%%ssa169) BeginBlock %ssa3
|
||||
%ssa170 i64 = Constant 0x41a9e1
|
||||
(%%ssa171) StoreContext %ssa170 i64, 0x8, 0x0
|
||||
(%%ssa172) ExitFunction
|
||||
(%%ssa173) EndBlock %ssa3
|
||||
(%%3) CodeBlock %%169, %%173, %%4
|
||||
(%%169) BeginBlock %3
|
||||
%170 i64 = Constant 0x41a9e1
|
||||
(%%171) StoreContext %170 i64, 0x8, 0x0
|
||||
(%%172) ExitFunction
|
||||
(%%173) EndBlock %3
|
||||
```
|
||||
|
||||
* BeginBlock points back to the CodeBlock SSA which helps with iterating across multiple blocks
|
||||
|
||||
+17
-17
@@ -17,23 +17,23 @@ Ex:
|
||||
Translates to the IR of:
|
||||
```
|
||||
BeginBlock
|
||||
%ssa8 i32 = Constant 0x1
|
||||
StoreContext 0x8, 0x8, %ssa8
|
||||
%ssa64 i32 = Constant 0x1
|
||||
StoreContext 0x8, 0x30, %ssa64
|
||||
%ssa120 i32 = Constant 0x1f
|
||||
StoreContext 0x8, 0x28, %ssa120
|
||||
%ssa176 i32 = Constant 0x1
|
||||
StoreContext 0x8, 0x20, %ssa176
|
||||
%ssa232 i64 = LoadContext 0x8, 0x8
|
||||
%ssa264 i64 = LoadContext 0x8, 0x30
|
||||
%ssa296 i64 = LoadContext 0x8, 0x28
|
||||
%ssa328 i64 = LoadContext 0x8, 0x20
|
||||
%ssa360 i64 = LoadContext 0x8, 0x58
|
||||
%ssa392 i64 = LoadContext 0x8, 0x48
|
||||
%ssa424 i64 = LoadContext 0x8, 0x50
|
||||
%ssa456 i64 = Syscall%ssa232, %ssa264, %ssa296, %ssa328, %ssa360, %ssa392, %ssa424
|
||||
StoreContext 0x8, 0x8, %ssa456
|
||||
%8 i32 = Constant 0x1
|
||||
StoreContext 0x8, 0x8, %8
|
||||
%64 i32 = Constant 0x1
|
||||
StoreContext 0x8, 0x30, %64
|
||||
%120 i32 = Constant 0x1f
|
||||
StoreContext 0x8, 0x28, %120
|
||||
%176 i32 = Constant 0x1
|
||||
StoreContext 0x8, 0x20, %176
|
||||
%232 i64 = LoadContext 0x8, 0x8
|
||||
%264 i64 = LoadContext 0x8, 0x30
|
||||
%296 i64 = LoadContext 0x8, 0x28
|
||||
%328 i64 = LoadContext 0x8, 0x20
|
||||
%360 i64 = LoadContext 0x8, 0x58
|
||||
%392 i64 = LoadContext 0x8, 0x48
|
||||
%424 i64 = LoadContext 0x8, 0x50
|
||||
%456 i64 = Syscall%232, %264, %296, %328, %360, %392, %424
|
||||
StoreContext 0x8, 0x8, %456
|
||||
BeginBlock
|
||||
EndBlock 0x1e
|
||||
ExitFunction
|
||||
|
||||
+1
-1
@@ -30,7 +30,7 @@ Large amount of x86-64 instructions load or store registers in order from the co
|
||||
We can merge these in to loadstore pair ops to improve perf
|
||||
### Function level heuristic pass
|
||||
Once we know that a function is a true full recompile we can do some additional optimizations.
|
||||
Remove any final flag stores. We know that a compiler won't pass flags past a function call boundry(It doesn't exist in the ABI)
|
||||
Remove any final flag stores. We know that a compiler won't pass flags past a function call boundary(It doesn't exist in the ABI)
|
||||
Remove any loadstores to the context mid function, only do a final store at the end of the function and do loads at the start. Which means ops just map registers directly throughout the entire function.
|
||||
### SIMD coalescing pass?
|
||||
When operating on older MMX ops(64bit SIMD) and they may end up up generating some independent ops that can be coalesced in to a 128bit op
|
||||
+43
-46
@@ -2,12 +2,16 @@
|
||||
|
||||
#include <FEXCore/Core/Context.h>
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/EnumOperators.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/fextl/fmt.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/list.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
|
||||
#include <charconv>
|
||||
#include <optional>
|
||||
#include <stdint.h>
|
||||
|
||||
@@ -52,6 +56,9 @@ namespace Handler {
|
||||
#include <FEXCore/Config/ConfigValues.inl>
|
||||
};
|
||||
|
||||
#define ENUMDEFINES
|
||||
#include <FEXCore/Config/ConfigOptions.inl>
|
||||
|
||||
enum ConfigCore {
|
||||
CONFIG_INTERPRETER,
|
||||
CONFIG_IRJIT,
|
||||
@@ -83,6 +90,42 @@ namespace Handler {
|
||||
LAYER_TOP,
|
||||
};
|
||||
|
||||
template<typename PairTypes>
|
||||
static inline fextl::string EnumParser(PairTypes const &EnumPairs, std::string_view const View) {
|
||||
uint64_t EnumMask{};
|
||||
auto Results = std::from_chars(View.data(), View.data() + View.size(), EnumMask);
|
||||
if (Results.ec == std::errc()) {
|
||||
// If the data is a valid number, just pass it through.
|
||||
return View.data();
|
||||
}
|
||||
|
||||
auto Begin = 0;
|
||||
auto End = View.find_first_of(',');
|
||||
std::string_view Option = View.substr(Begin, End);
|
||||
while (Option.size() != 0) {
|
||||
auto EnumValue = std::find_if(EnumPairs.begin(), EnumPairs.end(),
|
||||
[Option](const DisassembleConfigPair &Value) -> bool {
|
||||
return Value.first == Option;
|
||||
});
|
||||
|
||||
if (EnumValue == EnumPairs.end()) {
|
||||
LogMan::Msg::IFmt("Skipping Unknown option: {}", Option);
|
||||
}
|
||||
else {
|
||||
EnumMask |= FEXCore::ToUnderlying(EnumValue->second);
|
||||
}
|
||||
|
||||
if (End == std::string::npos) {
|
||||
break;
|
||||
}
|
||||
Begin = End + 1;
|
||||
End = View.find_first_of(',', Begin);
|
||||
Option = View.substr(Begin, End);
|
||||
}
|
||||
|
||||
return fextl::fmt::format("{}", EnumMask);
|
||||
}
|
||||
|
||||
namespace DefaultValues {
|
||||
#define P(x) x
|
||||
#define OPT_BASE(type, group, enum, json, default) extern const P(type) P(enum);
|
||||
@@ -244,50 +287,4 @@ namespace Type {
|
||||
|
||||
static void GetListIfExists(FEXCore::Config::ConfigOption Option, fextl::list<fextl::string> *List);
|
||||
};
|
||||
|
||||
// Application loaders
|
||||
class FEX_DEFAULT_VISIBILITY OptionMapper : public FEXCore::Config::Layer {
|
||||
public:
|
||||
explicit OptionMapper(FEXCore::Config::LayerType Layer);
|
||||
|
||||
protected:
|
||||
void MapNameToOption(const char *ConfigName, const char *ConfigString);
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Loads the global FEX config
|
||||
*
|
||||
* @return unique_ptr for that layer
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY fextl::unique_ptr<FEXCore::Config::Layer> CreateGlobalMainLayer();
|
||||
|
||||
/**
|
||||
* @brief Loads the main application config
|
||||
*
|
||||
* @param File Optional override to load a specific config file in to the main layer
|
||||
* Shouldn't be commonly used
|
||||
*
|
||||
* @return unique_ptr for that layer
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY fextl::unique_ptr<FEXCore::Config::Layer> CreateMainLayer(fextl::string const *File = nullptr);
|
||||
|
||||
/**
|
||||
* @brief Create an application configuration loader
|
||||
*
|
||||
* @param Filename Application filename component
|
||||
* @param Global Load the global configuration or user accessible file
|
||||
*
|
||||
* @return unique_ptr for that layer
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY fextl::unique_ptr<FEXCore::Config::Layer> CreateAppLayer(const fextl::string& Filename, FEXCore::Config::LayerType Type);
|
||||
|
||||
/**
|
||||
* @brief iCreate an environment configuration loader
|
||||
*
|
||||
* @param _envp[] The environment array from main
|
||||
*
|
||||
* @return unique_ptr for that layer
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY fextl::unique_ptr<FEXCore::Config::Layer> CreateEnvironmentLayer(char *const _envp[]);
|
||||
|
||||
}
|
||||
@@ -35,6 +35,8 @@ namespace CodeSerialize {
|
||||
namespace CPU {
|
||||
struct CPUBackendFeatures {
|
||||
bool SupportsStaticRegisterAllocation = false;
|
||||
bool SupportsShiftedBitwise = false;
|
||||
bool SupportsFlags = false;
|
||||
};
|
||||
|
||||
class CPUBackend {
|
||||
@@ -89,6 +91,28 @@ namespace CPU {
|
||||
size_t Size;
|
||||
// RIP that the block's entry comes from.
|
||||
uint64_t RIP;
|
||||
|
||||
// Number of RIP entries for this JIT Code section.
|
||||
uint32_t NumberOfRIPEntries;
|
||||
|
||||
// Offset after this block to the start of the RIP entries.
|
||||
uint32_t OffsetToRIPEntries;
|
||||
};
|
||||
|
||||
// Entries that live after the JITCodeTail.
|
||||
// These entries correlate JIT code regions with guest RIP regions.
|
||||
// Using these entries FEX is able to reconstruct the guest RIP accurately when an instruction cause a signal fault.
|
||||
// Packed using 16-bit entries to ensure the size isn't too large.
|
||||
// These smaller sizes means that each entry is relative to each other instead of absolute offset from the start of the JIT block.
|
||||
// When reconstructing the RIP, each entry must be walked linearly and accumulated with the previous entries.
|
||||
// This is a trade-off between compression inside the JIT code space and execution time when reconstruction the RIP.
|
||||
// RIP reconstruction when faulting is less likely so we are requiring the accumulation.
|
||||
struct JITRIPReconstructEntries {
|
||||
// The Host PC offset from the previous entry.
|
||||
uint16_t HostPCOffset;
|
||||
|
||||
// How much to offset the RIP from the previous entry.
|
||||
uint16_t GuestRIPOffset;
|
||||
};
|
||||
|
||||
/**
|
||||
|
||||
@@ -5,5 +5,9 @@ namespace FEXCore::CPUID {
|
||||
struct FunctionResults {
|
||||
uint32_t eax, ebx, ecx, edx;
|
||||
};
|
||||
|
||||
struct XCRResults {
|
||||
uint32_t eax, edx;
|
||||
};
|
||||
}
|
||||
|
||||
+38
-21
@@ -86,9 +86,17 @@ namespace FEXCore::Context {
|
||||
void *VDSO_kernel_rt_sigreturn;
|
||||
};
|
||||
|
||||
using CustomCPUFactoryType = std::function<fextl::unique_ptr<FEXCore::CPU::CPUBackend> (FEXCore::Context::Context*, FEXCore::Core::InternalThreadState *Thread)>;
|
||||
using CodeRangeInvalidationFn = std::function<void(uint64_t start, uint64_t Length)>;
|
||||
|
||||
using ExitHandler = std::function<void(uint64_t ThreadId, FEXCore::Context::ExitReason)>;
|
||||
using CustomCPUFactoryType = std::function<fextl::unique_ptr<CPU::CPUBackend>(Context*, Core::InternalThreadState *Thread)>;
|
||||
using CustomIREntrypointHandler = std::function<void(uintptr_t Entrypoint, IR::IREmitter *)>;
|
||||
|
||||
using ExitHandler = std::function<void(uint64_t ThreadId, ExitReason)>;
|
||||
|
||||
using AOTIRCodeFileWriterFn = std::function<void(const fextl::string& fileid, const fextl::string& filename)>;
|
||||
using AOTIRLoaderCBFn = std::function<int(const fextl::string&)>;
|
||||
using AOTIRRenamerCBFn = std::function<void(const fextl::string&)>;
|
||||
using AOTIRWriterCBFn = std::function<fextl::unique_ptr<AOTIRWriter>(const fextl::string&)>;
|
||||
|
||||
class Context {
|
||||
public:
|
||||
@@ -225,17 +233,6 @@ namespace FEXCore::Context {
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetCustomCPUBackendFactory(CustomCPUFactoryType Factory) = 0;
|
||||
|
||||
/**
|
||||
* @brief Sets up memory regions on the guest for mirroring within the guest's VM space
|
||||
*
|
||||
* @param VirtualAddress The address we want to set to mirror a physical memory region
|
||||
* @param PhysicalAddress The physical memory region we are mapping
|
||||
* @param Size Size of the region to mirror
|
||||
*
|
||||
* @return true when successfully mapped. false if there was an error adding
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual bool AddVirtualMemoryMapping(uint64_t VirtualAddress, uint64_t PhysicalAddress, uint64_t Size) = 0;
|
||||
|
||||
/**
|
||||
* @brief Retrieves a feature struct indicating certain supported aspects from
|
||||
* the hose.
|
||||
@@ -254,28 +251,32 @@ namespace FEXCore::Context {
|
||||
FEX_DEFAULT_VISIBILITY virtual void RunThread(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void StopThread(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void DestroyThread(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void CleanupAfterFork(FEXCore::Core::InternalThreadState *Thread) = 0;
|
||||
#ifndef _WIN32
|
||||
FEX_DEFAULT_VISIBILITY virtual void LockBeforeFork(FEXCore::Core::InternalThreadState *Thread) {}
|
||||
FEX_DEFAULT_VISIBILITY virtual void UnlockAfterFork(FEXCore::Core::InternalThreadState *Thread, bool Child) {}
|
||||
#endif
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetSignalDelegator(FEXCore::SignalDelegator *SignalDelegation) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetSyscallHandler(FEXCore::HLE::SyscallHandler *Handler) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::CPUID::FunctionResults RunCPUIDFunction(uint32_t Function, uint32_t Leaf) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::CPUID::XCRResults RunXCRFunction(uint32_t Function) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::CPUID::FunctionResults RunCPUIDFunctionName(uint32_t Function, uint32_t Leaf, uint32_t CPU) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::IR::AOTIRCacheEntry *LoadAOTIRCacheEntry(const fextl::string& Name) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void UnloadAOTIRCacheEntry(FEXCore::IR::AOTIRCacheEntry *Entry) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetAOTIRLoader(std::function<int(const fextl::string&)> CacheReader) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetAOTIRWriter(std::function<fextl::unique_ptr<AOTIRWriter>(const fextl::string&)> CacheWriter) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetAOTIRRenamer(std::function<void(const fextl::string&)> CacheRenamer) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetAOTIRLoader(AOTIRLoaderCBFn CacheReader) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetAOTIRWriter(AOTIRWriterCBFn CacheWriter) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetAOTIRRenamer(AOTIRRenamerCBFn CacheRenamer) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void FinalizeAOTIRCache() = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void WriteFilesWithCode(std::function<void(const fextl::string& fileid, const fextl::string& filename)> Writer) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void InvalidateGuestCodeRange(uint64_t Start, uint64_t Length) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void InvalidateGuestCodeRange(uint64_t Start, uint64_t Length, std::function<void(uint64_t start, uint64_t Length)> callback) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void WriteFilesWithCode(AOTIRCodeFileWriterFn Writer) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, CodeRangeInvalidationFn callback) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void MarkMemoryShared() = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void ConfigureAOTGen(FEXCore::Core::InternalThreadState *Thread, fextl::set<uint64_t> *ExternalBranches, uint64_t SectionMaxAddress) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)> Handler, void *Creator = nullptr, void *Data = nullptr) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual CustomIRResult AddCustomIREntrypoint(uintptr_t Entrypoint, CustomIREntrypointHandler Handler, void *Creator = nullptr, void *Data = nullptr) = 0;
|
||||
|
||||
/**
|
||||
* @brief Allows the frontend to register its own thunk handlers independent of what is controlled in the backend.
|
||||
@@ -288,6 +289,22 @@ namespace FEXCore::Context {
|
||||
FEX_DEFAULT_VISIBILITY virtual void GetVDSOSigReturn(VDSOSigReturn *VDSOPointers) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void IncrementIdleRefCount() = 0;
|
||||
|
||||
/**
|
||||
* @brief Informs the context if hardware TSO is supported.
|
||||
* Once hardware TSO is enabled, then TSO emulation through atomics is disabled and relies on the hardware.
|
||||
*
|
||||
* @param HardwareTSOSupported If the hardware supports the TSO memory model or not.
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual void SetHardwareTSOSupport(bool HardwareTSOSupported) = 0;
|
||||
|
||||
/**
|
||||
* @brief Enable exiting the JIT when HLT is hit.
|
||||
*
|
||||
* This is to workaround a bug in Wine's longjump function which breaks our unittests.
|
||||
*
|
||||
*/
|
||||
FEX_DEFAULT_VISIBILITY virtual void EnableExitOnHLT() = 0;
|
||||
private:
|
||||
};
|
||||
|
||||
|
||||
@@ -6,10 +6,69 @@
|
||||
|
||||
#include <atomic>
|
||||
#include <cstddef>
|
||||
#include <cstring>
|
||||
#include <stdint.h>
|
||||
#include <string_view>
|
||||
#include <type_traits>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
// Wrapper around std::atomic using std::memory_order_relaxed.
|
||||
// This allows compilers to emit more performant code at the expense of visibly tearing.
|
||||
// In particular, increments/decrements may visibly tear if a signal is received half-way through.
|
||||
//
|
||||
// Prefer std::atomic with default memory ordering unless you really know what you're doing.
|
||||
// Primarily this ensure program ordering when signals are concerned.
|
||||
template<typename T>
|
||||
class NonAtomicRefCounter {
|
||||
public:
|
||||
void Increment(T Value) {
|
||||
// Specifically avoiding fetch_add here because that will turn in to ldxr+stxr or lock xadd.
|
||||
// FEX very specifically wants to use simple loadstore instructions for this
|
||||
//
|
||||
// ARM64 ex:
|
||||
// ldr x0, [x1];
|
||||
// add x0, x0, #1;
|
||||
// str x0, [x1];
|
||||
//
|
||||
// x86-64 ex:
|
||||
// inc qword [rax];
|
||||
auto Current = AtomicVariable.load(std::memory_order_relaxed);
|
||||
AtomicVariable.store(Current + Value, std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
// Returns original value.
|
||||
// x86-64 needs to know the result on decrement.
|
||||
T Decrement(T Value) {
|
||||
// Specifically avoiding fetch_sub here because that will turn into ldxr+stxr or lock xadd.
|
||||
// FEX very specifically wants to use simple loadstore instructions for this
|
||||
//
|
||||
// ARM64 ex:
|
||||
// ldr x0, [x1];
|
||||
// sub x0, x0, #1;
|
||||
// str x0, [x1];
|
||||
//
|
||||
// x86-64 ex:
|
||||
// dec qword [rax];
|
||||
auto Current = AtomicVariable.load(std::memory_order_relaxed);
|
||||
AtomicVariable.store(Current - Value, std::memory_order_relaxed);
|
||||
return Current;
|
||||
}
|
||||
|
||||
T Load() const {
|
||||
return AtomicVariable.load(std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
void Store(T Value) {
|
||||
AtomicVariable.store(Value, std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
private:
|
||||
std::atomic<T> AtomicVariable;
|
||||
};
|
||||
static_assert(std::is_standard_layout_v<NonAtomicRefCounter<uint64_t>>, "Needs to be standard layout");
|
||||
static_assert(std::is_trivially_copyable_v<NonAtomicRefCounter<uint64_t>>, "needs to be trivially copyable");
|
||||
static_assert(sizeof(NonAtomicRefCounter<uint64_t>) == sizeof(uint64_t), "Needs to be correct size");
|
||||
|
||||
struct FEX_PACKED CPUState {
|
||||
// Allows more efficient handling of the register
|
||||
// file in the event AVX is not supported.
|
||||
@@ -49,6 +108,13 @@ namespace FEXCore::Core {
|
||||
uint16_t FCW;
|
||||
uint16_t FTW;
|
||||
|
||||
uint32_t _pad2[1];
|
||||
// Reference counter for FEX's per-thread deferred signals.
|
||||
// Counts the nesting depth of program sections that cause signals to be deferred.
|
||||
NonAtomicRefCounter<uint64_t> DeferredSignalRefCount;
|
||||
// Since this memory region is thread local, we use NonAtomicRefCounter for fast atomic access.
|
||||
NonAtomicRefCounter<uint64_t> *DeferredSignalFaultAddress;
|
||||
|
||||
static constexpr size_t FLAG_SIZE = sizeof(flags[0]);
|
||||
static constexpr size_t GDT_SIZE = sizeof(gdt[0]);
|
||||
static constexpr size_t GPR_REG_SIZE = sizeof(gregs[0]);
|
||||
@@ -63,9 +129,29 @@ namespace FEXCore::Core {
|
||||
static constexpr size_t NUM_GPRS = sizeof(gregs) / GPR_REG_SIZE;
|
||||
static constexpr size_t NUM_XMMS = sizeof(xmm) / XMM_AVX_REG_SIZE;
|
||||
static constexpr size_t NUM_MMS = sizeof(mm) / MM_REG_SIZE;
|
||||
CPUState() {
|
||||
// Initialize default CPU state
|
||||
rip = ~0ULL;
|
||||
memset(gregs, 0, sizeof(gregs));
|
||||
|
||||
for (auto& xmm : xmm.avx.data) {
|
||||
xmm[0] = 0xDEADBEEFULL;
|
||||
xmm[1] = 0xBAD0DAD1ULL;
|
||||
xmm[2] = 0xDEADCAFEULL;
|
||||
xmm[3] = 0xBAD2CAD3ULL;
|
||||
}
|
||||
memset(&flags, 0, Core::CPUState::NUM_EFLAG_BITS);
|
||||
flags[1] = 1; ///< Reserved - Always 1.
|
||||
flags[9] = 1; ///< Interrupt flag - Always 1.
|
||||
FCW = 0x37F;
|
||||
FTW = 0xFFFF;
|
||||
}
|
||||
};
|
||||
static_assert(std::is_trivially_copyable_v<CPUState>, "Needs to be trivial");
|
||||
static_assert(std::is_standard_layout_v<CPUState>, "This needs to be standard layout");
|
||||
static_assert(offsetof(CPUState, xmm) % 32 == 0, "xmm needs to be 256-bit aligned!");
|
||||
static_assert(offsetof(CPUState, mm) % 16 == 0, "mm needs to be 128-bit aligned!");
|
||||
static_assert(offsetof(CPUState, DeferredSignalRefCount) % 8 == 0, "Needs to be 8-byte aligned");
|
||||
|
||||
struct InternalThreadState;
|
||||
|
||||
@@ -144,6 +230,7 @@ namespace FEXCore::Core {
|
||||
uint64_t ThreadRemoveCodeEntryFromJIT{};
|
||||
uint64_t CPUIDObj{};
|
||||
uint64_t CPUIDFunction{};
|
||||
uint64_t XCRFunction{};
|
||||
uint64_t SyscallHandlerObj{};
|
||||
uint64_t SyscallHandlerFunc{};
|
||||
uint64_t ExitFunctionLink{};
|
||||
|
||||
@@ -29,6 +29,7 @@ class HostFeatures final {
|
||||
bool SupportsBMI2{};
|
||||
bool SupportsCLWB{};
|
||||
bool SupportsPMULL_128Bit{};
|
||||
bool SupportsCSSC{};
|
||||
|
||||
// Float exception behaviour
|
||||
bool SupportsFlushInputsToZero{};
|
||||
|
||||
Loaded 100 of 252 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user